From c3f5e47559bd19010a6925d0e8336d5213f1a7c0 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 17:42:49 +0200 Subject: [PATCH 01/67] add ansible.cfg --- ansible/ansible.cfg | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 ansible/ansible.cfg diff --git a/ansible/ansible.cfg b/ansible/ansible.cfg new file mode 100644 index 0000000..56620fa --- /dev/null +++ b/ansible/ansible.cfg @@ -0,0 +1,14 @@ +[defaults] +inventory = inventory.ini +roles_path = roles +collections_path = collections +interpreter_python = auto_silent +stdout_callback = yaml +retry_files_enabled = False +host_key_checking = True +forks = 10 +# vault_password_file = .vault_pass # uncomment in Stage 2 + +[ssh_connection] +pipelining = True +ssh_args = -o ControlMaster=auto -o ControlPersist=300s From 6029317f3bfe3475bb4496bfd399a2833d4531a6 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 17:53:12 +0200 Subject: [PATCH 02/67] ansible: vault encrypt secrets --- ansible/infra/nodito/nodito_secrets.yml | 17 +++++++++ ansible/infra_secrets.yml | 50 +++++++++++++++++++++++++ 2 files changed, 67 insertions(+) create mode 100644 ansible/infra/nodito/nodito_secrets.yml create mode 100644 ansible/infra_secrets.yml diff --git a/ansible/infra/nodito/nodito_secrets.yml b/ansible/infra/nodito/nodito_secrets.yml new file mode 100644 index 0000000..7983cdb --- /dev/null +++ b/ansible/infra/nodito/nodito_secrets.yml @@ -0,0 +1,17 @@ +$ANSIBLE_VAULT;1.1;AES256 +33303038356165616435663034356134363963323865356133353239373263623430373263613430 +6463323436386666633666333331636132653963663735330a333063303038313839653330303231 +37313332653033636265353730616431356337366566656533363838303939636266626565333365 +3764316366653361310a386330326335343936356430333864336431376230323734303030613032 +31343737323636363366623530353433313862663038376439343331633038363138316639613531 +63663866613037336630316464383464613165326632376337383338336432326564396261383561 +61313337336466653431613338656136653862646662323435396462623366366438626566373331 +34663433333965626161356636633930313961313531656233303638346265316232326530613062 +31343732636466393334623734346639636637363831393337363562343461616639316235336362 +62306464383263626235613533336130326232613563336630373661623637303961353863303633 +66306664653035363236323063616363313739633633373231333833393562633666356530383065 +36373830393965663861663231633838303863363763613663363734383961623466653735636138 +34376233363831363732643631646364363066626230353330316562376337633462323966376239 +30666263326263366563323739653538616462303566633037393861613664646331633264613338 +35636131383937393430666430646530663866626438643430316430376133653135333838636436 +37613465666265373336 diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml new file mode 100644 index 0000000..c48d456 --- /dev/null +++ b/ansible/infra_secrets.yml @@ -0,0 +1,50 @@ +$ANSIBLE_VAULT;1.1;AES256 +32643038333165373363353439313639326538343435393633343765386662643231333936386630 +6564313061366166353662373564393261396666663432610a633161643066663232336163306663 +61623535643231366638616630666562333936383232373130623631306334653634386132396431 +3036383735316339650a313565313062636461396563336461633838343037666237623038633133 +35333531313036666531633230646464636631666133313636353362646336623135363736316562 +36656564396230333231643266303833633134303862323633636136363136373836346363663364 +32306234393863613534663031646632643734366531343339386133363837396138616435313365 +65353037326266386538633665386638656539376133636435346136643163643238656434346132 +34363036623935343162313031343036393536613136306435323730346561393261333966663465 +35333864323638356565316632353662366262363962376134366432333361343639633133626533 +65303964643330393264643136346432313164396664333535363135646464326539626433623732 +34666530666234376637623930633433656163353136646134326365353563343833666161666435 +30313866396136623036613933376164346533383964303664633830376636363565323832343136 +62653137366632313133626234386235356630383433623934303737653438623136366638396237 +66303435356539343061313337646261343738373866646465373339616161323531346361383363 +64643363616130323935656264326631336231323236616438366632356333333534303861663238 +61376464623030363938653230316265366533623364653930343537326532393433363230356136 +62643135636130626436386239366561393537333736383361336162636434656339653630353732 +37343265366234656637336135633461333735363932636130323564623661303730303363383833 +34373963373765643437303534366233626662656531373261633565353137643761356135376430 +30373639336565356364313333333237323338346631666235613961656336643438653834313137 +35303666366433646334643264633837343134373637643535633864313062613336333538303931 +34396266326166666165333639363961346639343130656235323332346631643539363861666436 +63343733656362313165623130383838376138613534363535336662366564306132353362666463 +61613534613431623263323831646635323362343634616566393834353964643861633639346132 +66363233373964323139313966616365343535306365373532313063383236353964343265383761 +64663833363035363736643236373634663466666138346266643930643033343839313534376666 +37363461623965636164663339386237643635353237353263633131363434643066393538363639 +39346530366361656536383233663766616234383535643234616233666262353364633036663062 +64386663383363633365656631326332353261303562346666636565313564306635333562363930 +33306237333638323131613161336165636237623332643137363038303330313663346639326361 +39383565366431363361313361303436343761626462306537633337393636333135343833623464 +32323663616166653063303861396664613434643536633066626466303634663031623938313436 +32303763616562376137303061633864393863373334373338626566653633383132306635396566 +64616463663265643732366465666662663334666362383036326330356535333465326534363266 +33326265326365353231613839383336656562333837376565306230646466333238323936316463 +64396266306562303438363635343865376364333239376664346534636664326532353265313563 +66636331333430616462326334623637646539643638323036303335656632363565373739373234 +35666532346364376335346238376562336164386363623862356539316466336531643936336131 +31356332306639626532646338393638373764653961616638643066313832626561336435383839 +64363232313633643133383066636539663933323064366536666462646466323166306562623538 +64393835383431663561343231303961383337656331663261393439336136623066666434336362 +36313239336537376134396361643532633437633163373931353139303534356665383538393838 +31306336326538353232336533643961383337313161663266353730643931656165353264303938 +30636363666136653963306130666433633562313464653862656262313035306635306230653666 +34336631623235323734616163623039613636393863383966336435313230313231356563636337 +61373336336563653030366130356331616438336638323263663164366435653266643161653331 +31323733313936313833386665633235633339376330373161326361336266633838326336356363 +37383165393866656565316337386639346531323439626465613834303636663933 From 9db0b0ae847dd9f3d5057513612e0544a18104ec Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 17:59:41 +0200 Subject: [PATCH 03/67] now using password --- .gitignore | 11 ++++++----- ansible/ansible.cfg | 2 +- 2 files changed, 7 insertions(+), 6 deletions(-) diff --git a/.gitignore b/.gitignore index 6c0a084..750a233 100644 --- a/.gitignore +++ b/.gitignore @@ -14,8 +14,9 @@ inventory.ini venv/* .env -# Secrets and sensitive files -*_secrets.yml -*_secrets.yaml -secrets/ -.secrets/ +# Secrets are ansible-vault encrypted and ARE committed. +# Anything matching *_secrets.plain.yml is a working decryption — never commit those. +*_secrets.plain.yml + +# Vault password — never commit +ansible/.vault_pass \ No newline at end of file diff --git a/ansible/ansible.cfg b/ansible/ansible.cfg index 56620fa..aa10181 100644 --- a/ansible/ansible.cfg +++ b/ansible/ansible.cfg @@ -7,7 +7,7 @@ stdout_callback = yaml retry_files_enabled = False host_key_checking = True forks = 10 -# vault_password_file = .vault_pass # uncomment in Stage 2 +vault_password_file = .vault_pass [ssh_connection] pipelining = True From 5e06021938f3be8e576d963c45db031a7af4caad Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 20:08:19 +0200 Subject: [PATCH 04/67] secrets note in readme --- README.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/README.md b/README.md index f343cbc..80dbaf6 100644 --- a/README.md +++ b/README.md @@ -6,6 +6,12 @@ My repo documenting my personal infra, along with artifacts, scripts, etc. Go through the different numbered markdowns in the repo root to do the different parts. +## How to edit secrets + +`ansible-vault edit ansible/your_file_with_secrets.yml` + +Assumes that you've set `ansible/.vault_pass` with `chmod 600`. + ## Overview ### Services From 123107b7fbaf9798e39483404582ac027e1ef5d8 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:32:23 +0200 Subject: [PATCH 05/67] track inventory --- .gitignore | 2 -- ansible/inventory.ini | 25 +++++++++++++++++++++++++ 2 files changed, 25 insertions(+), 2 deletions(-) create mode 100644 ansible/inventory.ini diff --git a/.gitignore b/.gitignore index 750a233..9ca7b8e 100644 --- a/.gitignore +++ b/.gitignore @@ -9,8 +9,6 @@ crash.log *.tfvars *.tfvars.json -test-inventory.ini -inventory.ini venv/* .env diff --git a/ansible/inventory.ini b/ansible/inventory.ini new file mode 100644 index 0000000..3d6c209 --- /dev/null +++ b/ansible/inventory.ini @@ -0,0 +1,25 @@ +[vps] +vipy ansible_host=167.172.107.33 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +watchtower ansible_host=164.92.239.72 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +spacey ansible_host=64.227.112.128 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +[nodito_host] +nodito ansible_host=192.168.1.139 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua + + +[nodito_vms] +knots_box_local ansible_host=192.168.1.135 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +fulcrum_box_local ansible_host=192.168.1.142 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +mempool_box_local ansible_host=192.168.1.140 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +memos_box_local ansible_host=192.168.1.130 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +forgejo_runner_local ansible_host=192.168.1.147 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +arbret_staging_local ansible_host=192.168.1.142 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +small_backups_local ansible_host=192.168.1.148 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +nonkeiwaisi_local ansible_host=192.168.1.151 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua + +# Local connection to laptop: this assumes you're running ansible commands from your personal laptop +[lapy] +localhost ansible_connection=local ansible_user=counterweight gpg_recipient=counterweightoperator@protonmail.com gpg_key_id=883EDBAA726BD96C + +[arbret] +prd-arbret ansible_host=167.99.242.62 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua + From acb8f9d82ff2c87f1770d25800d2efe954fbbe1c Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:32:56 +0200 Subject: [PATCH 06/67] remove old example password --- ansible/inventory.ini.example | 16 ---------------- 1 file changed, 16 deletions(-) delete mode 100644 ansible/inventory.ini.example diff --git a/ansible/inventory.ini.example b/ansible/inventory.ini.example deleted file mode 100644 index bde96dd..0000000 --- a/ansible/inventory.ini.example +++ /dev/null @@ -1,16 +0,0 @@ -[vps] -vipy ansible_host=your.services.vps.ip ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/your-key -watchtower ansible_host=your.monitoring.vps.ip ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/your-key -spacey ansible_host=your.headscale.vps.ip ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/your-key - -[nodito_host] -nodito ansible_host=your.proxmox.ip.here ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/your-key ansible_ssh_pass=your_root_password - -[nodito_vms] -# Example node, replace with your VM names and addresses -# memos_box ansible_host=192.168.1.150 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/your-key - -# Local connection to laptop: this assumes you're running ansible commands from your personal laptop -# Make sure to adjust the username -[lapy] -localhost ansible_connection=local ansible_user=your laptop user gpg_recipient=your_email@example.com gpg_key_id=your_gpg_key_id_here \ No newline at end of file From 28a0bb280670fed018c88ea3f598bee808544dac Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:51:50 +0200 Subject: [PATCH 07/67] new groups, stop using all --- .../01_user_and_access_setup_playbook.yml | 2 +- .../02_firewall_and_fail2ban_playbook.yml | 2 +- ansible/infra/410_disk_usage_alerts.yml | 2 +- ansible/infra/420_system_healthcheck.yml | 2 +- ansible/infra/900_install_rsync.yml | 2 +- ansible/infra/910_docker_playbook.yml | 2 +- ansible/infra/920_join_headscale_mesh.yml | 2 +- ansible/inventory.ini | 36 +++++++++++++++++++ 8 files changed, 43 insertions(+), 7 deletions(-) diff --git a/ansible/infra/01_user_and_access_setup_playbook.yml b/ansible/infra/01_user_and_access_setup_playbook.yml index 13e5149..c6eed18 100644 --- a/ansible/infra/01_user_and_access_setup_playbook.yml +++ b/ansible/infra/01_user_and_access_setup_playbook.yml @@ -1,5 +1,5 @@ - name: Secure Debian - hosts: all + hosts: managed vars_files: - ../infra_vars.yml become: true diff --git a/ansible/infra/02_firewall_and_fail2ban_playbook.yml b/ansible/infra/02_firewall_and_fail2ban_playbook.yml index e83cbcb..9f37c70 100644 --- a/ansible/infra/02_firewall_and_fail2ban_playbook.yml +++ b/ansible/infra/02_firewall_and_fail2ban_playbook.yml @@ -1,5 +1,5 @@ - name: Secure Debian - hosts: all + hosts: managed vars_files: - ../infra_vars.yml become: true diff --git a/ansible/infra/410_disk_usage_alerts.yml b/ansible/infra/410_disk_usage_alerts.yml index de02f53..7905cf3 100644 --- a/ansible/infra/410_disk_usage_alerts.yml +++ b/ansible/infra/410_disk_usage_alerts.yml @@ -1,5 +1,5 @@ - name: Deploy Disk Usage Monitoring - hosts: all + hosts: managed become: yes vars_files: - ../infra_vars.yml diff --git a/ansible/infra/420_system_healthcheck.yml b/ansible/infra/420_system_healthcheck.yml index 2580ff0..5a45ce1 100644 --- a/ansible/infra/420_system_healthcheck.yml +++ b/ansible/infra/420_system_healthcheck.yml @@ -1,5 +1,5 @@ - name: Deploy System Healthcheck Monitoring - hosts: all + hosts: managed become: yes vars_files: - ../infra_vars.yml diff --git a/ansible/infra/900_install_rsync.yml b/ansible/infra/900_install_rsync.yml index c0b7318..6c6b10e 100644 --- a/ansible/infra/900_install_rsync.yml +++ b/ansible/infra/900_install_rsync.yml @@ -1,5 +1,5 @@ - name: Install rsync - hosts: all + hosts: managed vars_files: - ../infra_vars.yml become: true diff --git a/ansible/infra/910_docker_playbook.yml b/ansible/infra/910_docker_playbook.yml index f137b6a..62b7147 100644 --- a/ansible/infra/910_docker_playbook.yml +++ b/ansible/infra/910_docker_playbook.yml @@ -1,5 +1,5 @@ - name: Install Docker and Docker Compose on Debian 12 - hosts: all + hosts: managed become: yes tasks: diff --git a/ansible/infra/920_join_headscale_mesh.yml b/ansible/infra/920_join_headscale_mesh.yml index 8d06d44..5b5e4ce 100644 --- a/ansible/infra/920_join_headscale_mesh.yml +++ b/ansible/infra/920_join_headscale_mesh.yml @@ -1,5 +1,5 @@ - name: Join machine to headscale mesh network - hosts: all + hosts: managed become: yes vars_files: - ../infra_vars.yml diff --git a/ansible/inventory.ini b/ansible/inventory.ini index 3d6c209..a061637 100644 --- a/ansible/inventory.ini +++ b/ansible/inventory.ini @@ -23,3 +23,39 @@ localhost ansible_connection=local ansible_user=counterweight gpg_recipient=coun [arbret] prd-arbret ansible_host=167.99.242.62 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +[edge] +vipy + +[monitoring] +watchtower + +[vpn_control] +spacey + +[hypervisor] +nodito + +[bitcoin] +knots_box_local + +[electrum] +fulcrum_box_local + +[mempool] +mempool_box_local + +[memos] +memos_box_local + +[ci_runner] +forgejo_runner_local + +[control] +localhost + +# Every machine Ansible may configure as a server. +# Deliberately EXCLUDES [control] (your laptop) and [arbret]. +[managed:children] +vps +nodito_host +nodito_vms \ No newline at end of file From e63fa1ff110396168324b3633dd0563626916ff1 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:12 +0200 Subject: [PATCH 08/67] personal-blog: target edge group instead of vipy Co-Authored-By: Claude Opus 5 (1M context) --- .../services/personal-blog/deploy_personal_blog_playbook.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml index f4ee8ec..63bbae5 100644 --- a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml +++ b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy personal blog static site with Caddy file server - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From a26c0eca4635ff54af4ccb6065d67b950c4f0ac7 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:12 +0200 Subject: [PATCH 09/67] ntfy-emergency-app: target edge group instead of vipy Co-Authored-By: Claude Opus 5 (1M context) --- .../ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml index b8c0064..a04d6af 100644 --- a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml +++ b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy ntfy-emergency-app with Docker Compose and configure Caddy reverse proxy - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From 007fdb43af98f40ac90b91402e78f46f5d6f4c8b Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:12 +0200 Subject: [PATCH 10/67] memos: target memos and edge groups instead of hostnames Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/memos/deploy_memos_playbook.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/ansible/services/memos/deploy_memos_playbook.yml b/ansible/services/memos/deploy_memos_playbook.yml index da56bd6..a823135 100644 --- a/ansible/services/memos/deploy_memos_playbook.yml +++ b/ansible/services/memos/deploy_memos_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy Memos on memos-box - hosts: memos_box_local + hosts: memos become: yes vars_files: - ../../infra_vars.yml @@ -146,7 +146,7 @@ - name: Configure Caddy reverse proxy for Memos on vipy (proxying via Tailscale) - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From 31efa8365b1e57d12b89ac00dad240e68747c308 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:12 +0200 Subject: [PATCH 11/67] uptime_kuma: target monitoring group instead of watchtower Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/uptime_kuma/deploy_uptime_kuma_playbook.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ansible/services/uptime_kuma/deploy_uptime_kuma_playbook.yml b/ansible/services/uptime_kuma/deploy_uptime_kuma_playbook.yml index 4af3858..8754dce 100644 --- a/ansible/services/uptime_kuma/deploy_uptime_kuma_playbook.yml +++ b/ansible/services/uptime_kuma/deploy_uptime_kuma_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy Uptime Kuma with Docker Compose and configure Caddy reverse proxy - hosts: watchtower + hosts: monitoring become: yes vars_files: - ../../infra_vars.yml From 791f6f3b69a36ad5fc6f8ed30ea43ebd50cd9cdb Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:13 +0200 Subject: [PATCH 12/67] ntfy: target monitoring group instead of watchtower Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/ntfy/deploy_ntfy_playbook.yml | 2 +- ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/ansible/services/ntfy/deploy_ntfy_playbook.yml b/ansible/services/ntfy/deploy_ntfy_playbook.yml index 0729baa..d030123 100644 --- a/ansible/services/ntfy/deploy_ntfy_playbook.yml +++ b/ansible/services/ntfy/deploy_ntfy_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy ntfy and configure Caddy reverse proxy - hosts: watchtower + hosts: monitoring become: yes vars_files: - ../../infra_vars.yml diff --git a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml index 5ba03f1..8041905 100644 --- a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml +++ b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml @@ -1,5 +1,5 @@ - name: Setup ntfy as Uptime Kuma Notification Channel - hosts: watchtower + hosts: monitoring become: no vars_files: - ../../infra_vars.yml From cc1aecd098dfd7e4c91ec110bdff925fb9ca0406 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:13 +0200 Subject: [PATCH 13/67] vaultwarden: target edge group instead of vipy Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml | 2 +- .../vaultwarden/disable_vaultwarden_sign_ups_playbook.yml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml index 0340538..65f1f64 100644 --- a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml +++ b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy Vaultwarden with Docker Compose and configure Caddy reverse proxy - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml diff --git a/ansible/services/vaultwarden/disable_vaultwarden_sign_ups_playbook.yml b/ansible/services/vaultwarden/disable_vaultwarden_sign_ups_playbook.yml index b041e8e..eebb214 100644 --- a/ansible/services/vaultwarden/disable_vaultwarden_sign_ups_playbook.yml +++ b/ansible/services/vaultwarden/disable_vaultwarden_sign_ups_playbook.yml @@ -1,5 +1,5 @@ - name: Disable Vaultwarden Signups - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From 6680918fb7fc0c9bdc743605e4a5519cc90601af Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:13 +0200 Subject: [PATCH 14/67] forgejo: target edge group instead of vipy Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/forgejo/deploy_forgejo_playbook.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ansible/services/forgejo/deploy_forgejo_playbook.yml b/ansible/services/forgejo/deploy_forgejo_playbook.yml index e17d08f..ff865de 100644 --- a/ansible/services/forgejo/deploy_forgejo_playbook.yml +++ b/ansible/services/forgejo/deploy_forgejo_playbook.yml @@ -1,5 +1,5 @@ - name: Install Forgejo on Debian 12 with Caddy reverse proxy - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From f3ff1169db526a4e7cf6dd66d3ebe80e3846424f Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:13 +0200 Subject: [PATCH 15/67] forgejo-runner: target ci_runner group instead of forgejo_runner_local Co-Authored-By: Claude Opus 5 (1M context) --- .../services/forgejo-runner/deploy_forgejo_runner_playbook.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml index a194178..d94f73c 100644 --- a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml +++ b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml @@ -1,5 +1,5 @@ - name: Install Forgejo Runner on Debian 13 - hosts: forgejo_runner_local + hosts: ci_runner become: yes vars_files: - ../../infra_vars.yml From 900a0b4826be730c673409d05b8d660f32880f71 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:14 +0200 Subject: [PATCH 16/67] headscale: target vpn_control group instead of spacey Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/headscale/deploy_headscale_playbook.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ansible/services/headscale/deploy_headscale_playbook.yml b/ansible/services/headscale/deploy_headscale_playbook.yml index 1bcf5bf..7ed81ed 100644 --- a/ansible/services/headscale/deploy_headscale_playbook.yml +++ b/ansible/services/headscale/deploy_headscale_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy headscale and configure Caddy reverse proxy - hosts: spacey + hosts: vpn_control become: no vars_files: - ../../infra_vars.yml From 9ae0cc36d89de0ea83d09807f01fa3c712ff1c55 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:14 +0200 Subject: [PATCH 17/67] lnbits: target edge group instead of vipy Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/lnbits/deploy_lnbits_playbook.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ansible/services/lnbits/deploy_lnbits_playbook.yml b/ansible/services/lnbits/deploy_lnbits_playbook.yml index e5d546b..6df63f5 100644 --- a/ansible/services/lnbits/deploy_lnbits_playbook.yml +++ b/ansible/services/lnbits/deploy_lnbits_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy LNBits with Poetry and configure Caddy reverse proxy - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From 8d402b64fcffb1255aedad62ba2b81c275a75aea Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:14 +0200 Subject: [PATCH 18/67] phoenixd: target edge group instead of vipy Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/phoenixd/deploy_phoenixd_playbook.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml index 65ad7c9..9e465eb 100644 --- a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml +++ b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml @@ -20,7 +20,7 @@ # means losing the funds. See setup_backup_phoenixd_to_lapy.yml. - name: Deploy phoenixd on vipy - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From fd7803755bee1a52de3c8c1261fb99d4e209073e Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:14 +0200 Subject: [PATCH 19/67] mempool: target mempool and edge groups instead of hostnames Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/mempool/deploy_mempool_playbook.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/ansible/services/mempool/deploy_mempool_playbook.yml b/ansible/services/mempool/deploy_mempool_playbook.yml index 658180a..c0fe51d 100644 --- a/ansible/services/mempool/deploy_mempool_playbook.yml +++ b/ansible/services/mempool/deploy_mempool_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy Mempool Block Explorer with Docker - hosts: mempool_box_local + hosts: mempool become: yes vars_files: - ../../infra_vars.yml @@ -584,7 +584,7 @@ - /tmp/mempool_push_urls.yml - name: Configure Caddy reverse proxy for Mempool on vipy - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From a6c621e95c608286f86d3d0168c4d353bdd08fd2 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:15 +0200 Subject: [PATCH 20/67] fulcrum: target electrum and edge groups instead of hostnames Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/fulcrum/deploy_fulcrum_playbook.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml index 1255cd6..6b17280 100644 --- a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml +++ b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy Fulcrum Electrum Server - hosts: fulcrum_box_local + hosts: electrum become: yes vars_files: - ../../infra_vars.yml @@ -524,7 +524,7 @@ - name: Setup public Fulcrum SSL forwarding on vipy via systemd-socket-proxyd - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From 52d4a377f7062eb284ea7cccb64db4ce1f94f21e Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:15 +0200 Subject: [PATCH 21/67] bitcoin-knots: target bitcoin and edge groups instead of hostnames Co-Authored-By: Claude Opus 5 (1M context) --- .../services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml index d818945..f84d627 100644 --- a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +++ b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml @@ -1,5 +1,5 @@ - name: Build and Deploy Bitcoin Knots from Source - hosts: knots_box_local + hosts: bitcoin become: yes vars_files: - ../../infra_vars.yml @@ -724,7 +724,7 @@ - name: Setup public Bitcoin P2P forwarding on vipy via systemd-socket-proxyd - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From 4349006d51ea2769bc21b0717a2b240e2b49f197 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:15 +0200 Subject: [PATCH 22/67] datum-gateway: target bitcoin and edge groups instead of hostnames Co-Authored-By: Claude Opus 5 (1M context) --- .../datum-gateway/deploy_datum_gateway_playbook.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index f33ded7..5fa360c 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -25,7 +25,7 @@ # bitcoin_rpc_password - Shared with the bitcoin-knots deployment - name: Deploy DATUM Gateway on knots_box_local - hosts: knots_box_local + hosts: bitcoin become: yes vars_files: - ../../infra_vars.yml @@ -470,7 +470,7 @@ # Caddy Reverse Proxy for DATUM Dashboard (on vipy) # =========================================== - name: Configure Caddy reverse proxy for DATUM Gateway dashboard on vipy - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml @@ -663,7 +663,7 @@ # over the Tailscale network, matching the Bitcoin P2P proxy pattern. # =========================================== - name: Setup public Stratum port forwarding on vipy via systemd-socket-proxyd - hosts: vipy + hosts: edge become: yes vars_files: - ../../infra_vars.yml From ceb5c4ad2743d3c45fe446916407cf450c219b99 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:24 +0200 Subject: [PATCH 23/67] infra/nodito: target hypervisor group instead of nodito_host/nodito Co-Authored-By: Claude Opus 5 (1M context) --- ansible/infra/430_cpu_temp_alerts.yml | 2 +- ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml | 2 +- ansible/infra/nodito/31_proxmox_community_repos_playbook.yml | 2 +- ansible/infra/nodito/32_zfs_pool_setup_playbook.yml | 4 ++-- ansible/infra/nodito/33_proxmox_debian_cloud_template.yml | 2 +- ansible/infra/nodito/34_nut_ups_setup_playbook.yml | 4 ++-- 6 files changed, 8 insertions(+), 8 deletions(-) diff --git a/ansible/infra/430_cpu_temp_alerts.yml b/ansible/infra/430_cpu_temp_alerts.yml index 3b87102..b06f0b6 100644 --- a/ansible/infra/430_cpu_temp_alerts.yml +++ b/ansible/infra/430_cpu_temp_alerts.yml @@ -1,5 +1,5 @@ - name: Deploy CPU Temperature Monitoring - hosts: nodito_host + hosts: hypervisor become: yes vars_files: - ../infra_vars.yml diff --git a/ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml b/ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml index 02c6679..86f693f 100644 --- a/ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml +++ b/ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml @@ -1,5 +1,5 @@ - name: Bootstrap Nodito SSH Key Access - hosts: nodito_host + hosts: hypervisor become: true vars_files: - ../infra_vars.yml diff --git a/ansible/infra/nodito/31_proxmox_community_repos_playbook.yml b/ansible/infra/nodito/31_proxmox_community_repos_playbook.yml index b0be2ef..378a674 100644 --- a/ansible/infra/nodito/31_proxmox_community_repos_playbook.yml +++ b/ansible/infra/nodito/31_proxmox_community_repos_playbook.yml @@ -1,5 +1,5 @@ - name: Switch Proxmox VE from Enterprise to Community Repositories - hosts: nodito_host + hosts: hypervisor become: true vars_files: - ../infra_vars.yml diff --git a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml index cb72328..2939ce8 100644 --- a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml +++ b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml @@ -1,5 +1,5 @@ - name: Setup ZFS RAID 1 Pool for Proxmox Storage - hosts: nodito_host + hosts: hypervisor become: true vars_files: - ../infra_vars.yml @@ -172,7 +172,7 @@ when: "'ONLINE' not in final_zfs_status.stdout" - name: Setup ZFS Pool Health Monitoring and Monthly Scrubs - hosts: nodito + hosts: hypervisor become: true vars_files: - ../../infra_vars.yml diff --git a/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml b/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml index e8f8332..3b687ee 100644 --- a/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml +++ b/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml @@ -1,5 +1,5 @@ - name: Create Proxmox template from Debian cloud image (no VM clone) - hosts: nodito_host + hosts: hypervisor become: true vars_files: - ../../infra_vars.yml diff --git a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml index 02468d5..6b366ba 100644 --- a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml +++ b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml @@ -1,5 +1,5 @@ - name: Setup NUT (Network UPS Tools) for CyberPower UPS - hosts: nodito_host + hosts: hypervisor become: true vars_files: - ../../infra_vars.yml @@ -251,7 +251,7 @@ - name: Setup UPS Heartbeat Monitoring with Uptime Kuma - hosts: nodito + hosts: hypervisor become: true vars_files: - ../../infra_vars.yml From ed99e17aaeddfd1821b879eba02f4e38ad27aa7e Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 21:56:24 +0200 Subject: [PATCH 24/67] backups: target control group instead of lapy Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml | 2 +- ansible/services/headscale/setup_backup_headscale_to_lapy.yml | 2 +- ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml | 2 +- ansible/services/memos/setup_backup_memos_to_lapy.yml | 2 +- ansible/services/personal-blog/setup_deploy_alias_lapy.yml | 2 +- ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml | 2 +- .../services/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml | 2 +- .../services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml | 2 +- 8 files changed, 8 insertions(+), 8 deletions(-) diff --git a/ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml b/ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml index b90f0fb..c3ff8d6 100644 --- a/ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml +++ b/ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml @@ -1,6 +1,6 @@ --- - name: Configure local backup for Forgejo from remote - hosts: lapy + hosts: control gather_facts: no vars_files: - ../../infra_vars.yml diff --git a/ansible/services/headscale/setup_backup_headscale_to_lapy.yml b/ansible/services/headscale/setup_backup_headscale_to_lapy.yml index 5f9a764..acc82ce 100644 --- a/ansible/services/headscale/setup_backup_headscale_to_lapy.yml +++ b/ansible/services/headscale/setup_backup_headscale_to_lapy.yml @@ -1,5 +1,5 @@ - name: Configure local backup for Headscale from remote - hosts: lapy + hosts: control gather_facts: no vars_files: - ../../infra_vars.yml diff --git a/ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml b/ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml index 5d10dec..48296cb 100644 --- a/ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml +++ b/ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml @@ -1,5 +1,5 @@ - name: Configure local backup for LNBits from remote - hosts: lapy + hosts: control gather_facts: no vars_files: - ../../infra_vars.yml diff --git a/ansible/services/memos/setup_backup_memos_to_lapy.yml b/ansible/services/memos/setup_backup_memos_to_lapy.yml index 6d9c161..c32c332 100644 --- a/ansible/services/memos/setup_backup_memos_to_lapy.yml +++ b/ansible/services/memos/setup_backup_memos_to_lapy.yml @@ -1,5 +1,5 @@ - name: Configure local backup for Memos from memos-box - hosts: lapy + hosts: control gather_facts: no vars_files: - ../../infra_vars.yml diff --git a/ansible/services/personal-blog/setup_deploy_alias_lapy.yml b/ansible/services/personal-blog/setup_deploy_alias_lapy.yml index 2b90d68..99f9b34 100644 --- a/ansible/services/personal-blog/setup_deploy_alias_lapy.yml +++ b/ansible/services/personal-blog/setup_deploy_alias_lapy.yml @@ -1,5 +1,5 @@ - name: Configure deployment alias for personal blog in lapy .bashrc - hosts: lapy + hosts: control gather_facts: no vars_files: - ../../infra_vars.yml diff --git a/ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml b/ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml index cc008e6..7474af1 100644 --- a/ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml +++ b/ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml @@ -13,7 +13,7 @@ # Because both files are static, phoenixd does not need to be stopped. - name: Configure local backup for phoenixd from remote - hosts: lapy + hosts: control gather_facts: no vars_files: - ../../infra_vars.yml diff --git a/ansible/services/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml b/ansible/services/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml index 9ae9713..54c0b79 100644 --- a/ansible/services/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml +++ b/ansible/services/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml @@ -1,5 +1,5 @@ - name: Configure local backup for Uptime Kuma from remote - hosts: lapy + hosts: control gather_facts: no vars_files: - ../../infra_vars.yml diff --git a/ansible/services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml b/ansible/services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml index 064d633..096f9d3 100644 --- a/ansible/services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml +++ b/ansible/services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml @@ -1,5 +1,5 @@ - name: Configure local backup for Vaultwarden from remote - hosts: lapy + hosts: control gather_facts: no vars_files: - ../../infra_vars.yml From 3c2aacad46c18bfafae6ac32bb2ac563de4a50af Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 22:00:43 +0200 Subject: [PATCH 25/67] vars: derive remote_host_name from role groups instead of hostnames Resolved values verified unchanged: same host, IP, user, key and port for all 8. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/services/forgejo/forgejo_vars.yml | 2 +- ansible/services/headscale/headscale_vars.yml | 2 +- ansible/services/lnbits/lnbits_vars.yml | 2 +- ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml | 2 +- ansible/services/personal-blog/personal_blog_vars.yml | 2 +- ansible/services/phoenixd/phoenixd_vars.yml | 2 +- ansible/services/uptime_kuma/uptime_kuma_vars.yml | 2 +- ansible/services/vaultwarden/vaultwarden_vars.yml | 2 +- 8 files changed, 8 insertions(+), 8 deletions(-) diff --git a/ansible/services/forgejo/forgejo_vars.yml b/ansible/services/forgejo/forgejo_vars.yml index 0bbb5a5..2a24133 100644 --- a/ansible/services/forgejo/forgejo_vars.yml +++ b/ansible/services/forgejo/forgejo_vars.yml @@ -12,7 +12,7 @@ forgejo_user: "git" # (caddy_sites_dir and subdomain now in services_config.yml) # Remote access -remote_host_name: "vipy" +remote_host_name: "{{ groups['edge'] | first }}" remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" diff --git a/ansible/services/headscale/headscale_vars.yml b/ansible/services/headscale/headscale_vars.yml index 653c175..0b0ee50 100644 --- a/ansible/services/headscale/headscale_vars.yml +++ b/ansible/services/headscale/headscale_vars.yml @@ -13,7 +13,7 @@ headscale_data_dir: /var/lib/headscale # Namespace now configured in services_config.yml under service_settings.headscale.namespace # Remote access -remote_host_name: "spacey" +remote_host_name: "{{ groups['vpn_control'] | first }}" remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" diff --git a/ansible/services/lnbits/lnbits_vars.yml b/ansible/services/lnbits/lnbits_vars.yml index bdb97df..eabd6fc 100644 --- a/ansible/services/lnbits/lnbits_vars.yml +++ b/ansible/services/lnbits/lnbits_vars.yml @@ -6,7 +6,7 @@ lnbits_port: 8765 # (caddy_sites_dir and subdomain now in services_config.yml) # Remote access -remote_host_name: "vipy" +remote_host_name: "{{ groups['edge'] | first }}" remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" diff --git a/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml b/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml index 415bc4d..a3bb480 100644 --- a/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml +++ b/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml @@ -9,7 +9,7 @@ ntfy_emergency_app_topic: "emergencia" ntfy_emergency_app_ui_message: "Leave Pablo a message, he will respond as soon as possible" # Remote access -remote_host_name: "vipy" +remote_host_name: "{{ groups['edge'] | first }}" remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" diff --git a/ansible/services/personal-blog/personal_blog_vars.yml b/ansible/services/personal-blog/personal_blog_vars.yml index 59e0921..a1f34e3 100644 --- a/ansible/services/personal-blog/personal_blog_vars.yml +++ b/ansible/services/personal-blog/personal_blog_vars.yml @@ -4,7 +4,7 @@ personal_blog_web_root: "/var/www/pablohere.contrapeso.xyz" # Remote access for deployment -remote_host_name: "vipy" +remote_host_name: "{{ groups['edge'] | first }}" remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" diff --git a/ansible/services/phoenixd/phoenixd_vars.yml b/ansible/services/phoenixd/phoenixd_vars.yml index 57218dc..f67c302 100644 --- a/ansible/services/phoenixd/phoenixd_vars.yml +++ b/ansible/services/phoenixd/phoenixd_vars.yml @@ -41,7 +41,7 @@ phoenixd_healthcheck_service_name: phoenixd-healthcheck phoenixd_monitor_name: "Phoenixd" # Remote access -remote_host_name: "vipy" +remote_host_name: "{{ groups['edge'] | first }}" remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" diff --git a/ansible/services/uptime_kuma/uptime_kuma_vars.yml b/ansible/services/uptime_kuma/uptime_kuma_vars.yml index 3263f49..33ffb6d 100644 --- a/ansible/services/uptime_kuma/uptime_kuma_vars.yml +++ b/ansible/services/uptime_kuma/uptime_kuma_vars.yml @@ -4,7 +4,7 @@ uptime_kuma_data_dir: "{{ uptime_kuma_dir }}/data" uptime_kuma_port: 3001 # Remote access -remote_host_name: "watchtower" +remote_host_name: "{{ groups['monitoring'] | first }}" remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" diff --git a/ansible/services/vaultwarden/vaultwarden_vars.yml b/ansible/services/vaultwarden/vaultwarden_vars.yml index 75e527d..e60f1d4 100644 --- a/ansible/services/vaultwarden/vaultwarden_vars.yml +++ b/ansible/services/vaultwarden/vaultwarden_vars.yml @@ -6,7 +6,7 @@ vaultwarden_port: 8222 # (caddy_sites_dir and subdomain now in services_config.yml) # Remote access -remote_host_name: "vipy" +remote_host_name: "{{ groups['edge'] | first }}" remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" From b0f14ea3655f96cb817f9435731b14a26db5ea1a Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 22:17:13 +0200 Subject: [PATCH 26/67] vars --- ansible/group_vars/all/main.yml | 4 +++ ansible/group_vars/all/vault.yml | 50 +++++++++++++++++++++++++++++++ ansible/host_vars/nodito/main.yml | 28 +++++++++++++++++ 3 files changed, 82 insertions(+) create mode 100644 ansible/group_vars/all/main.yml create mode 100644 ansible/group_vars/all/vault.yml create mode 100644 ansible/host_vars/nodito/main.yml diff --git a/ansible/group_vars/all/main.yml b/ansible/group_vars/all/main.yml new file mode 100644 index 0000000..952df93 --- /dev/null +++ b/ansible/group_vars/all/main.yml @@ -0,0 +1,4 @@ +new_user: counterweight +ssh_port: 22 +allow_ssh_from: "any" +root_domain: contrapeso.xyz diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml new file mode 100644 index 0000000..c48d456 --- /dev/null +++ b/ansible/group_vars/all/vault.yml @@ -0,0 +1,50 @@ +$ANSIBLE_VAULT;1.1;AES256 +32643038333165373363353439313639326538343435393633343765386662643231333936386630 +6564313061366166353662373564393261396666663432610a633161643066663232336163306663 +61623535643231366638616630666562333936383232373130623631306334653634386132396431 +3036383735316339650a313565313062636461396563336461633838343037666237623038633133 +35333531313036666531633230646464636631666133313636353362646336623135363736316562 +36656564396230333231643266303833633134303862323633636136363136373836346363663364 +32306234393863613534663031646632643734366531343339386133363837396138616435313365 +65353037326266386538633665386638656539376133636435346136643163643238656434346132 +34363036623935343162313031343036393536613136306435323730346561393261333966663465 +35333864323638356565316632353662366262363962376134366432333361343639633133626533 +65303964643330393264643136346432313164396664333535363135646464326539626433623732 +34666530666234376637623930633433656163353136646134326365353563343833666161666435 +30313866396136623036613933376164346533383964303664633830376636363565323832343136 +62653137366632313133626234386235356630383433623934303737653438623136366638396237 +66303435356539343061313337646261343738373866646465373339616161323531346361383363 +64643363616130323935656264326631336231323236616438366632356333333534303861663238 +61376464623030363938653230316265366533623364653930343537326532393433363230356136 +62643135636130626436386239366561393537333736383361336162636434656339653630353732 +37343265366234656637336135633461333735363932636130323564623661303730303363383833 +34373963373765643437303534366233626662656531373261633565353137643761356135376430 +30373639336565356364313333333237323338346631666235613961656336643438653834313137 +35303666366433646334643264633837343134373637643535633864313062613336333538303931 +34396266326166666165333639363961346639343130656235323332346631643539363861666436 +63343733656362313165623130383838376138613534363535336662366564306132353362666463 +61613534613431623263323831646635323362343634616566393834353964643861633639346132 +66363233373964323139313966616365343535306365373532313063383236353964343265383761 +64663833363035363736643236373634663466666138346266643930643033343839313534376666 +37363461623965636164663339386237643635353237353263633131363434643066393538363639 +39346530366361656536383233663766616234383535643234616233666262353364633036663062 +64386663383363633365656631326332353261303562346666636565313564306635333562363930 +33306237333638323131613161336165636237623332643137363038303330313663346639326361 +39383565366431363361313361303436343761626462306537633337393636333135343833623464 +32323663616166653063303861396664613434643536633066626466303634663031623938313436 +32303763616562376137303061633864393863373334373338626566653633383132306635396566 +64616463663265643732366465666662663334666362383036326330356535333465326534363266 +33326265326365353231613839383336656562333837376565306230646466333238323936316463 +64396266306562303438363635343865376364333239376664346534636664326532353265313563 +66636331333430616462326334623637646539643638323036303335656632363565373739373234 +35666532346364376335346238376562336164386363623862356539316466336531643936336131 +31356332306639626532646338393638373764653961616638643066313832626561336435383839 +64363232313633643133383066636539663933323064366536666462646466323166306562623538 +64393835383431663561343231303961383337656331663261393439336136623066666434336362 +36313239336537376134396361643532633437633163373931353139303534356665383538393838 +31306336326538353232336533643961383337313161663266353730643931656165353264303938 +30636363666136653963306130666433633562313464653862656262313035306635306230653666 +34336631623235323734616163623039613636393863383966336435313230313231356563636337 +61373336336563653030366130356331616438336638323263663164366435653266643161653331 +31323733313936313833386665633235633339376330373161326361336266633838326336356363 +37383165393866656565316337386639346531323439626465613834303636663933 diff --git a/ansible/host_vars/nodito/main.yml b/ansible/host_vars/nodito/main.yml new file mode 100644 index 0000000..c0002f3 --- /dev/null +++ b/ansible/host_vars/nodito/main.yml @@ -0,0 +1,28 @@ +# Nodito CPU Temperature Monitoring Configuration + +# Temperature Monitoring Configuration +temp_threshold_celsius: 80 +temp_check_interval_minutes: 1 + +# Script Configuration +monitoring_script_dir: /opt/nodito-monitoring +monitoring_script_path: "{{ monitoring_script_dir }}/cpu_temp_monitor.sh" +log_file: "{{ monitoring_script_dir }}/cpu_temp_monitor.log" + +# System Configuration +systemd_service_name: nodito-cpu-temp-monitor + +# ZFS Pool Configuration +zfs_pool_name: "proxmox-tank-1" +zfs_disk_1: "/dev/disk/by-id/ata-ST4000NT001-3M2101_WX11TN0Z" # First disk for RAID 1 mirror +zfs_disk_2: "/dev/disk/by-id/ata-ST4000NT001-3M2101_WX11TN2P" # Second disk for RAID 1 mirror +zfs_pool_mountpoint: "/var/lib/vz" + +# UPS Configuration (CyberPower CP900EPFCLCD via USB) +ups_name: cyberpower +ups_desc: "CyberPower CP900EPFCLCD" +ups_driver: usbhid-ups +ups_port: auto +ups_user: counterweight +ups_offdelay: 120 # Seconds after shutdown before UPS cuts outlet power +ups_ondelay: 30 # Seconds after mains returns before UPS restores outlet power From 79942525b12808673754792d4fcf6e1cde08b3ca Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 22:43:15 +0200 Subject: [PATCH 27/67] archive: record Uptime Kuma monitors and setup before decommissioning MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Captured from the live instance rather than the repo: the playbooks created 17 monitors, the server had 75. The rest existed only in the UI. Push tokens are excluded deliberately — they are live credentials. Co-Authored-By: Claude Opus 5 (1M context) --- archive/uptime_kuma/MONITORS.md | 152 +++ archive/uptime_kuma/README.md | 51 + .../deploy_uptime_kuma_playbook.yml | 0 archive/uptime_kuma/monitors.json | 1202 +++++++++++++++++ .../setup_backup_uptime_kuma_to_lapy.yml | 0 .../uptime_kuma/uptime_kuma_vars.yml | 0 6 files changed, 1405 insertions(+) create mode 100644 archive/uptime_kuma/MONITORS.md create mode 100644 archive/uptime_kuma/README.md rename {ansible/services => archive}/uptime_kuma/deploy_uptime_kuma_playbook.yml (100%) create mode 100644 archive/uptime_kuma/monitors.json rename {ansible/services => archive}/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml (100%) rename {ansible/services => archive}/uptime_kuma/uptime_kuma_vars.yml (100%) diff --git a/archive/uptime_kuma/MONITORS.md b/archive/uptime_kuma/MONITORS.md new file mode 100644 index 0000000..53b1d89 --- /dev/null +++ b/archive/uptime_kuma/MONITORS.md @@ -0,0 +1,152 @@ +# Uptime Kuma — monitor inventory (archived) + +Captured from the live instance at `https://uptime.contrapeso.xyz` on 2026-09-11, +immediately before decommissioning. This is the **authoritative** record: most of +these monitors existed only in the Uptime Kuma UI and were never described by any +playbook in this repo. + +**75 monitors total** — 16 group, 8 http, 3 port, 48 push. All were active. + +Push tokens are deliberately **not** recorded here: they are live credentials, and +anything holding one could report a false 'up'. They die with the server. + +--- + +## arbret - production *(7 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| arbret.com - arbret-analytics | push | — | 120s | Healthy when timer is scheduled and last run succeeded | +| arbret.com - arbret-backup | push | — | 120s | Healthy when timer is scheduled and last run succeeded | +| arbret.com - arbret-server | push | — | 120s | Healthy when arbret-server.service is active | +| arbret.com - arbret-worker | push | — | 120s | Healthy when arbret-worker.service is active | +| arbret.com - health | push | — | 120s | Healthy when GET /api/health returns status ok | +| arbret.com - https | push | — | 120s | Healthy when HTTPS front-door returns 200 | +| arbret.com - postgresql | push | — | 120s | Healthy when postgresql.service is active | + +## arbret - staging *(7 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| arbretstaging.contrapeso.xyz - arbret-analytics | push | — | 120s | Healthy when timer is scheduled and last run succeeded | +| arbretstaging.contrapeso.xyz - arbret-backup | push | — | 120s | Healthy when timer is scheduled and last run succeeded | +| arbretstaging.contrapeso.xyz - arbret-server | push | — | 120s | Healthy when arbret-server.service is active | +| arbretstaging.contrapeso.xyz - arbret-worker | push | — | 120s | Healthy when arbret-worker.service is active | +| arbretstaging.contrapeso.xyz - health | push | — | 120s | Healthy when GET /api/health returns status ok | +| arbretstaging.contrapeso.xyz - https | push | — | 120s | Healthy when HTTPS front-door returns 200 | +| arbretstaging.contrapeso.xyz - postgresql | push | — | 120s | Healthy when postgresql.service is active | + +## arbret-staging-box - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-arbret-staging-box-root | push | — | 960s | upside-down, Disk Usage: arbret-staging-box (/) - Alerts when usage excee | +| system-healthcheck-arbret-staging-box | push | — | 90s | System Healthcheck: arbret-staging-box - Regular healthcheck | + +## forgejo-runner-box - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-forgejo-runner-box-root | push | — | 960s | upside-down, Disk Usage: forgejo-runner-box (/) - Alerts when usage excee | +| system-healthcheck-forgejo-runner-box | push | — | 90s | System Healthcheck: forgejo-runner-box - Regular healthcheck | + +## fulcrum-box - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-fulcrum-box-root | push | — | 960s | upside-down, Disk Usage: fulcrum-box (/) - Alerts when usage exceeds 80% | +| system-healthcheck-fulcrum-box | push | — | 90s | System Healthcheck: fulcrum-box - Regular healthcheck ping e | + +## knots-box - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-knots-box-root | push | — | 960s | upside-down, Disk Usage: knots-box (/) - Alerts when usage exceeds 80% | +| system-healthcheck-knots-box | push | — | 90s | System Healthcheck: knots-box - Regular healthcheck ping eve | + +## memos-box - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-memos-box-root | push | — | 960s | upside-down, Disk Usage: memos-box (/) - Alerts when usage exceeds 80% | +| system-healthcheck-memos-box | push | — | 90s | System Healthcheck: memos-box - Regular healthcheck ping eve | + +## mempool-box - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-mempool-box-root | push | — | 960s | upside-down, Disk Usage: mempool-box (/) - Alerts when usage exceeds 80% | +| system-healthcheck-mempool-box | push | — | 90s | System Healthcheck: mempool-box - Regular healthcheck ping e | + +## nodito - infra *(4 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| UPS ONLINE | push | https:// | 90s | — | +| cpu-temp-nodito | push | — | 120s | upside-down, CPU Temperature: nodito - Alerts when temperature exceeds 80 | +| system-healthcheck-nodito | push | — | 300s | System Healthcheck: nodito - Regular healthcheck ping every | +| zfs-health-nodito | push | — | 90000s | ZFS Pool Health: nodito - Daily health check for pool proxmo | + +## nonkeiwaisi-box - infra *(0 monitors)* + +_(empty)_ + +## prd-arbret - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-prd-arbret-root | push | — | 960s | upside-down, Disk Usage: prd-arbret (/) - Alerts when usage exceeds 80% | +| system-healthcheck-prd-arbret | push | — | 90s | System Healthcheck: prd-arbret - Regular healthcheck ping ev | + +## prd-spacey - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-prd-spacey-root | push | — | 960s | upside-down, Disk Usage: prd-spacey (/) - Alerts when usage exceeds 80% | +| system-healthcheck-prd-spacey | push | — | 90s | System Healthcheck: prd-spacey - Regular healthcheck ping ev | + +## prd-vipy - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-prd-vipy-root | push | — | 960s | upside-down, Disk Usage: prd-vipy (/) - Alerts when usage exceeds 80% | +| system-healthcheck-prd-vipy | push | — | 90s | System Healthcheck: prd-vipy - Regular healthcheck ping ever | + +## prd-watchtower - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-prd-watchtower-root | push | — | 960s | upside-down, Disk Usage: prd-watchtower (/) - Alerts when usage exceeds 8 | +| system-healthcheck-prd-watchtower | push | — | 90s | System Healthcheck: prd-watchtower - Regular healthcheck pin | + +## services *(19 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| Forgejo | http | https://forgejo.contrapeso.xyz/api/healthz | 90s | — | +| Headscale | http | https://headscale.contrapeso.xyz/health | 60s | — | +| LNBits | http | https://wallet.contrapeso.xyz/api/v1/health | 60s | — | +| Memos | http | https://memos.contrapeso.xyz/healthz | 60s | — | +| Mempool Public Access | http | https://mempool.contrapeso.xyz | 60s | — | +| Personal Blog | http | https://pablohere.contrapeso.xyz | 60s | — | +| Vaultwarden | http | https://vault.contrapeso.xyz/alive | 60s | — | +| ntfy-emergency-app | http | https://avisame.contrapeso.xyz | 60s | — | +| Bitcoin Knots P2P Public | port | 167.172.107.33:8333 | 60s | — | +| DATUM Stratum (public) | port | 167.172.107.33:23334 | 60s | — | +| Fulcrum SSL Public | port | 167.172.107.33:50002 | 60s | — | +| Bitcoin Knots | push | — | 90s | — | +| DATUM Gateway | push | — | 90s | — | +| Fulcrum | push | — | 90s | — | +| Mempool Backend | push | — | 180s | — | +| Mempool Frontend | push | — | 90s | — | +| Mempool MariaDB | push | — | 90s | — | +| Phoenixd | push | — | 90s | — | +| forgejo-runner-healthcheck | push | — | 90s | Forgejo Runner healthcheck - ping every 60s | + +## small-backups-box - infra *(2 monitors)* + +| Monitor | Type | Target | Interval | Notes | +|---|---|---|---|---| +| disk-usage-small-backups-box-root | push | — | 960s | upside-down, Disk Usage: small-backups-box (/) - Alerts when usage exceed | +| system-healthcheck-small-backups-box | push | — | 90s | System Healthcheck: small-backups-box - Regular healthcheck | + diff --git a/archive/uptime_kuma/README.md b/archive/uptime_kuma/README.md new file mode 100644 index 0000000..5405ae0 --- /dev/null +++ b/archive/uptime_kuma/README.md @@ -0,0 +1,51 @@ +# Uptime Kuma — archived + +Uptime Kuma was the monitoring stack for this infrastructure until **2026-09-11**, when +it was decommissioned. Everything that referenced it has been removed from the live +playbooks; this folder is the record of what it was, kept so the setup can be understood +later without digging through git history. + +## Contents + +| File | What it is | +|---|---| +| `MONITORS.md` | Every monitor that existed, grouped as it was in the UI. The authoritative record. | +| `monitors.json` | The same data, machine-readable, as returned by the API. | +| `deploy_uptime_kuma_playbook.yml` | How the server itself was deployed (Docker Compose on `monitoring`, behind Caddy). | +| `uptime_kuma_vars.yml` | Variables the deploy playbook needed — it is unreadable without these. | +| `setup_backup_uptime_kuma_to_lapy.yml` | How its data was backed up, and to where. Useful when disposing of the old volumes. | + +## Why the inventory was captured from the live server, not the repo + +The playbooks only ever created **17** monitors. The live instance had **75**. The +difference was created by hand in the UI and existed nowhere else — so a repo-derived +list would have silently lost two thirds of the picture. `MONITORS.md` is a snapshot of +the real thing, taken immediately before removal. + +Push tokens are deliberately excluded. They are live credentials — anything holding one +can report a false "up" — and they become meaningless once the server is gone. + +## What removal did NOT do + +Removing the playbook code does not touch the machines. **48 push monitors** were driven +by scripts and systemd timers installed *on the hosts*, which keep firing on their +schedule and curling an endpoint that no longer answers. They are harmless but they are +still there, writing logs and failing quietly. + +Left behind, per host: + +- `/opt/disk-monitoring` + `disk-usage-monitor.{service,timer}` — on all 12 `managed` hosts +- `/opt/system-healthcheck` + `system-healthcheck.{service,timer}` — on all 12 `managed` hosts +- `/opt/nodito-monitoring` + `nodito-cpu-temp-monitor.{service,timer}` — `nodito` +- `/opt/zfs-monitoring` — `nodito` +- `/opt/ups-monitoring` — `nodito` +- `bitcoin-knots-healthcheck.{service,timer}` — `bitcoin` +- `datum-gateway-healthcheck.{service,timer}` — `bitcoin` +- `fulcrum-healthcheck.{service,timer}` — `electrum` +- `mempool-{backend,frontend,mariadb}-healthcheck.service` — `mempool` +- phoenixd and forgejo-runner healthcheck units — `edge`, `ci_runner` + +Note `nut-monitor.service` on `nodito` is **NUT's own daemon**, not a monitoring +leftover — do not remove it with the rest. + +Cleaning these up is a separate decommissioning pass and was not part of the removal. diff --git a/ansible/services/uptime_kuma/deploy_uptime_kuma_playbook.yml b/archive/uptime_kuma/deploy_uptime_kuma_playbook.yml similarity index 100% rename from ansible/services/uptime_kuma/deploy_uptime_kuma_playbook.yml rename to archive/uptime_kuma/deploy_uptime_kuma_playbook.yml diff --git a/archive/uptime_kuma/monitors.json b/archive/uptime_kuma/monitors.json new file mode 100644 index 0000000..d1b0b0a --- /dev/null +++ b/archive/uptime_kuma/monitors.json @@ -0,0 +1,1202 @@ +[ + { + "id": 1, + "name": "services", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 3, + "name": "Vaultwarden", + "type": "http", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": "https://vault.contrapeso.xyz/alive", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 4, + "name": "prd-spacey - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 5, + "name": "prd-watchtower - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 6, + "name": "prd-vipy - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 7, + "name": "disk-usage-prd-vipy-root", + "type": "push", + "active": true, + "parent": 6, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: prd-vipy (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 8, + "name": "disk-usage-prd-watchtower-root", + "type": "push", + "active": true, + "parent": 5, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: prd-watchtower (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 9, + "name": "disk-usage-prd-spacey-root", + "type": "push", + "active": true, + "parent": 4, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: prd-spacey (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 10, + "name": "system-healthcheck-prd-vipy", + "type": "push", + "active": true, + "parent": 6, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: prd-vipy - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 11, + "name": "system-healthcheck-prd-spacey", + "type": "push", + "active": true, + "parent": 4, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: prd-spacey - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 12, + "name": "system-healthcheck-prd-watchtower", + "type": "push", + "active": true, + "parent": 5, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: prd-watchtower - Regular healthcheck ping every 60s", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 13, + "name": "Forgejo", + "type": "http", + "active": true, + "parent": 1, + "interval": 90, + "retries": null, + "url": "https://forgejo.contrapeso.xyz/api/healthz", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 10, + "resendInterval": 0 + }, + { + "id": 14, + "name": "Personal Blog", + "type": "http", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": "https://pablohere.contrapeso.xyz", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 15, + "name": "LNBits", + "type": "http", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": "https://wallet.contrapeso.xyz/api/v1/health", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 16, + "name": "ntfy-emergency-app", + "type": "http", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": "https://avisame.contrapeso.xyz", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 17, + "name": "Headscale", + "type": "http", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": "https://headscale.contrapeso.xyz/health", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 18, + "name": "knots-box - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 19, + "name": "disk-usage-knots-box-root", + "type": "push", + "active": true, + "parent": 18, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: knots-box (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 20, + "name": "system-healthcheck-knots-box", + "type": "push", + "active": true, + "parent": 18, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: knots-box - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 21, + "name": "nodito - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 22, + "name": "system-healthcheck-nodito", + "type": "push", + "active": true, + "parent": 21, + "interval": 300, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: nodito - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 23, + "name": "cpu-temp-nodito", + "type": "push", + "active": true, + "parent": 21, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "CPU Temperature: nodito - Alerts when temperature exceeds 80\u00b0C", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 24, + "name": "Bitcoin Knots", + "type": "push", + "active": true, + "parent": 1, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 25, + "name": "fulcrum-box - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 26, + "name": "disk-usage-fulcrum-box-root", + "type": "push", + "active": true, + "parent": 25, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: fulcrum-box (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 27, + "name": "system-healthcheck-fulcrum-box", + "type": "push", + "active": true, + "parent": 25, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: fulcrum-box - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 28, + "name": "Fulcrum", + "type": "push", + "active": true, + "parent": 1, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 35, + "name": "mempool-box - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 36, + "name": "disk-usage-mempool-box-root", + "type": "push", + "active": true, + "parent": 35, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: mempool-box (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 37, + "name": "system-healthcheck-mempool-box", + "type": "push", + "active": true, + "parent": 35, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: mempool-box - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 38, + "name": "Mempool MariaDB", + "type": "push", + "active": true, + "parent": 1, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 39, + "name": "Mempool Backend", + "type": "push", + "active": true, + "parent": 1, + "interval": 180, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 40, + "name": "Mempool Frontend", + "type": "push", + "active": true, + "parent": 1, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 41, + "name": "Mempool Public Access", + "type": "http", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": "https://mempool.contrapeso.xyz", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 42, + "name": "memos-box - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 43, + "name": "disk-usage-memos-box-root", + "type": "push", + "active": true, + "parent": 42, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: memos-box (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 44, + "name": "system-healthcheck-memos-box", + "type": "push", + "active": true, + "parent": 42, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: memos-box - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 45, + "name": "Memos", + "type": "http", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": "https://memos.contrapeso.xyz/healthz", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 46, + "name": "Bitcoin Knots P2P Public", + "type": "port", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": null, + "hostname": "167.172.107.33", + "port": 8333, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 47, + "name": "Fulcrum SSL Public", + "type": "port", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": null, + "hostname": "167.172.107.33", + "port": 50002, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 49, + "name": "zfs-health-nodito", + "type": "push", + "active": true, + "parent": 21, + "interval": 90000, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "ZFS Pool Health: nodito - Daily health check for pool proxmox-tank-1", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 50, + "name": "UPS ONLINE", + "type": "push", + "active": true, + "parent": 21, + "interval": 90, + "retries": null, + "url": "https://", + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 51, + "name": "forgejo-runner-box - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 52, + "name": "disk-usage-forgejo-runner-box-root", + "type": "push", + "active": true, + "parent": 51, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: forgejo-runner-box (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 53, + "name": "system-healthcheck-forgejo-runner-box", + "type": "push", + "active": true, + "parent": 51, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: forgejo-runner-box - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 55, + "name": "forgejo-runner-healthcheck", + "type": "push", + "active": true, + "parent": 1, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Forgejo Runner healthcheck - ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 59, + "name": "arbret-staging-box - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 60, + "name": "disk-usage-arbret-staging-box-root", + "type": "push", + "active": true, + "parent": 59, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: arbret-staging-box (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 61, + "name": "system-healthcheck-arbret-staging-box", + "type": "push", + "active": true, + "parent": 59, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: arbret-staging-box - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 66, + "name": "arbret - staging", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 67, + "name": "arbretstaging.contrapeso.xyz - health", + "type": "push", + "active": true, + "parent": 66, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when GET /api/health returns status ok", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 68, + "name": "arbretstaging.contrapeso.xyz - https", + "type": "push", + "active": true, + "parent": 66, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when HTTPS front-door returns 200", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 69, + "name": "arbretstaging.contrapeso.xyz - postgresql", + "type": "push", + "active": true, + "parent": 66, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when postgresql.service is active", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 70, + "name": "arbretstaging.contrapeso.xyz - arbret-server", + "type": "push", + "active": true, + "parent": 66, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when arbret-server.service is active", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 71, + "name": "arbretstaging.contrapeso.xyz - arbret-worker", + "type": "push", + "active": true, + "parent": 66, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when arbret-worker.service is active", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 72, + "name": "arbretstaging.contrapeso.xyz - arbret-analytics", + "type": "push", + "active": true, + "parent": 66, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when timer is scheduled and last run succeeded", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 73, + "name": "arbretstaging.contrapeso.xyz - arbret-backup", + "type": "push", + "active": true, + "parent": 66, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when timer is scheduled and last run succeeded", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 74, + "name": "prd-arbret - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 85, + "name": "disk-usage-prd-arbret-root", + "type": "push", + "active": true, + "parent": 74, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: prd-arbret (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 86, + "name": "system-healthcheck-prd-arbret", + "type": "push", + "active": true, + "parent": 74, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: prd-arbret - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 87, + "name": "arbret - production", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 88, + "name": "arbret.com - health", + "type": "push", + "active": true, + "parent": 87, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when GET /api/health returns status ok", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 89, + "name": "arbret.com - https", + "type": "push", + "active": true, + "parent": 87, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when HTTPS front-door returns 200", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 90, + "name": "arbret.com - postgresql", + "type": "push", + "active": true, + "parent": 87, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when postgresql.service is active", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 91, + "name": "arbret.com - arbret-server", + "type": "push", + "active": true, + "parent": 87, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when arbret-server.service is active", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 92, + "name": "arbret.com - arbret-worker", + "type": "push", + "active": true, + "parent": 87, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when arbret-worker.service is active", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 93, + "name": "arbret.com - arbret-analytics", + "type": "push", + "active": true, + "parent": 87, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when timer is scheduled and last run succeeded", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 94, + "name": "arbret.com - arbret-backup", + "type": "push", + "active": true, + "parent": 87, + "interval": 120, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "Healthy when timer is scheduled and last run succeeded", + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 95, + "name": "small-backups-box - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 96, + "name": "disk-usage-small-backups-box-root", + "type": "push", + "active": true, + "parent": 95, + "interval": 960, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": true, + "description": "Disk Usage: small-backups-box (/) - Alerts when usage exceeds 80%", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 97, + "name": "system-healthcheck-small-backups-box", + "type": "push", + "active": true, + "parent": 95, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": "System Healthcheck: small-backups-box - Regular healthcheck ping every 60s", + "maxretries": 1, + "resendInterval": 0 + }, + { + "id": 109, + "name": "DATUM Gateway", + "type": "push", + "active": true, + "parent": 1, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 111, + "name": "DATUM Stratum (public)", + "type": "port", + "active": true, + "parent": 1, + "interval": 60, + "retries": null, + "url": null, + "hostname": "167.172.107.33", + "port": 23334, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 112, + "name": "Phoenixd", + "type": "push", + "active": true, + "parent": 1, + "interval": 90, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 3, + "resendInterval": 0 + }, + { + "id": 113, + "name": "nonkeiwaisi-box - infra", + "type": "group", + "active": true, + "parent": null, + "interval": 60, + "retries": null, + "url": null, + "hostname": null, + "port": null, + "upsideDown": false, + "description": null, + "maxretries": 1, + "resendInterval": 0 + } +] diff --git a/ansible/services/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml b/archive/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml similarity index 100% rename from ansible/services/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml rename to archive/uptime_kuma/setup_backup_uptime_kuma_to_lapy.yml diff --git a/ansible/services/uptime_kuma/uptime_kuma_vars.yml b/archive/uptime_kuma/uptime_kuma_vars.yml similarity index 100% rename from ansible/services/uptime_kuma/uptime_kuma_vars.yml rename to archive/uptime_kuma/uptime_kuma_vars.yml From 80c9e6f3e36dd6f10fef1055062e936d704d2ba8 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 22:43:15 +0200 Subject: [PATCH 28/67] uptime-kuma: remove credentials from the vaults MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Drops uptime_kuma_username/password from infra_secrets.yml and its group_vars copy and example, plus a dead push token in nodito_secrets.yml that nothing referenced. They remain in git history — rotation is what actually retires them. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 93 ++++++++++++------------- ansible/infra/nodito/nodito_secrets.yml | 27 +++---- ansible/infra_secrets.yml | 93 ++++++++++++------------- ansible/infra_secrets.yml.example | 2 - 4 files changed, 99 insertions(+), 116 deletions(-) diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index c48d456..798e2d8 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,50 +1,45 @@ $ANSIBLE_VAULT;1.1;AES256 -32643038333165373363353439313639326538343435393633343765386662643231333936386630 -6564313061366166353662373564393261396666663432610a633161643066663232336163306663 -61623535643231366638616630666562333936383232373130623631306334653634386132396431 -3036383735316339650a313565313062636461396563336461633838343037666237623038633133 -35333531313036666531633230646464636631666133313636353362646336623135363736316562 -36656564396230333231643266303833633134303862323633636136363136373836346363663364 -32306234393863613534663031646632643734366531343339386133363837396138616435313365 -65353037326266386538633665386638656539376133636435346136643163643238656434346132 -34363036623935343162313031343036393536613136306435323730346561393261333966663465 -35333864323638356565316632353662366262363962376134366432333361343639633133626533 -65303964643330393264643136346432313164396664333535363135646464326539626433623732 -34666530666234376637623930633433656163353136646134326365353563343833666161666435 -30313866396136623036613933376164346533383964303664633830376636363565323832343136 -62653137366632313133626234386235356630383433623934303737653438623136366638396237 -66303435356539343061313337646261343738373866646465373339616161323531346361383363 -64643363616130323935656264326631336231323236616438366632356333333534303861663238 -61376464623030363938653230316265366533623364653930343537326532393433363230356136 -62643135636130626436386239366561393537333736383361336162636434656339653630353732 -37343265366234656637336135633461333735363932636130323564623661303730303363383833 -34373963373765643437303534366233626662656531373261633565353137643761356135376430 -30373639336565356364313333333237323338346631666235613961656336643438653834313137 -35303666366433646334643264633837343134373637643535633864313062613336333538303931 -34396266326166666165333639363961346639343130656235323332346631643539363861666436 -63343733656362313165623130383838376138613534363535336662366564306132353362666463 -61613534613431623263323831646635323362343634616566393834353964643861633639346132 -66363233373964323139313966616365343535306365373532313063383236353964343265383761 -64663833363035363736643236373634663466666138346266643930643033343839313534376666 -37363461623965636164663339386237643635353237353263633131363434643066393538363639 -39346530366361656536383233663766616234383535643234616233666262353364633036663062 -64386663383363633365656631326332353261303562346666636565313564306635333562363930 -33306237333638323131613161336165636237623332643137363038303330313663346639326361 -39383565366431363361313361303436343761626462306537633337393636333135343833623464 -32323663616166653063303861396664613434643536633066626466303634663031623938313436 -32303763616562376137303061633864393863373334373338626566653633383132306635396566 -64616463663265643732366465666662663334666362383036326330356535333465326534363266 -33326265326365353231613839383336656562333837376565306230646466333238323936316463 -64396266306562303438363635343865376364333239376664346534636664326532353265313563 -66636331333430616462326334623637646539643638323036303335656632363565373739373234 -35666532346364376335346238376562336164386363623862356539316466336531643936336131 -31356332306639626532646338393638373764653961616638643066313832626561336435383839 -64363232313633643133383066636539663933323064366536666462646466323166306562623538 -64393835383431663561343231303961383337656331663261393439336136623066666434336362 -36313239336537376134396361643532633437633163373931353139303534356665383538393838 -31306336326538353232336533643961383337313161663266353730643931656165353264303938 -30636363666136653963306130666433633562313464653862656262313035306635306230653666 -34336631623235323734616163623039613636393863383966336435313230313231356563636337 -61373336336563653030366130356331616438336638323263663164366435653266643161653331 -31323733313936313833386665633235633339376330373161326361336266633838326336356363 -37383165393866656565316337386639346531323439626465613834303636663933 +65343264303166346334396163363362326334626531336363363766326135393462373564313539 +6633346163343664393232363965383334303563396236360a316265623239333431636133306663 +65623237393564323936303936633036343137646637633963396166313462626165616463616665 +3235373562626238640a636163623535376565373835653831666361636563616230353639626131 +38393530363461346334356437613934306330666563326661343465653362313038346535386337 +62623964633764343964653061373137366539326433396231396339313231633632353465323139 +33333866383834376165303964353639386234646535626333363932333764646137663036303530 +32326635343631623535323830333934376466323338353234323464666435383436633139343037 +62666333383231356163383530353339336230373431616665333638633234656231393166303639 +32386134386264613839313564316164336461373738646630636630363731643066633132336636 +64663532663661653539653431353763303436323234383864353064333631643435633733366435 +32656332626633313134623361343965643962663134613363626265353865643738363738316232 +62356136623166303038653830393138646637636335366362323239316333616331383765663861 +33643331323766646463663862396232306233343134326436326235336165616662383636616539 +39373962336264636363303734343666646534343733666632626338393936393032626130643233 +30633963656166343834396431323233363036353165623934336532333662323932336630636235 +63326631366234383732316464383364316337393034303433646365343763633135336336376161 +33326436303863326666616234396438623864333538313363623261353962336430316132343830 +36353761326133666166393930343233643861366138363930623362336566386230316236353365 +65663632383436333536636161393133316334376262653330303238636431353966646666663665 +30386137623066343135616534323336313162643066636330323934646162303362373838313532 +61386163356537613334616537656134373835326265393765303436396233383239393365336136 +64623130383538306331323939386634316663663263306362363830383930346232313965613035 +33353235663230663635316136666335343233646539613832643431393134303865616232636435 +66366437353863336235646438353431616465396365373562636461623938303663326461656134 +66353735343237363164326236663363376333316163353462626234393861663535323463623233 +66623934336662323830616362653035353563343564636431383837643961303637366131633132 +33376133376465623030643935643737636339333236316566636663323934626265633863343963 +38626563323166343264643465626264393838643330393739386330656461656561306539653232 +63393838626233386531353063366131366237633162393537363963633361386238643232336230 +36363062386262303364303736373061316265373730366437326166336163386335623665323234 +35383865646466363534383833653764303833323230613863313735393264336132303934633465 +65616138636134613361316539343739353735666538653438333264623034306633663634306438 +33643165353534623039626536636433373963383866303535333931633633393138663464336562 +36643432366233323235613139626236346135633237613265343933646434373939396265626366 +34313835346463303132326537323166363032306133623662343736393435383330356433306138 +65653931613832313832323936353638303937613863646430633266306330656135386533306462 +30663366396566393134653663356531316639303635343236666333363637636433336533356663 +35346261363134326165316463626439306130653165636134616434663736616666643963336464 +61643834613164336234386237393333636234306162306563363430646262363963366666653666 +64323036323731626236343361613566643565666138386631633462613836613766626562353934 +32363936626466346464356131653232326330386239666638626333346238623964383666393961 +38393264663636353161663636393733396332333838333962393439346533353332653362356230 +62613039376539636363 diff --git a/ansible/infra/nodito/nodito_secrets.yml b/ansible/infra/nodito/nodito_secrets.yml index 7983cdb..c34b1e8 100644 --- a/ansible/infra/nodito/nodito_secrets.yml +++ b/ansible/infra/nodito/nodito_secrets.yml @@ -1,17 +1,12 @@ $ANSIBLE_VAULT;1.1;AES256 -33303038356165616435663034356134363963323865356133353239373263623430373263613430 -6463323436386666633666333331636132653963663735330a333063303038313839653330303231 -37313332653033636265353730616431356337366566656533363838303939636266626565333365 -3764316366653361310a386330326335343936356430333864336431376230323734303030613032 -31343737323636363366623530353433313862663038376439343331633038363138316639613531 -63663866613037336630316464383464613165326632376337383338336432326564396261383561 -61313337336466653431613338656136653862646662323435396462623366366438626566373331 -34663433333965626161356636633930313961313531656233303638346265316232326530613062 -31343732636466393334623734346639636637363831393337363562343461616639316235336362 -62306464383263626235613533336130326232613563336630373661623637303961353863303633 -66306664653035363236323063616363313739633633373231333833393562633666356530383065 -36373830393965663861663231633838303863363763613663363734383961623466653735636138 -34376233363831363732643631646364363066626230353330316562376337633462323966376239 -30666263326263366563323739653538616462303566633037393861613664646331633264613338 -35636131383937393430666430646530663866626438643430316430376133653135333838636436 -37613465666265373336 +39386134336137333139343861353965353665626161303634323563386563356364373265616163 +3237623438636161323366313363356335613566303338630a663732316663333136653737663935 +34373466316538663962636134303530323837316563336136653636643430363566376363323666 +3531613636393331310a306633613030343138656237663061653438336562323135633732356563 +33396136656564353465636361613161326465303166373131343562363338303966666362643532 +33383561366539393930633766363533313363653733393263376634316362643863376635366638 +36396333633231616662323465323932656565396138313264613832346261616231333265636439 +61666334613662373839613833613663333436373365376534643662656335316536303739616437 +65333635613330353536353162646234323266316338643435653864666364313734386665303830 +34383961633734646134373866623330663038366130306265656466653562643764346162313333 +353664363335303433393330306332666336 diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml index c48d456..798e2d8 100644 --- a/ansible/infra_secrets.yml +++ b/ansible/infra_secrets.yml @@ -1,50 +1,45 @@ $ANSIBLE_VAULT;1.1;AES256 -32643038333165373363353439313639326538343435393633343765386662643231333936386630 -6564313061366166353662373564393261396666663432610a633161643066663232336163306663 -61623535643231366638616630666562333936383232373130623631306334653634386132396431 -3036383735316339650a313565313062636461396563336461633838343037666237623038633133 -35333531313036666531633230646464636631666133313636353362646336623135363736316562 -36656564396230333231643266303833633134303862323633636136363136373836346363663364 -32306234393863613534663031646632643734366531343339386133363837396138616435313365 -65353037326266386538633665386638656539376133636435346136643163643238656434346132 -34363036623935343162313031343036393536613136306435323730346561393261333966663465 -35333864323638356565316632353662366262363962376134366432333361343639633133626533 -65303964643330393264643136346432313164396664333535363135646464326539626433623732 -34666530666234376637623930633433656163353136646134326365353563343833666161666435 -30313866396136623036613933376164346533383964303664633830376636363565323832343136 -62653137366632313133626234386235356630383433623934303737653438623136366638396237 -66303435356539343061313337646261343738373866646465373339616161323531346361383363 -64643363616130323935656264326631336231323236616438366632356333333534303861663238 -61376464623030363938653230316265366533623364653930343537326532393433363230356136 -62643135636130626436386239366561393537333736383361336162636434656339653630353732 -37343265366234656637336135633461333735363932636130323564623661303730303363383833 -34373963373765643437303534366233626662656531373261633565353137643761356135376430 -30373639336565356364313333333237323338346631666235613961656336643438653834313137 -35303666366433646334643264633837343134373637643535633864313062613336333538303931 -34396266326166666165333639363961346639343130656235323332346631643539363861666436 -63343733656362313165623130383838376138613534363535336662366564306132353362666463 -61613534613431623263323831646635323362343634616566393834353964643861633639346132 -66363233373964323139313966616365343535306365373532313063383236353964343265383761 -64663833363035363736643236373634663466666138346266643930643033343839313534376666 -37363461623965636164663339386237643635353237353263633131363434643066393538363639 -39346530366361656536383233663766616234383535643234616233666262353364633036663062 -64386663383363633365656631326332353261303562346666636565313564306635333562363930 -33306237333638323131613161336165636237623332643137363038303330313663346639326361 -39383565366431363361313361303436343761626462306537633337393636333135343833623464 -32323663616166653063303861396664613434643536633066626466303634663031623938313436 -32303763616562376137303061633864393863373334373338626566653633383132306635396566 -64616463663265643732366465666662663334666362383036326330356535333465326534363266 -33326265326365353231613839383336656562333837376565306230646466333238323936316463 -64396266306562303438363635343865376364333239376664346534636664326532353265313563 -66636331333430616462326334623637646539643638323036303335656632363565373739373234 -35666532346364376335346238376562336164386363623862356539316466336531643936336131 -31356332306639626532646338393638373764653961616638643066313832626561336435383839 -64363232313633643133383066636539663933323064366536666462646466323166306562623538 -64393835383431663561343231303961383337656331663261393439336136623066666434336362 -36313239336537376134396361643532633437633163373931353139303534356665383538393838 -31306336326538353232336533643961383337313161663266353730643931656165353264303938 -30636363666136653963306130666433633562313464653862656262313035306635306230653666 -34336631623235323734616163623039613636393863383966336435313230313231356563636337 -61373336336563653030366130356331616438336638323263663164366435653266643161653331 -31323733313936313833386665633235633339376330373161326361336266633838326336356363 -37383165393866656565316337386639346531323439626465613834303636663933 +65343264303166346334396163363362326334626531336363363766326135393462373564313539 +6633346163343664393232363965383334303563396236360a316265623239333431636133306663 +65623237393564323936303936633036343137646637633963396166313462626165616463616665 +3235373562626238640a636163623535376565373835653831666361636563616230353639626131 +38393530363461346334356437613934306330666563326661343465653362313038346535386337 +62623964633764343964653061373137366539326433396231396339313231633632353465323139 +33333866383834376165303964353639386234646535626333363932333764646137663036303530 +32326635343631623535323830333934376466323338353234323464666435383436633139343037 +62666333383231356163383530353339336230373431616665333638633234656231393166303639 +32386134386264613839313564316164336461373738646630636630363731643066633132336636 +64663532663661653539653431353763303436323234383864353064333631643435633733366435 +32656332626633313134623361343965643962663134613363626265353865643738363738316232 +62356136623166303038653830393138646637636335366362323239316333616331383765663861 +33643331323766646463663862396232306233343134326436326235336165616662383636616539 +39373962336264636363303734343666646534343733666632626338393936393032626130643233 +30633963656166343834396431323233363036353165623934336532333662323932336630636235 +63326631366234383732316464383364316337393034303433646365343763633135336336376161 +33326436303863326666616234396438623864333538313363623261353962336430316132343830 +36353761326133666166393930343233643861366138363930623362336566386230316236353365 +65663632383436333536636161393133316334376262653330303238636431353966646666663665 +30386137623066343135616534323336313162643066636330323934646162303362373838313532 +61386163356537613334616537656134373835326265393765303436396233383239393365336136 +64623130383538306331323939386634316663663263306362363830383930346232313965613035 +33353235663230663635316136666335343233646539613832643431393134303865616232636435 +66366437353863336235646438353431616465396365373562636461623938303663326461656134 +66353735343237363164326236663363376333316163353462626234393861663535323463623233 +66623934336662323830616362653035353563343564636431383837643961303637366131633132 +33376133376465623030643935643737636339333236316566636663323934626265633863343963 +38626563323166343264643465626264393838643330393739386330656461656561306539653232 +63393838626233386531353063366131366237633162393537363963633361386238643232336230 +36363062386262303364303736373061316265373730366437326166336163386335623665323234 +35383865646466363534383833653764303833323230613863313735393264336132303934633465 +65616138636134613361316539343739353735666538653438333264623034306633663634306438 +33643165353534623039626536636433373963383866303535333931633633393138663464336562 +36643432366233323235613139626236346135633237613265343933646434373939396265626366 +34313835346463303132326537323166363032306133623662343736393435383330356433306138 +65653931613832313832323936353638303937613863646430633266306330656135386533306462 +30663366396566393134653663356531316639303635343236666333363637636433336533356663 +35346261363134326165316463626439306130653165636134616434663736616666643963336464 +61643834613164336234386237393333636234306162306563363430646262363963366666653666 +64323036323731626236343361613566643565666138386631633462613836613766626562353934 +32363936626466346464356131653232326330386239666638626333346238623964383666393961 +38393264663636353161663636393733396332333838333962393439346533353332653362356230 +62613039376539636363 diff --git a/ansible/infra_secrets.yml.example b/ansible/infra_secrets.yml.example index d539282..c95234d 100644 --- a/ansible/infra_secrets.yml.example +++ b/ansible/infra_secrets.yml.example @@ -1,8 +1,6 @@ # Uptime Kuma login credentials # Used by the disk monitoring playbook to create monitors automatically -uptime_kuma_username: "admin" -uptime_kuma_password: "your_password_here" # ntfy credentials # Used for notification channel setup in Uptime Kuma From 2dabc6ad62bff830005c83fe71da531b59c593d2 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 22:43:55 +0200 Subject: [PATCH 29/67] uptime-kuma: add uptime_kuma_enabled flag and make monitoring blocks inert MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Removing the credentials would otherwise break these playbooks mid-deploy: they template uptime_kuma_password with no assert to stop them first. 100 tasks are now guarded by uptime_kuma_enabled (false), so deployments run normally and the monitoring sections skip. A further 28 tasks were already self-guarding on monitor_setup/push_url being defined; verified that a skipped task's registered variable makes those skip cleanly rather than error. The blocks are kept on purpose — the health-check logic is the durable part and should be rewired to whatever replaces Uptime Kuma. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/main.yml | 5 ++++ ansible/infra_vars.yml | 5 ++++ .../deploy_bitcoin_knots_playbook.yml | 22 +++++++++++++++ .../deploy_datum_gateway_playbook.yml | 28 +++++++++++++++++++ .../deploy_forgejo_runner_playbook.yml | 17 +++++++++++ .../forgejo/deploy_forgejo_playbook.yml | 14 ++++++++++ .../fulcrum/deploy_fulcrum_playbook.yml | 22 +++++++++++++++ .../headscale/deploy_headscale_playbook.yml | 14 ++++++++++ .../lnbits/deploy_lnbits_playbook.yml | 14 ++++++++++ .../services/memos/deploy_memos_playbook.yml | 15 ++++++++++ .../mempool/deploy_mempool_playbook.yml | 25 +++++++++++++++++ .../deploy_ntfy_emergency_app_playbook.yml | 14 ++++++++++ .../deploy_personal_blog_playbook.yml | 14 ++++++++++ .../phoenixd/deploy_phoenixd_playbook.yml | 17 +++++++++++ .../deploy_vaultwarden_playbook.yml | 14 ++++++++++ 15 files changed, 240 insertions(+) diff --git a/ansible/group_vars/all/main.yml b/ansible/group_vars/all/main.yml index 952df93..36d35f8 100644 --- a/ansible/group_vars/all/main.yml +++ b/ansible/group_vars/all/main.yml @@ -2,3 +2,8 @@ new_user: counterweight ssh_port: 22 allow_ssh_from: "any" root_domain: contrapeso.xyz + +# Uptime Kuma was decommissioned on 2026-09-11. The monitoring blocks in the +# playbooks are kept deliberately — the check logic is meant to be rewired to +# whatever replaces it. This flag keeps them inert until then. See archive/uptime_kuma/. +uptime_kuma_enabled: false diff --git a/ansible/infra_vars.yml b/ansible/infra_vars.yml index 952df93..36d35f8 100644 --- a/ansible/infra_vars.yml +++ b/ansible/infra_vars.yml @@ -2,3 +2,8 @@ new_user: counterweight ssh_port: 22 allow_ssh_from: "any" root_domain: contrapeso.xyz + +# Uptime Kuma was decommissioned on 2026-09-11. The monitoring blocks in the +# playbooks are kept deliberately — the check logic is meant to be rewired to +# whatever replaces it. This flag keeps them inert until then. See archive/uptime_kuma/. +uptime_kuma_enabled: false diff --git a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml index f84d627..fcc02b4 100644 --- a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +++ b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml @@ -439,7 +439,18 @@ debug: msg: "Bitcoin Knots RPC is {{ 'available' if rpc_check.status == 200 else 'not yet available' }}" + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create Bitcoin Knots health check and push script + when: uptime_kuma_enabled | default(false) copy: dest: /usr/local/bin/bitcoin-knots-healthcheck-push.sh content: | @@ -534,6 +545,7 @@ mode: '0644' - name: Create systemd service for Bitcoin Knots health check + when: uptime_kuma_enabled | default(false) copy: dest: /etc/systemd/system/bitcoin-knots-healthcheck.service content: | @@ -566,6 +578,7 @@ state: started - name: Create Uptime Kuma push monitor setup script for Bitcoin Knots + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -655,6 +668,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -667,6 +681,7 @@ mode: '0644' - name: Run Uptime Kuma push monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_bitcoin_knots_monitor.py delegate_to: localhost become: no @@ -707,6 +722,7 @@ when: uptime_kuma_push_url | default('') != '' - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: @@ -718,6 +734,7 @@ handlers: - name: Restart bitcoind + when: uptime_kuma_enabled | default(false) systemd: name: bitcoind state: restarted @@ -794,6 +811,7 @@ ignore_errors: yes - name: Display public endpoint + when: uptime_kuma_enabled | default(false) debug: msg: "Bitcoin P2P public endpoint: {{ ansible_host }}:{{ bitcoin_p2p_port }}" @@ -801,6 +819,7 @@ # Uptime Kuma TCP Monitor for Public P2P # =========================================== - name: Create Uptime Kuma TCP monitor setup script for Bitcoin P2P + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -873,6 +892,7 @@ mode: '0755' - name: Create temporary config for TCP monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -887,6 +907,7 @@ mode: '0644' - name: Run Uptime Kuma TCP monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_bitcoin_p2p_tcp_monitor.py delegate_to: localhost become: no @@ -900,6 +921,7 @@ when: tcp_monitor_setup.stdout is defined - name: Clean up TCP monitor temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index 5fa360c..3fa36d4 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -206,7 +206,18 @@ # =========================================== # Health Check Script + Systemd Timer # =========================================== + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create DATUM Gateway health check script + when: uptime_kuma_enabled | default(false) copy: dest: /usr/local/bin/datum-gateway-healthcheck-push.sh content: | @@ -287,6 +298,7 @@ daemon_reload: yes - name: Enable and start datum-gateway health check timer + when: uptime_kuma_enabled | default(false) systemd: name: datum-gateway-healthcheck.timer enabled: yes @@ -296,6 +308,7 @@ # Uptime Kuma Push Monitor Setup # =========================================== - name: Create Uptime Kuma push monitor setup script for DATUM Gateway + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -396,6 +409,7 @@ mode: "0755" - name: Create temporary config for push monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -408,6 +422,7 @@ mode: "0644" - name: Run Uptime Kuma push monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_datum_gateway_monitor.py delegate_to: localhost become: no @@ -421,6 +436,7 @@ when: monitor_setup.stdout is defined - name: Read push URL from file + when: uptime_kuma_enabled | default(false) slurp: src: /tmp/datum_gateway_push_url.txt delegate_to: localhost @@ -442,6 +458,7 @@ notify: Restart datum-gateway health check timer - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: @@ -460,6 +477,7 @@ daemon_reload: yes - name: Restart datum-gateway health check timer + when: uptime_kuma_enabled | default(false) systemd: name: datum-gateway-healthcheck.timer state: restarted @@ -539,6 +557,7 @@ msg: "{{ caddy_reload.stdout_lines + caddy_reload.stderr_lines }}" - name: Display DATUM Gateway dashboard URL + when: uptime_kuma_enabled | default(false) debug: msg: "DATUM Gateway dashboard: https://{{ datum_gateway_domain }}" @@ -546,6 +565,7 @@ # Uptime Kuma HTTP Monitor for Public Dashboard # =========================================== - name: Create Uptime Kuma HTTP monitor setup script for DATUM dashboard + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -621,6 +641,7 @@ mode: "0755" - name: Create temporary config for HTTP monitor + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -634,6 +655,7 @@ mode: "0644" - name: Run Uptime Kuma HTTP monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_datum_http_monitor.py delegate_to: localhost become: no @@ -647,6 +669,7 @@ when: http_monitor_setup.stdout is defined - name: Clean up HTTP monitor temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: @@ -733,6 +756,7 @@ ignore_errors: yes - name: Display public Stratum endpoint + when: uptime_kuma_enabled | default(false) debug: msg: "DATUM Stratum public endpoint: {{ ansible_host }}:{{ datum_gateway_stratum_port }}" @@ -740,6 +764,7 @@ # Uptime Kuma TCP Monitor for Public Stratum # =========================================== - name: Create Uptime Kuma TCP monitor setup script for Stratum + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -816,6 +841,7 @@ mode: "0755" - name: Create temporary config for Stratum TCP monitor + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -830,6 +856,7 @@ mode: "0644" - name: Run Uptime Kuma TCP monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_datum_stratum_tcp_monitor.py delegate_to: localhost become: no @@ -843,6 +870,7 @@ when: tcp_monitor_setup.stdout is defined - name: Clean up Stratum TCP monitor temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml index d94f73c..bdc8428 100644 --- a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml +++ b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml @@ -147,7 +147,18 @@ register: runner_active changed_when: false + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Assert runner is running + when: uptime_kuma_enabled | default(false) assert: that: - runner_active.stdout == "active" @@ -155,6 +166,7 @@ # ── 10. Set up Uptime Kuma push monitor ──────────────────────────── - name: Create Uptime Kuma push monitor setup script + when: uptime_kuma_enabled | default(false) copy: dest: /tmp/setup_forgejo_runner_monitor.py content: | @@ -238,6 +250,7 @@ become: no - name: Run Uptime Kuma push monitor setup + when: uptime_kuma_enabled | default(false) command: > {{ ansible_playbook_python }} /tmp/setup_forgejo_runner_monitor.py @@ -256,10 +269,12 @@ changed_when: false - name: Parse monitor setup result + when: uptime_kuma_enabled | default(false) set_fact: monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - name: Set push URL + when: uptime_kuma_enabled | default(false) set_fact: uptime_kuma_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" @@ -272,6 +287,7 @@ mode: '0755' - name: Create forgejo-runner healthcheck script + when: uptime_kuma_enabled | default(false) copy: dest: "{{ healthcheck_script_path }}" content: | @@ -385,6 +401,7 @@ Timeout: {{ healthcheck_timeout_seconds }}s - name: Clean up temporary monitor setup script + when: uptime_kuma_enabled | default(false) file: path: /tmp/setup_forgejo_runner_monitor.py state: absent diff --git a/ansible/services/forgejo/deploy_forgejo_playbook.yml b/ansible/services/forgejo/deploy_forgejo_playbook.yml index ff865de..acb1473 100644 --- a/ansible/services/forgejo/deploy_forgejo_playbook.yml +++ b/ansible/services/forgejo/deploy_forgejo_playbook.yml @@ -118,7 +118,18 @@ - name: Reload Caddy to apply new config command: systemctl reload caddy + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create Uptime Kuma monitor setup script for Forgejo + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -193,6 +204,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -206,6 +218,7 @@ mode: '0644' - name: Run Uptime Kuma monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_forgejo_monitor.py delegate_to: localhost become: no @@ -214,6 +227,7 @@ ignore_errors: yes - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml index 6b17280..a1548a7 100644 --- a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml +++ b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml @@ -259,7 +259,18 @@ debug: msg: "Fulcrum service is {{ 'running' if fulcrum_service_status.status.ActiveState == 'active' else 'not running' }}" + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create Fulcrum health check and push script + when: uptime_kuma_enabled | default(false) copy: dest: /usr/local/bin/fulcrum-healthcheck-push.sh content: | @@ -332,6 +343,7 @@ mode: '0644' - name: Create systemd service for Fulcrum health check + when: uptime_kuma_enabled | default(false) copy: dest: /etc/systemd/system/fulcrum-healthcheck.service content: | @@ -364,6 +376,7 @@ state: started - name: Create Uptime Kuma push monitor setup script for Fulcrum + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -453,6 +466,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -465,6 +479,7 @@ mode: '0644' - name: Run Uptime Kuma push monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_fulcrum_monitor.py delegate_to: localhost become: no @@ -505,6 +520,7 @@ when: uptime_kuma_push_url | default('') != '' - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: @@ -518,6 +534,7 @@ handlers: - name: Restart fulcrum + when: uptime_kuma_enabled | default(false) systemd: name: fulcrum state: restarted @@ -593,6 +610,7 @@ ignore_errors: yes - name: Display public endpoint + when: uptime_kuma_enabled | default(false) debug: msg: "Fulcrum SSL public endpoint: {{ ansible_host }}:{{ fulcrum_ssl_port }}" @@ -600,6 +618,7 @@ # Uptime Kuma TCP Monitor for Public SSL Port # =========================================== - name: Create Uptime Kuma TCP monitor setup script for Fulcrum SSL + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -672,6 +691,7 @@ mode: '0755' - name: Create temporary config for TCP monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -686,6 +706,7 @@ mode: '0644' - name: Run Uptime Kuma TCP monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_fulcrum_ssl_tcp_monitor.py delegate_to: localhost become: no @@ -699,6 +720,7 @@ when: tcp_monitor_setup.stdout is defined - name: Clean up TCP monitor temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/headscale/deploy_headscale_playbook.yml b/ansible/services/headscale/deploy_headscale_playbook.yml index 7ed81ed..61ba22e 100644 --- a/ansible/services/headscale/deploy_headscale_playbook.yml +++ b/ansible/services/headscale/deploy_headscale_playbook.yml @@ -264,7 +264,18 @@ become: yes command: systemctl reload caddy + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create Uptime Kuma monitor setup script for Headscale + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -339,6 +350,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -352,6 +364,7 @@ mode: '0644' - name: Run Uptime Kuma monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_headscale_monitor.py delegate_to: localhost become: no @@ -360,6 +373,7 @@ ignore_errors: yes - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/lnbits/deploy_lnbits_playbook.yml b/ansible/services/lnbits/deploy_lnbits_playbook.yml index 6df63f5..d596052 100644 --- a/ansible/services/lnbits/deploy_lnbits_playbook.yml +++ b/ansible/services/lnbits/deploy_lnbits_playbook.yml @@ -181,7 +181,18 @@ - name: Reload Caddy to apply new config command: systemctl reload caddy + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create Uptime Kuma monitor setup script for LNBits + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -256,6 +267,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -269,6 +281,7 @@ mode: '0644' - name: Run Uptime Kuma monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_lnbits_monitor.py delegate_to: localhost become: no @@ -277,6 +290,7 @@ ignore_errors: yes - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/memos/deploy_memos_playbook.yml b/ansible/services/memos/deploy_memos_playbook.yml index a823135..5ab254e 100644 --- a/ansible/services/memos/deploy_memos_playbook.yml +++ b/ansible/services/memos/deploy_memos_playbook.yml @@ -139,7 +139,18 @@ msg: "Memos is running on port {{ memos_port }}. Access via Tailscale at http://{{ memos_tailscale_hostname }}:{{ memos_port }}" handlers: + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Restart memos + when: uptime_kuma_enabled | default(false) systemd: name: memos state: restarted @@ -196,6 +207,7 @@ command: systemctl reload caddy - name: Create Uptime Kuma monitor setup script for Memos + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -275,6 +287,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -288,6 +301,7 @@ mode: '0644' - name: Run Uptime Kuma monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_memos_monitor.py delegate_to: localhost become: no @@ -296,6 +310,7 @@ ignore_errors: yes - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/mempool/deploy_mempool_playbook.yml b/ansible/services/mempool/deploy_mempool_playbook.yml index c0fe51d..1dfea1b 100644 --- a/ansible/services/mempool/deploy_mempool_playbook.yml +++ b/ansible/services/mempool/deploy_mempool_playbook.yml @@ -226,7 +226,18 @@ delay: 5 ignore_errors: yes + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Display deployment status + when: uptime_kuma_enabled | default(false) debug: msg: - "Mempool deployment complete!" @@ -239,6 +250,7 @@ # Health Check Scripts for Uptime Kuma Push Monitors # =========================================== - name: Create Mempool MariaDB health check script + when: uptime_kuma_enabled | default(false) copy: dest: /usr/local/bin/mempool-mariadb-healthcheck-push.sh content: | @@ -273,6 +285,7 @@ mode: '0755' - name: Create Mempool backend health check script + when: uptime_kuma_enabled | default(false) copy: dest: /usr/local/bin/mempool-backend-healthcheck-push.sh content: | @@ -307,6 +320,7 @@ mode: '0755' - name: Create Mempool frontend health check script + when: uptime_kuma_enabled | default(false) copy: dest: /usr/local/bin/mempool-frontend-healthcheck-push.sh content: | @@ -396,6 +410,7 @@ daemon_reload: yes - name: Enable and start health check timers + when: uptime_kuma_enabled | default(false) systemd: name: "mempool-{{ item }}-healthcheck.timer" enabled: yes @@ -409,6 +424,7 @@ # Uptime Kuma Push Monitor Setup # =========================================== - name: Create Uptime Kuma push monitor setup script for Mempool + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -492,6 +508,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -507,6 +524,7 @@ mode: '0644' - name: Run Uptime Kuma push monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_mempool_monitors.py delegate_to: localhost become: no @@ -520,6 +538,7 @@ when: monitor_setup.stdout is defined - name: Read push URLs from file + when: uptime_kuma_enabled | default(false) slurp: src: /tmp/mempool_push_urls.yml delegate_to: localhost @@ -573,6 +592,7 @@ when: push_urls is defined - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: @@ -637,6 +657,7 @@ state: reloaded - name: Display Mempool URL + when: uptime_kuma_enabled | default(false) debug: msg: "Mempool is now available at https://{{ mempool_domain }}" @@ -644,6 +665,7 @@ # Uptime Kuma HTTP Monitor for Public Endpoint # =========================================== - name: Create Uptime Kuma HTTP monitor setup script for Mempool + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -714,6 +736,7 @@ mode: '0755' - name: Create temporary config for HTTP monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -727,6 +750,7 @@ mode: '0644' - name: Run Uptime Kuma HTTP monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_mempool_http_monitor.py delegate_to: localhost become: no @@ -740,6 +764,7 @@ when: http_monitor_setup.stdout is defined - name: Clean up HTTP monitor temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml index a04d6af..a8d97a6 100644 --- a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml +++ b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml @@ -79,7 +79,18 @@ - name: Reload Caddy to apply new config command: systemctl reload caddy + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create Uptime Kuma monitor setup script for ntfy-emergency-app + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -159,6 +170,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -172,6 +184,7 @@ mode: '0644' - name: Run Uptime Kuma monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_ntfy_emergency_app_monitor.py delegate_to: localhost become: no @@ -180,6 +193,7 @@ ignore_errors: yes - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml index 63bbae5..af7a1f3 100644 --- a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml +++ b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml @@ -82,7 +82,18 @@ - name: Reload Caddy to apply new config command: systemctl reload caddy + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create Uptime Kuma monitor setup script for Personal Blog + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -157,6 +168,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -170,6 +182,7 @@ mode: '0644' - name: Run Uptime Kuma monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_personal_blog_monitor.py delegate_to: localhost become: no @@ -178,6 +191,7 @@ ignore_errors: yes - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml index 9e465eb..f20ecdc 100644 --- a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml +++ b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml @@ -253,7 +253,18 @@ # =========================================== # Health Check Script + Systemd Timer # =========================================== + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create phoenixd health check script + when: uptime_kuma_enabled | default(false) copy: dest: "{{ phoenixd_healthcheck_script_path }}" content: | @@ -340,6 +351,7 @@ daemon_reload: yes - name: Enable and start phoenixd health check timer + when: uptime_kuma_enabled | default(false) systemd: name: "{{ phoenixd_healthcheck_service_name }}.timer" enabled: yes @@ -349,6 +361,7 @@ # Uptime Kuma Push Monitor Setup # =========================================== - name: Create Uptime Kuma push monitor setup script for phoenixd + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -443,6 +456,7 @@ mode: "0755" - name: Create temporary config for push monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -455,6 +469,7 @@ mode: "0644" - name: Run Uptime Kuma push monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_phoenixd_monitor.py delegate_to: localhost become: no @@ -468,6 +483,7 @@ when: monitor_setup.stdout is defined - name: Read push URL from file + when: uptime_kuma_enabled | default(false) slurp: src: /tmp/phoenixd_push_url.txt delegate_to: localhost @@ -489,6 +505,7 @@ notify: Restart phoenixd health check timer - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: diff --git a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml index 65f1f64..9868d13 100644 --- a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml +++ b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml @@ -111,7 +111,18 @@ - name: Reload Caddy to apply new config command: systemctl reload caddy + # ═════════════════════════════════════════════════════════════════════════ + # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. + # + # Every task below is inert: uptime_kuma_enabled is false in + # group_vars/all/main.yml, so they all skip and the deployment above still + # runs normally. Kept because the health-check logic is the durable part — + # when a replacement exists, rewire the push transport and flip the flag. + # + # What was being monitored: archive/uptime_kuma/MONITORS.md + # ═════════════════════════════════════════════════════════════════════════ - name: Create Uptime Kuma monitor setup script for Vaultwarden + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -191,6 +202,7 @@ mode: '0755' - name: Create temporary config for monitor setup + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no copy: @@ -204,6 +216,7 @@ mode: '0644' - name: Run Uptime Kuma monitor setup + when: uptime_kuma_enabled | default(false) command: python3 /tmp/setup_vaultwarden_monitor.py delegate_to: localhost become: no @@ -212,6 +225,7 @@ ignore_errors: yes - name: Clean up temporary files + when: uptime_kuma_enabled | default(false) delegate_to: localhost become: no file: From b5b028c3567012b94adf9c7124c89034a0507891 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 22:43:55 +0200 Subject: [PATCH 30/67] uptime-kuma: deprecation banners on the monitoring-only plays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit These five plus the ntfy notification playbook assert on the credentials, so they now fail immediately instead of running — deliberately, before anything is installed. The banner says so and points at archive/uptime_kuma/. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/infra/410_disk_usage_alerts.yml | 13 +++++++++++++ ansible/infra/420_system_healthcheck.yml | 13 +++++++++++++ ansible/infra/430_cpu_temp_alerts.yml | 13 +++++++++++++ ansible/infra/nodito/32_zfs_pool_setup_playbook.yml | 13 +++++++++++++ ansible/infra/nodito/34_nut_ups_setup_playbook.yml | 13 +++++++++++++ .../ntfy/setup_ntfy_uptime_kuma_notification.yml | 13 +++++++++++++ 6 files changed, 78 insertions(+) diff --git a/ansible/infra/410_disk_usage_alerts.yml b/ansible/infra/410_disk_usage_alerts.yml index 7905cf3..dcd9bdc 100644 --- a/ansible/infra/410_disk_usage_alerts.yml +++ b/ansible/infra/410_disk_usage_alerts.yml @@ -1,3 +1,16 @@ +# ═════════════════════════════════════════════════════════════════════════════ +# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. +# +# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username +# and uptime_kuma_password were removed from the vault, so the "Validate Uptime +# Kuma configuration" assert stops it before anything is installed or changed. +# +# It is kept because the CHECK LOGIC is the durable part — what gets measured, +# the thresholds, and the systemd timer plumbing. When something replaces Uptime +# Kuma, only the push transport needs rewriting; the rest still applies. +# +# What was being monitored: archive/uptime_kuma/MONITORS.md +# ═════════════════════════════════════════════════════════════════════════════ - name: Deploy Disk Usage Monitoring hosts: managed become: yes diff --git a/ansible/infra/420_system_healthcheck.yml b/ansible/infra/420_system_healthcheck.yml index 5a45ce1..05532a7 100644 --- a/ansible/infra/420_system_healthcheck.yml +++ b/ansible/infra/420_system_healthcheck.yml @@ -1,3 +1,16 @@ +# ═════════════════════════════════════════════════════════════════════════════ +# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. +# +# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username +# and uptime_kuma_password were removed from the vault, so the "Validate Uptime +# Kuma configuration" assert stops it before anything is installed or changed. +# +# It is kept because the CHECK LOGIC is the durable part — what gets measured, +# the thresholds, and the systemd timer plumbing. When something replaces Uptime +# Kuma, only the push transport needs rewriting; the rest still applies. +# +# What was being monitored: archive/uptime_kuma/MONITORS.md +# ═════════════════════════════════════════════════════════════════════════════ - name: Deploy System Healthcheck Monitoring hosts: managed become: yes diff --git a/ansible/infra/430_cpu_temp_alerts.yml b/ansible/infra/430_cpu_temp_alerts.yml index b06f0b6..5c9e855 100644 --- a/ansible/infra/430_cpu_temp_alerts.yml +++ b/ansible/infra/430_cpu_temp_alerts.yml @@ -1,3 +1,16 @@ +# ═════════════════════════════════════════════════════════════════════════════ +# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. +# +# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username +# and uptime_kuma_password were removed from the vault, so the "Validate Uptime +# Kuma configuration" assert stops it before anything is installed or changed. +# +# It is kept because the CHECK LOGIC is the durable part — what gets measured, +# the thresholds, and the systemd timer plumbing. When something replaces Uptime +# Kuma, only the push transport needs rewriting; the rest still applies. +# +# What was being monitored: archive/uptime_kuma/MONITORS.md +# ═════════════════════════════════════════════════════════════════════════════ - name: Deploy CPU Temperature Monitoring hosts: hypervisor become: yes diff --git a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml index 2939ce8..8cdde6a 100644 --- a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml +++ b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml @@ -171,6 +171,19 @@ msg: "ZFS pool {{ zfs_pool_name }} is not in a healthy state" when: "'ONLINE' not in final_zfs_status.stdout" +# ═════════════════════════════════════════════════════════════════════════════ +# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. +# +# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username +# and uptime_kuma_password were removed from the vault, so the "Validate Uptime +# Kuma configuration" assert stops it before anything is installed or changed. +# +# It is kept because the CHECK LOGIC is the durable part — what gets measured, +# the thresholds, and the systemd timer plumbing. When something replaces Uptime +# Kuma, only the push transport needs rewriting; the rest still applies. +# +# What was being monitored: archive/uptime_kuma/MONITORS.md +# ═════════════════════════════════════════════════════════════════════════════ - name: Setup ZFS Pool Health Monitoring and Monthly Scrubs hosts: hypervisor become: true diff --git a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml index 6b366ba..f39fb56 100644 --- a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml +++ b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml @@ -250,6 +250,19 @@ - nut-monitor +# ═════════════════════════════════════════════════════════════════════════════ +# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. +# +# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username +# and uptime_kuma_password were removed from the vault, so the "Validate Uptime +# Kuma configuration" assert stops it before anything is installed or changed. +# +# It is kept because the CHECK LOGIC is the durable part — what gets measured, +# the thresholds, and the systemd timer plumbing. When something replaces Uptime +# Kuma, only the push transport needs rewriting; the rest still applies. +# +# What was being monitored: archive/uptime_kuma/MONITORS.md +# ═════════════════════════════════════════════════════════════════════════════ - name: Setup UPS Heartbeat Monitoring with Uptime Kuma hosts: hypervisor become: true diff --git a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml index 8041905..2d3d22a 100644 --- a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml +++ b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml @@ -1,3 +1,16 @@ +# ═════════════════════════════════════════════════════════════════════════════ +# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. +# +# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username +# and uptime_kuma_password were removed from the vault, so the "Validate Uptime +# Kuma configuration" assert stops it before anything is installed or changed. +# +# It is kept because the CHECK LOGIC is the durable part — what gets measured, +# the thresholds, and the systemd timer plumbing. When something replaces Uptime +# Kuma, only the push transport needs rewriting; the rest still applies. +# +# What was being monitored: archive/uptime_kuma/MONITORS.md +# ═════════════════════════════════════════════════════════════════════════════ - name: Setup ntfy as Uptime Kuma Notification Channel hosts: monitoring become: no From 338f2ae6362cf100b04714a9ff5fc20faed1eff8 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 22:43:56 +0200 Subject: [PATCH 31/67] uptime-kuma: annotate config and drop the unused collection lucasheld.uptime_kuma was pinned but never used - every monitor was created by hand-rolled Python. The uptime subdomain stays because the deprecated blocks still template it. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/requirements.yml | 13 ++++++------- ansible/services_config.yml | 3 +++ requirements.txt | 3 +++ 3 files changed, 12 insertions(+), 7 deletions(-) diff --git a/ansible/requirements.yml b/ansible/requirements.yml index dd7eef4..16ef1fd 100644 --- a/ansible/requirements.yml +++ b/ansible/requirements.yml @@ -1,11 +1,10 @@ ---- # Ansible Galaxy Collections Requirements # Install with: ansible-galaxy collection install -r requirements.yml -collections: - # Uptime Kuma Ansible Collection - # Used by: infra/41_disk_usage_alerts.yml - # Provides modules to manage Uptime Kuma monitors programmatically - - name: lucasheld.uptime_kuma - version: ">=1.0.0" +# No collections are currently required. +# +# lucasheld.uptime_kuma was pinned here but never used — every monitor was created +# by hand-rolled Python instead. Removed 2026-09-11 along with Uptime Kuma itself. +# See archive/uptime_kuma/. +collections: [] diff --git a/ansible/services_config.yml b/ansible/services_config.yml index 1bc9db0..a20c853 100644 --- a/ansible/services_config.yml +++ b/ansible/services_config.yml @@ -5,6 +5,9 @@ subdomains: # Monitoring Services (on watchtower) ntfy: ntfy + # DEPRECATED 2026-09-11 — Uptime Kuma is decommissioned and this subdomain no + # longer resolves to anything. Kept only because the deprecated monitoring + # blocks still template it into uptime_kuma_api_url. See archive/uptime_kuma/. uptime_kuma: uptime # VPN Infrastructure (on spacey) diff --git a/requirements.txt b/requirements.txt index 972dc4f..653523e 100644 --- a/requirements.txt +++ b/requirements.txt @@ -8,4 +8,7 @@ packaging==25.0 pycparser==2.22 PyYAML==6.0.2 resolvelib==1.0.1 +# Only needed by the deprecated Uptime Kuma monitoring blocks, which are kept for +# reference but never run (uptime_kuma_enabled: false). Drop this once they are +# rewired to a replacement. See archive/uptime_kuma/. uptime-kuma-api>=1.2.1 From 8a3fddbe4967f34e43af1265bfeecb6cd22f908c Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 22:43:56 +0200 Subject: [PATCH 32/67] docs: mark Uptime Kuma as decommissioned README, both setup guides and the forgejo-runner notes now point at archive/uptime_kuma/ instead of describing a live service. Co-Authored-By: Claude Opus 5 (1M context) --- 01_infra_setup.md | 6 ++++++ 02_vps_core_services_setup.md | 6 ++++++ README.md | 2 +- ansible/services/forgejo-runner/SETUP.md | 4 +++- 4 files changed, 16 insertions(+), 2 deletions(-) diff --git a/01_infra_setup.md b/01_infra_setup.md index 6ef0978..a8ba9c0 100644 --- a/01_infra_setup.md +++ b/01_infra_setup.md @@ -162,6 +162,12 @@ Note that, by applying these playbooks, both the root user and the `counterweigh ```bash cp ansible/infra_secrets.yml.example ansible/infra_secrets.yml ``` + > **DEPRECATED (2026-09-11).** Uptime Kuma has been decommissioned. The server + > deployment was removed from this repo; what it monitored and how it was set up is + > preserved in [`archive/uptime_kuma/`](archive/uptime_kuma/). The monitoring blocks in + > the playbooks are kept but inert (`uptime_kuma_enabled: false`) so the check logic + > survives for whatever replaces it. The credentials below no longer exist in the vault. + * Edit `ansible/infra_secrets.yml` and add your Uptime Kuma credentials: ```yaml uptime_kuma_username: "admin" diff --git a/02_vps_core_services_setup.md b/02_vps_core_services_setup.md index 19cd122..5d75de9 100644 --- a/02_vps_core_services_setup.md +++ b/02_vps_core_services_setup.md @@ -49,6 +49,12 @@ Checklist: ## Uptime Kuma +> **DEPRECATED (2026-09-11).** Uptime Kuma has been decommissioned. The server +> deployment was removed from this repo; what it monitored and how it was set up is +> preserved in [`archive/uptime_kuma/`](archive/uptime_kuma/). The monitoring blocks in +> the playbooks are kept but inert (`uptime_kuma_enabled: false`) so the check logic +> survives for whatever replaces it. The credentials below no longer exist in the vault. + Uptime Kuma gets used to monitor the availability of services, keep track of their uptime and notify issues. ### Deploy diff --git a/README.md b/README.md index 80dbaf6..89ef092 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ Assumes that you've set `ansible/.vault_pass` with `chmod 600`. + Plan install + File based config + Crossbackup to Desky via rsync -* Uptime Kuma +* ~~Uptime Kuma~~ — decommissioned 2026-09-11, see `archive/uptime_kuma/` + Deployed on Vipy + Crossbackup to Desky via rsync * Vaultwarden diff --git a/ansible/services/forgejo-runner/SETUP.md b/ansible/services/forgejo-runner/SETUP.md index a66d295..5cb2a3f 100644 --- a/ansible/services/forgejo-runner/SETUP.md +++ b/ansible/services/forgejo-runner/SETUP.md @@ -25,4 +25,6 @@ ansible-playbook ansible/services/forgejo-runner/deploy_forgejo_runner_playbook. 1. On the VM: `systemctl status forgejo-runner` should show active 2. In Forgejo: **Site Administration** > **Actions** > **Runners** should show the runner as online -3. In Uptime Kuma: the `forgejo-runner-healthcheck` push monitor should be receiving pings +3. ~~In Uptime Kuma: the `forgejo-runner-healthcheck` push monitor should be receiving pings~~ + *(Uptime Kuma was decommissioned 2026-09-11 — this check no longer applies. The + healthcheck timer still runs on the host but pushes nowhere. See `archive/uptime_kuma/`.)* From b07ed72a92852a8da35d86f52db98f7160b361fb Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 23:03:25 +0200 Subject: [PATCH 33/67] caddy: add a [caddy] role group and target it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit services/caddy_playbook.yml was the one play still targeting a location group (vps) rather than a role group. The two coincide today — vps is exactly vipy, watchtower and spacey, the three hosts with /etc/caddy/sites-enabled — but adding a fourth VPS that does not run Caddy would have silently pulled it into the play. [caddy:children] is edge + monitoring + vpn_control. Verified the play selects the same three machines before and after. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/inventory.ini | 8 +++++++- ansible/services/caddy_playbook.yml | 2 +- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/ansible/inventory.ini b/ansible/inventory.ini index a061637..4643dbb 100644 --- a/ansible/inventory.ini +++ b/ansible/inventory.ini @@ -58,4 +58,10 @@ localhost [managed:children] vps nodito_host -nodito_vms \ No newline at end of file +nodito_vms + +# Hosts that run Caddy and therefore have /etc/caddy/sites-enabled. +[caddy:children] +edge +monitoring +vpn_control diff --git a/ansible/services/caddy_playbook.yml b/ansible/services/caddy_playbook.yml index de98c8f..29e74b2 100644 --- a/ansible/services/caddy_playbook.yml +++ b/ansible/services/caddy_playbook.yml @@ -1,5 +1,5 @@ - name: Install and configure Caddy on Debian 12 - hosts: vps + hosts: caddy become: yes tasks: From cc9340b7cc0801c4dd080a82a4044dcfeacae439 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 23:10:43 +0200 Subject: [PATCH 34/67] caddy: add the caddy_site role Replaces the four-task Caddy vhost block currently copy-pasted into 10 playbooks. Nothing calls it yet; this commit only adds the role. Verified by rendering all 10 sites through the template and diffing against what the current playbooks produce: 9 of 10 byte-identical. The tenth is datum-gateway, where the resolvers comment is standardised, rewriting one comment line Caddy ignores. Then dry-run against the live hosts (--check, nothing written): - vipy: forgejo, vaultwarden, lnbits, personal-blog, ntfy-emergency-app all report ok/unchanged against the real files - watchtower: ntfy renders identical via caddy_site_body, blank line and {host}{uri} placeholders intact - spacey: headscale renders identical when given the config that is actually running - memos, mempool, datum-gateway report changed - the comment, as expected All 14 site files on all 3 hosts confirmed unchanged afterwards. Two things the build turned up: - Ansible does not template dict *keys*, so caddy_site_basic_auth is a list of {user, hash}. As a dict, a Jinja username passes through literally. The assert refuses a mapping. - `caddy validate` does accept a single site fragment - rc=0 on a good one, rc=1 with a line number on a broken one. This was the plan's one untested claim. A failed validate leaves the live file untouched. The reload is now a handler, so it fires once at end of play rather than immediately; anything needing the new config live mid-play must flush_handlers first. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/roles/caddy_site/README.md | 94 +++++++++++++++++++ ansible/roles/caddy_site/defaults/main.yml | 23 +++++ ansible/roles/caddy_site/handlers/main.yml | 8 ++ ansible/roles/caddy_site/tasks/main.yml | 46 +++++++++ .../roles/caddy_site/templates/site.conf.j2 | 34 +++++++ 5 files changed, 205 insertions(+) create mode 100644 ansible/roles/caddy_site/README.md create mode 100644 ansible/roles/caddy_site/defaults/main.yml create mode 100644 ansible/roles/caddy_site/handlers/main.yml create mode 100644 ansible/roles/caddy_site/tasks/main.yml create mode 100644 ansible/roles/caddy_site/templates/site.conf.j2 diff --git a/ansible/roles/caddy_site/README.md b/ansible/roles/caddy_site/README.md new file mode 100644 index 0000000..6891070 --- /dev/null +++ b/ansible/roles/caddy_site/README.md @@ -0,0 +1,94 @@ +# `caddy_site` + +Writes one Caddy site file into `{{ caddy_sites_dir }}`, makes sure the main +Caddyfile imports that directory, validates the result, and reloads Caddy once. + +Replaces the four-task block that was copy-pasted into 10 playbooks. + +Runs on any host in the `[caddy]` group — `edge` (vipy), `monitoring` +(watchtower) and `vpn_control` (spacey). + +## Usage + +```yaml +- ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: forgejo # -> forgejo.conf + caddy_site_domain: "{{ forgejo_domain }}" + caddy_site_upstream: "localhost:{{ forgejo_port }}" +``` + +Use `include_role`, not a `roles:` block, so the call stays in task order next +to the tasks it depends on. Variables passed this way are scoped to the include +and do not leak into later calls — so **every call must pass everything it +needs**; nothing carries over. + +## Shapes + +Pick exactly one of `caddy_site_upstream`, `caddy_site_root`, `caddy_site_body`. + +| Want | Set | +|---|---| +| `reverse_proxy host:port` | `caddy_site_upstream` | +| static `root *` + `file_server` | `caddy_site_root` | +| anything else | `caddy_site_body` (raw, indented 4 for you) | + +`caddy_site_upstream` accepts two modifiers, which add a block to the +`reverse_proxy`: + +- `caddy_site_headers_up: {"X-Forwarded-Host": "..."}` +- `caddy_site_resolvers: "100.100.100.100"` — Tailscale MagicDNS + +and `caddy_site_basic_auth` wraps the site in a `basic_auth` block. + +## `caddy_site_basic_auth` is a LIST, not a dict + +```yaml +caddy_site_basic_auth: + - user: "{{ datum_dashboard_username }}" + hash: "{{ datum_dashboard_password_hash }}" +``` + +**Ansible does not template dictionary keys.** With `{ "{{ user }}": "hash" }` +the value is rendered and the key is not, so the literal string +`{{ datum_dashboard_username }}` lands in the config file. Found while building +this role; the `assert` refuses a mapping so it cannot happen again. + +## Secrets and `--diff` + +Rendered site files can carry credentials — `datum-gateway.conf` holds a bcrypt +hash — and `--diff` prints rendered content. The template task therefore sets +`diff: "{{ caddy_site_reveal | bool }}"`, default `false`, so `--diff` runs are +safe everywhere. Pass `-e caddy_site_reveal=true` to see what moved on a site +you know is not secret. + +## Validation + +`validate: "caddy validate --adapter caddyfile --config %s"` runs against the +rendered temp file before it is moved into place. Verified on vipy that a single +site fragment validates cleanly (rc=0, `Valid configuration`) and that a +malformed one is rejected (rc=1, with the syntax error and line number). A +failed validate leaves the live file untouched, so a broken config can no longer +reach a running Caddy. + +What it cannot catch is a conflict with the global `/etc/caddy/Caddyfile`. + +## The reload is a handler + +`Reload caddy` fires **once, at the end of the play**, however many sites +notified it. The code this replaced ran `command: systemctl reload caddy` +immediately, mid-play. If a later task in the same play needs the new config to +be live, flush first: + +```yaml +- ansible.builtin.meta: flush_handlers +``` + +## Known intentional difference + +The `resolvers` block is commented `# Use Tailscale MagicDNS to resolve the +upstream hostname` in every case. `datum-gateway` previously said `# Resolve via +Tailscale MagicDNS`. Migrating it therefore rewrites one comment line, which +Caddy ignores. Every other site renders byte-identical to what its playbook +produced. diff --git a/ansible/roles/caddy_site/defaults/main.yml b/ansible/roles/caddy_site/defaults/main.yml new file mode 100644 index 0000000..cafa3dc --- /dev/null +++ b/ansible/roles/caddy_site/defaults/main.yml @@ -0,0 +1,23 @@ +--- +# Required +caddy_site_name: "" # file basename -> .conf +caddy_site_domain: "" # site address line; may hold several, comma separated + +# Pick exactly one shape +caddy_site_upstream: "" # "localhost:3000" -> reverse_proxy +caddy_site_root: "" # filesystem path -> root * + file_server +caddy_site_body: "" # raw escape hatch for one-off sites; wins over both + +# reverse_proxy modifiers +caddy_site_resolvers: "" # "100.100.100.100" for Tailscale MagicDNS +caddy_site_headers_up: {} # {"X-Forwarded-Host": "wallet.example.com"} +# A LIST, not a dict: Ansible does not template dict *keys*, so a Jinja +# expression for the username silently passes through as literal text. +caddy_site_basic_auth: [] # [{user: "{{ x_user }}", hash: "{{ x_hash }}"}] + +# Placement. caddy_sites_dir comes from services_config.yml; this is the fallback. +caddy_sites_dir: /etc/caddy/sites-enabled + +# Rendered site files can carry credentials (basic_auth hashes), so --diff is +# suppressed by default. Pass -e caddy_site_reveal=true to see what moved. +caddy_site_reveal: false diff --git a/ansible/roles/caddy_site/handlers/main.yml b/ansible/roles/caddy_site/handlers/main.yml new file mode 100644 index 0000000..fb57280 --- /dev/null +++ b/ansible/roles/caddy_site/handlers/main.yml @@ -0,0 +1,8 @@ +--- +# Fires once at the end of the play, however many sites notified it. +# Anything later in the same play that needs the new config live must be +# preceded by `- ansible.builtin.meta: flush_handlers`. +- name: Reload caddy + ansible.builtin.systemd: + name: caddy + state: reloaded diff --git a/ansible/roles/caddy_site/tasks/main.yml b/ansible/roles/caddy_site/tasks/main.yml new file mode 100644 index 0000000..5ad3976 --- /dev/null +++ b/ansible/roles/caddy_site/tasks/main.yml @@ -0,0 +1,46 @@ +--- +- name: Assert caddy_site parameters are sane + ansible.builtin.assert: + that: + - caddy_site_name | length > 0 + - caddy_site_domain | length > 0 + - (caddy_site_upstream | length > 0) or (caddy_site_root | length > 0) or (caddy_site_body | length > 0) + - caddy_site_basic_auth is not mapping + fail_msg: >- + caddy_site: '{{ caddy_site_name | default("") }}' needs a name, a domain and + one of caddy_site_upstream / caddy_site_root / caddy_site_body. + caddy_site_basic_auth must be a LIST of {user, hash} — Ansible does not template dict keys. + quiet: true + +- name: Ensure Caddy sites-enabled directory exists + ansible.builtin.file: + path: "{{ caddy_sites_dir }}" + state: directory + owner: root + group: root + mode: '0755' + +- name: Ensure Caddyfile imports sites-enabled + ansible.builtin.lineinfile: + path: /etc/caddy/Caddyfile + line: 'import sites-enabled/*' + insertafter: EOF + state: present + create: yes + mode: '0644' + backup: yes + +# `validate` runs `caddy validate` against the rendered temp file before it is +# moved into place: verified on vipy that a single site fragment validates +# cleanly (rc=0, "Valid configuration") and that a malformed one is rejected +# (rc=1). A failed validate leaves the live file untouched. +- name: "Write Caddy site '{{ caddy_site_name }}'" + ansible.builtin.template: + src: site.conf.j2 + dest: "{{ caddy_sites_dir }}/{{ caddy_site_name }}.conf" + owner: root + group: root + mode: '0644' + validate: "caddy validate --adapter caddyfile --config %s" + diff: "{{ caddy_site_reveal | bool }}" + notify: Reload caddy diff --git a/ansible/roles/caddy_site/templates/site.conf.j2 b/ansible/roles/caddy_site/templates/site.conf.j2 new file mode 100644 index 0000000..5d34c01 --- /dev/null +++ b/ansible/roles/caddy_site/templates/site.conf.j2 @@ -0,0 +1,34 @@ +{{ caddy_site_domain }} { +{% if caddy_site_body %} +{{ caddy_site_body | trim | indent(4, first=True) }} +{% else %} +{% if caddy_site_basic_auth %} + basic_auth { +{% for cred in caddy_site_basic_auth %} + {{ cred.user }} {{ cred.hash }} +{% endfor %} + } +{% endif %} +{% if caddy_site_root %} + root * {{ caddy_site_root }} + file_server +{% endif %} +{% if caddy_site_upstream %} +{% if caddy_site_headers_up or caddy_site_resolvers %} + reverse_proxy {{ caddy_site_upstream }} { +{% for key, value in caddy_site_headers_up.items() %} + header_up {{ key }} {{ value }} +{% endfor %} +{% if caddy_site_resolvers %} + # Use Tailscale MagicDNS to resolve the upstream hostname + transport http { + resolvers {{ caddy_site_resolvers }} + } +{% endif %} + } +{% else %} + reverse_proxy {{ caddy_site_upstream }} +{% endif %} +{% endif %} +{% endif %} +} From 39a1b43577fb6e46b7741ed58134731132dca2de Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 23:19:01 +0200 Subject: [PATCH 35/67] personal-blog: use the caddy_site role First service on the role. 31 lines of copy-pasted Caddy plumbing become 7. Verified: --check before and after the edit reports the same three unrelated tasks as changed, so the edit introduces nothing. Real run leaves all 14 site files on all 3 hosts byte-identical, and the blog still answers HTTP 200. A second consecutive run reports the site task ok with the handler not firing. Side effect worth noting: the playbook no longer has a perpetually-changed task. `command: systemctl reload caddy` always reported changed; the role's handler only fires when the file actually moves. Co-Authored-By: Claude Opus 5 (1M context) --- .../deploy_personal_blog_playbook.yml | 38 ++++--------------- 1 file changed, 7 insertions(+), 31 deletions(-) diff --git a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml index af7a1f3..96d030f 100644 --- a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml +++ b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml @@ -8,7 +8,6 @@ - ./personal_blog_vars.yml vars: personal_blog_subdomain: "{{ subdomains.personal_blog }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" personal_blog_domain: "{{ personal_blog_subdomain }}.{{ root_domain }}" uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" @@ -51,36 +50,13 @@ group: www-data mode: '0664' - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - - - name: Create Caddy file server configuration for personal blog - copy: - dest: "{{ caddy_sites_dir }}/personal-blog.conf" - content: | - {{ personal_blog_domain }} { - root * {{ personal_blog_web_root }} - file_server - } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - command: systemctl reload caddy + - name: Publish the blog through Caddy + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: personal-blog + caddy_site_domain: "{{ personal_blog_domain }}" + caddy_site_root: "{{ personal_blog_web_root }}" # ═════════════════════════════════════════════════════════════════════════ # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. From 4bee18297892196ffed1cabd0d2f7cbb28d36a4a Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 23:24:57 +0200 Subject: [PATCH 36/67] ntfy-emergency-app, vaultwarden, forgejo: use the caddy_site role The plain reverse_proxy shape. All three removed a byte-identical 23-line block (verified by md5 of the diff with the service name normalised) and gained the same 7-line include_role call. The caddy_sites_dir self-reference goes with it. Verified in check mode, nothing applied to the hosts yet: - ntfy-emergency-app: site task ok, changed=0 - vaultwarden: site task ok; the one changed task is a pre-existing always-restarts fail2ban step, identical before the edit - forgejo: check mode cannot run this playbook at all - get_url does not download in check mode so the next task fails on "Source /tmp/forgejo not found". Confirmed identical before the edit. Covered instead by the Stage 2 dry-run, which ran the role against vipy with forgejo's real parameters and reported ok/unchanged. All 14 site files on all 3 hosts still byte-identical. Real runs for these three are still outstanding. Co-Authored-By: Claude Opus 5 (1M context) --- .../forgejo/deploy_forgejo_playbook.yml | 37 ++++--------------- .../deploy_ntfy_emergency_app_playbook.yml | 37 ++++--------------- .../deploy_vaultwarden_playbook.yml | 37 ++++--------------- 3 files changed, 21 insertions(+), 90 deletions(-) diff --git a/ansible/services/forgejo/deploy_forgejo_playbook.yml b/ansible/services/forgejo/deploy_forgejo_playbook.yml index acb1473..db78e95 100644 --- a/ansible/services/forgejo/deploy_forgejo_playbook.yml +++ b/ansible/services/forgejo/deploy_forgejo_playbook.yml @@ -8,7 +8,6 @@ - ./forgejo_vars.yml vars: forgejo_subdomain: "{{ subdomains.forgejo }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" forgejo_domain: "{{ forgejo_subdomain }}.{{ root_domain }}" uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" @@ -88,35 +87,13 @@ enabled: yes state: started - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - - - name: Create Caddy reverse proxy configuration for forgejo - copy: - dest: "{{ caddy_sites_dir }}/forgejo.conf" - content: | - {{ forgejo_domain }} { - reverse_proxy localhost:{{ forgejo_port }} - } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - command: systemctl reload caddy + - name: Publish Forgejo through Caddy + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: forgejo + caddy_site_domain: "{{ forgejo_domain }}" + caddy_site_upstream: "localhost:{{ forgejo_port }}" # ═════════════════════════════════════════════════════════════════════════ # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. diff --git a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml index a8d97a6..7379d5f 100644 --- a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml +++ b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml @@ -8,7 +8,6 @@ - ./ntfy_emergency_app_vars.yml vars: ntfy_emergency_app_subdomain: "{{ subdomains.ntfy_emergency_app }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" ntfy_emergency_app_domain: "{{ ntfy_emergency_app_subdomain }}.{{ root_domain }}" ntfy_service_domain: "{{ subdomains.ntfy }}.{{ root_domain }}" ntfy_emergency_app_ntfy_url: "https://{{ ntfy_service_domain }}" @@ -49,35 +48,13 @@ args: chdir: "{{ ntfy_emergency_app_dir }}" - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - - - name: Create Caddy reverse proxy configuration for ntfy-emergency-app - copy: - dest: "{{ caddy_sites_dir }}/ntfy-emergency-app.conf" - content: | - {{ ntfy_emergency_app_domain }} { - reverse_proxy localhost:{{ ntfy_emergency_app_port }} - } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - command: systemctl reload caddy + - name: Publish ntfy-emergency-app through Caddy + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: ntfy-emergency-app + caddy_site_domain: "{{ ntfy_emergency_app_domain }}" + caddy_site_upstream: "localhost:{{ ntfy_emergency_app_port }}" # ═════════════════════════════════════════════════════════════════════════ # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. diff --git a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml index 9868d13..74e87d8 100644 --- a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml +++ b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml @@ -8,7 +8,6 @@ - ./vaultwarden_vars.yml vars: vaultwarden_subdomain: "{{ subdomains.vaultwarden }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" vaultwarden_domain: "{{ vaultwarden_subdomain }}.{{ root_domain }}" uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" @@ -81,35 +80,13 @@ name: fail2ban state: restarted - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - - - name: Create Caddy reverse proxy configuration for vaultwarden - copy: - dest: "{{ caddy_sites_dir }}/vaultwarden.conf" - content: | - {{ vaultwarden_domain }} { - reverse_proxy localhost:{{ vaultwarden_port }} - } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - command: systemctl reload caddy + - name: Publish Vaultwarden through Caddy + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: vaultwarden + caddy_site_domain: "{{ vaultwarden_domain }}" + caddy_site_upstream: "localhost:{{ vaultwarden_port }}" # ═════════════════════════════════════════════════════════════════════════ # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. From 82fca48b0877008c5c2668ae0b5047d24a02eabb Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 23:29:06 +0200 Subject: [PATCH 37/67] lnbits, memos, mempool: use the caddy_site role lnbits is the header_up shape; memos and mempool are the Tailscale MagicDNS shape. 108 lines removed, 25 added. mempool was the one playbook already reloading Caddy correctly (systemd: state: reloaded rather than command: systemctl reload caddy), so its end marker differed - the role's handler does the same thing. Verified: - lnbits: full --check, site task ok, byte-identical to the live file - memos, mempool: --check --diff via --limit edge shows exactly one added line each, the standardised MagicDNS comment. Both playbooks fail earlier in check mode on their VM play ("Extract memos binary", the same download-does-not-happen-in-check-mode artifact as forgejo), but the edits are confined to the hosts: edge play - memos at line 169+, play 2 starts at 159; mempool at 617+, play 2 starts at 606. The added comment means the next real run of memos/mempool rewrites one comment line. Those two host files were already stale against their playbooks before this change. All 14 site files on all 3 hosts still byte-identical. Co-Authored-By: Claude Opus 5 (1M context) --- .../lnbits/deploy_lnbits_playbook.yml | 43 ++++------------- .../services/memos/deploy_memos_playbook.yml | 43 ++++------------- .../mempool/deploy_mempool_playbook.yml | 47 ++++--------------- 3 files changed, 25 insertions(+), 108 deletions(-) diff --git a/ansible/services/lnbits/deploy_lnbits_playbook.yml b/ansible/services/lnbits/deploy_lnbits_playbook.yml index d596052..65bdfbf 100644 --- a/ansible/services/lnbits/deploy_lnbits_playbook.yml +++ b/ansible/services/lnbits/deploy_lnbits_playbook.yml @@ -8,7 +8,6 @@ - ./lnbits_vars.yml vars: lnbits_subdomain: "{{ subdomains.lnbits }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" lnbits_domain: "{{ lnbits_subdomain }}.{{ root_domain }}" uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" @@ -147,39 +146,15 @@ enabled: yes state: started - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - create: yes - mode: '0644' - - - name: Create Caddy reverse proxy configuration for lnbits - copy: - dest: "{{ caddy_sites_dir }}/lnbits.conf" - content: | - {{ lnbits_domain }} { - reverse_proxy localhost:{{ lnbits_port }} { - header_up X-Forwarded-Host {{ lnbits_domain }} - } - } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - command: systemctl reload caddy + - name: Publish LNBits through Caddy + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: lnbits + caddy_site_domain: "{{ lnbits_domain }}" + caddy_site_upstream: "localhost:{{ lnbits_port }}" + caddy_site_headers_up: + X-Forwarded-Host: "{{ lnbits_domain }}" # ═════════════════════════════════════════════════════════════════════════ # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. diff --git a/ansible/services/memos/deploy_memos_playbook.yml b/ansible/services/memos/deploy_memos_playbook.yml index 5ab254e..83c187e 100644 --- a/ansible/services/memos/deploy_memos_playbook.yml +++ b/ansible/services/memos/deploy_memos_playbook.yml @@ -166,45 +166,18 @@ - ./memos_vars.yml vars: memos_subdomain: "{{ subdomains.memos }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" memos_domain: "{{ memos_subdomain }}.{{ root_domain }}" uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - - - name: Create Caddy reverse proxy configuration for memos (via Tailscale) - copy: - dest: "{{ caddy_sites_dir }}/memos.conf" - content: | - {{ memos_domain }} { - reverse_proxy {{ memos_tailscale_hostname }}:{{ memos_port }} { - # Use Tailscale MagicDNS to resolve the upstream hostname - transport http { - resolvers 100.100.100.100 - } - } - } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - command: systemctl reload caddy + - name: Publish Memos through Caddy (via Tailscale) + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: memos + caddy_site_domain: "{{ memos_domain }}" + caddy_site_upstream: "{{ memos_tailscale_hostname }}:{{ memos_port }}" + caddy_site_resolvers: "100.100.100.100" - name: Create Uptime Kuma monitor setup script for Memos when: uptime_kuma_enabled | default(false) diff --git a/ansible/services/mempool/deploy_mempool_playbook.yml b/ansible/services/mempool/deploy_mempool_playbook.yml index 1dfea1b..ba3e9e9 100644 --- a/ansible/services/mempool/deploy_mempool_playbook.yml +++ b/ansible/services/mempool/deploy_mempool_playbook.yml @@ -614,47 +614,16 @@ vars: mempool_subdomain: "{{ subdomains.mempool }}" mempool_domain: "{{ mempool_subdomain }}.{{ root_domain }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" tasks: - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - create: yes - mode: '0644' - - - name: Create Caddy reverse proxy configuration for Mempool - copy: - dest: "{{ caddy_sites_dir }}/mempool.conf" - content: | - {{ mempool_domain }} { - reverse_proxy mempool-box:{{ mempool_frontend_port }} { - # Use Tailscale MagicDNS to resolve the upstream hostname - transport http { - resolvers 100.100.100.100 - } - } - } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - systemd: - name: caddy - state: reloaded + - name: Publish Mempool through Caddy (via Tailscale) + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: mempool + caddy_site_domain: "{{ mempool_domain }}" + caddy_site_upstream: "mempool-box:{{ mempool_frontend_port }}" + caddy_site_resolvers: "100.100.100.100" - name: Display Mempool URL when: uptime_kuma_enabled | default(false) From 16cbd189b8a9b715ad14a92563103533efaa1f4e Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 23:37:38 +0200 Subject: [PATCH 38/67] ntfy, datum-gateway, headscale: use the caddy_site role Completes Stage 3. No hand-rolled Caddy plumbing remains anywhere: `grep sites-enabled` outside roles/ returns nothing, and so does `grep "systemctl reload caddy"`. ntfy uses caddy_site_body for its plain-HTTP listener and @httpget redirect. Verified ok/unchanged against watchtower; the one other changed task is a pre-existing "Update APT cache". datum-gateway keeps a whole-Caddyfile validate after the role call. The role validates its own fragment, but only a whole-file validate catches a conflict between two sites, and this playbook was the only one that ever had it. Its two debug tasks that echoed command output are gone with the commands. headscale is the one that mattered. Its playbook wrote `reverse_proxy localhost:8080`, but spacey is actually running a /admin* route in front of Headplane behind Caddy basic auth. Running that playbook would have deleted the admin route and its auth - a hazard that predates this work. It now renders the config that is really there, verified ok/unchanged via --start-at-task (the play cannot reach Caddy in check mode: "Install headscale package" fails because the .deb is not really downloaded, before and after this edit alike). Supporting changes for headscale: - headscale_ui_password_hash added to infra_secrets.yml and the identical group_vars/all/vault.yml, read from the live config on spacey. The vault already had headscale_ui_username (= counterweight, confirmed) and headscale_ui_password; I did not verify the password is the plaintext of this hash. - headplane_port added to headscale_vars.yml. - The role's handler now sets become: true. Handlers do not inherit become from the task that notified them, and this play runs become: no. - The include uses `apply: become: yes`; `become:` on an include_role is rejected outright. All 14 site files on all 3 hosts still byte-identical. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 103 ++++++++++-------- ansible/infra_secrets.yml | 103 ++++++++++-------- ansible/roles/caddy_site/handlers/main.yml | 4 + .../deploy_datum_gateway_playbook.yml | 69 +++--------- .../headscale/deploy_headscale_playbook.yml | 60 +++++----- ansible/services/headscale/headscale_vars.yml | 3 + .../services/ntfy/deploy_ntfy_playbook.yml | 50 +++------ 7 files changed, 184 insertions(+), 208 deletions(-) diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index 798e2d8..d1f18ec 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,45 +1,60 @@ $ANSIBLE_VAULT;1.1;AES256 -65343264303166346334396163363362326334626531336363363766326135393462373564313539 -6633346163343664393232363965383334303563396236360a316265623239333431636133306663 -65623237393564323936303936633036343137646637633963396166313462626165616463616665 -3235373562626238640a636163623535376565373835653831666361636563616230353639626131 -38393530363461346334356437613934306330666563326661343465653362313038346535386337 -62623964633764343964653061373137366539326433396231396339313231633632353465323139 -33333866383834376165303964353639386234646535626333363932333764646137663036303530 -32326635343631623535323830333934376466323338353234323464666435383436633139343037 -62666333383231356163383530353339336230373431616665333638633234656231393166303639 -32386134386264613839313564316164336461373738646630636630363731643066633132336636 -64663532663661653539653431353763303436323234383864353064333631643435633733366435 -32656332626633313134623361343965643962663134613363626265353865643738363738316232 -62356136623166303038653830393138646637636335366362323239316333616331383765663861 -33643331323766646463663862396232306233343134326436326235336165616662383636616539 -39373962336264636363303734343666646534343733666632626338393936393032626130643233 -30633963656166343834396431323233363036353165623934336532333662323932336630636235 -63326631366234383732316464383364316337393034303433646365343763633135336336376161 -33326436303863326666616234396438623864333538313363623261353962336430316132343830 -36353761326133666166393930343233643861366138363930623362336566386230316236353365 -65663632383436333536636161393133316334376262653330303238636431353966646666663665 -30386137623066343135616534323336313162643066636330323934646162303362373838313532 -61386163356537613334616537656134373835326265393765303436396233383239393365336136 -64623130383538306331323939386634316663663263306362363830383930346232313965613035 -33353235663230663635316136666335343233646539613832643431393134303865616232636435 -66366437353863336235646438353431616465396365373562636461623938303663326461656134 -66353735343237363164326236663363376333316163353462626234393861663535323463623233 -66623934336662323830616362653035353563343564636431383837643961303637366131633132 -33376133376465623030643935643737636339333236316566636663323934626265633863343963 -38626563323166343264643465626264393838643330393739386330656461656561306539653232 -63393838626233386531353063366131366237633162393537363963633361386238643232336230 -36363062386262303364303736373061316265373730366437326166336163386335623665323234 -35383865646466363534383833653764303833323230613863313735393264336132303934633465 -65616138636134613361316539343739353735666538653438333264623034306633663634306438 -33643165353534623039626536636433373963383866303535333931633633393138663464336562 -36643432366233323235613139626236346135633237613265343933646434373939396265626366 -34313835346463303132326537323166363032306133623662343736393435383330356433306138 -65653931613832313832323936353638303937613863646430633266306330656135386533306462 -30663366396566393134653663356531316639303635343236666333363637636433336533356663 -35346261363134326165316463626439306130653165636134616434663736616666643963336464 -61643834613164336234386237393333636234306162306563363430646262363963366666653666 -64323036323731626236343361613566643565666138386631633462613836613766626562353934 -32363936626466346464356131653232326330386239666638626333346238623964383666393961 -38393264663636353161663636393733396332333838333962393439346533353332653362356230 -62613039376539636363 +61356165613635386631393135656434646436303665313031346566323336313138353433316463 +3363323534613064643132663335623238366431393062340a346538396662306537663163623366 +38626166383933616331623231373137306562623637313263333237633661663436666266616433 +3862346438643638650a306634333535653633613534646630386131333236366538333765323333 +38656163303837303732663561373232393132343331663164656262393730326434373731333636 +36613538646431396536363936336562616431656665653965373864633366663836353434626434 +34383932373461333564303439623565383661646365386665393831383463663662356536356236 +36373964666236626465366161636135393734356536633466383262326537343833636630343738 +62353066316131613737373162643363653662656261363465386364323962656537373061373032 +33343763353464383438363438343965653532393831343930393562633630383932653862623637 +38386239353237356631646436356166373961333464396639383538383662326534343339313330 +65373339356364636634616532363832386631323062363530313861336238353261353334306235 +62616338396431316537346638656365356564346666366366343638356261623664393263323937 +30343261363562383332323462336435376664386134646562643836363834313237373631353731 +35653535663864643266313332356635363262363533663232656531373130336539633066376139 +32316665393831663035623962656364363831333563366135636164346335383738363336663566 +63356532643563393939383635386462663561386434323939303431653438653131363538383034 +33623933333464363032656636643033326162626163353633343062633966343332383138363963 +36383831663562616533316436366566323061386535343538393861383462333166343562316633 +30623665623035393537393965626363323132656433313339396233356666346634316332616336 +37363635363330323230373332326565343530653335383437373230366563366237633665626331 +66623336626230663361636439316337393865383035326136653264666438666566646132353036 +38616264313833316536623238633339373466613866626366383835656638623863323838653030 +63303938376164653966356435386333363731656666313234663535666165646233313137343563 +36386437393139656438333262383437656666343831313239323961373637653163643664356565 +62346431393133656530316262303763646165643336396661666431383436323562336137653031 +62626538373839613734396366653065306534636630346338316237616161613037616364356431 +38643965376133616161336633383664326230383435363334353137303162663738313331346238 +39356161616533616134356231323530306338333162343363353531303263636632613036386638 +63633136386232306234323936303563646466313935326631396565383432386130656638616266 +32383334363237336539396665336366643764633131643663643137376438323666326435626461 +64346633636431393137633537306431646564386565303933636434386462346630626537346438 +32623333666133303061646564366366326665363163396262633164323631636337346130303239 +38373936663337356134666132303165393365663763396362623434633737373538653566646134 +62316164396438303532616266313062326666633130656338653139376634306664633031333037 +31636166306565353334633435656233336233363664306264626237623366336161303134353433 +63383739666462623336386537346662633666626466393039653439346436653937633537396436 +35393339383066326630353066623132333034656539363561346462626265363263303535343961 +39316461616630326539613731303039613736393633373338646266323938326162373831346336 +30656130343463366534323030646238313465306266383034623065636665623366333063383736 +38373063393837306462303564643962373334343139626338623935336435643730646532633630 +66363730386636633639346463363365343239373265303738353732653633653437636130636664 +39386365353334653765343335303263363461313965383664326563333734626533376436626530 +63353163323637303730353564383733653365613635353764333266393532653663326533646132 +39313763373735323835626437306435373238653432393936643165663663656665316132653330 +35383038353532656434366336346235363563636264303734633138323963396562306232646236 +38366536306561653937336333373434336164663336613839353439356435333833396363636437 +38323934613735643363656233333037336465336564313966623063376566663030303230323262 +65643564666534326234306164343365383632333061316238623565353538313538396364313337 +35336230313339643736653238386231343661623337306236383665356632366236356335323530 +66373831616239376231636361333430343433303233393066323865663434643433303832373262 +36366534646161323130393931626362626139663139643263366639656531613436313533363130 +65643736363833333939613566663339623964333262643863333030623138633464386238613934 +31323639333234336264313663376465323737353766643839303665313737336534386665363034 +30326439666433373232306136306365343764643434306561343339353132346430646436343362 +32333765616262363930353435616563333736313533653339656231316230346166363335356638 +37333561366339613136306438306130343230663732333862663838396463623661303961336433 +65646332373139303462303633346432366530643130366133363937653739653036366136373434 +343864373734666431326239373866313734 diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml index 798e2d8..4803d86 100644 --- a/ansible/infra_secrets.yml +++ b/ansible/infra_secrets.yml @@ -1,45 +1,60 @@ $ANSIBLE_VAULT;1.1;AES256 -65343264303166346334396163363362326334626531336363363766326135393462373564313539 -6633346163343664393232363965383334303563396236360a316265623239333431636133306663 -65623237393564323936303936633036343137646637633963396166313462626165616463616665 -3235373562626238640a636163623535376565373835653831666361636563616230353639626131 -38393530363461346334356437613934306330666563326661343465653362313038346535386337 -62623964633764343964653061373137366539326433396231396339313231633632353465323139 -33333866383834376165303964353639386234646535626333363932333764646137663036303530 -32326635343631623535323830333934376466323338353234323464666435383436633139343037 -62666333383231356163383530353339336230373431616665333638633234656231393166303639 -32386134386264613839313564316164336461373738646630636630363731643066633132336636 -64663532663661653539653431353763303436323234383864353064333631643435633733366435 -32656332626633313134623361343965643962663134613363626265353865643738363738316232 -62356136623166303038653830393138646637636335366362323239316333616331383765663861 -33643331323766646463663862396232306233343134326436326235336165616662383636616539 -39373962336264636363303734343666646534343733666632626338393936393032626130643233 -30633963656166343834396431323233363036353165623934336532333662323932336630636235 -63326631366234383732316464383364316337393034303433646365343763633135336336376161 -33326436303863326666616234396438623864333538313363623261353962336430316132343830 -36353761326133666166393930343233643861366138363930623362336566386230316236353365 -65663632383436333536636161393133316334376262653330303238636431353966646666663665 -30386137623066343135616534323336313162643066636330323934646162303362373838313532 -61386163356537613334616537656134373835326265393765303436396233383239393365336136 -64623130383538306331323939386634316663663263306362363830383930346232313965613035 -33353235663230663635316136666335343233646539613832643431393134303865616232636435 -66366437353863336235646438353431616465396365373562636461623938303663326461656134 -66353735343237363164326236663363376333316163353462626234393861663535323463623233 -66623934336662323830616362653035353563343564636431383837643961303637366131633132 -33376133376465623030643935643737636339333236316566636663323934626265633863343963 -38626563323166343264643465626264393838643330393739386330656461656561306539653232 -63393838626233386531353063366131366237633162393537363963633361386238643232336230 -36363062386262303364303736373061316265373730366437326166336163386335623665323234 -35383865646466363534383833653764303833323230613863313735393264336132303934633465 -65616138636134613361316539343739353735666538653438333264623034306633663634306438 -33643165353534623039626536636433373963383866303535333931633633393138663464336562 -36643432366233323235613139626236346135633237613265343933646434373939396265626366 -34313835346463303132326537323166363032306133623662343736393435383330356433306138 -65653931613832313832323936353638303937613863646430633266306330656135386533306462 -30663366396566393134653663356531316639303635343236666333363637636433336533356663 -35346261363134326165316463626439306130653165636134616434663736616666643963336464 -61643834613164336234386237393333636234306162306563363430646262363963366666653666 -64323036323731626236343361613566643565666138386631633462613836613766626562353934 -32363936626466346464356131653232326330386239666638626333346238623964383666393961 -38393264663636353161663636393733396332333838333962393439346533353332653362356230 -62613039376539636363 +34616531626537376463326332306434383163393761363536633133363161373631376437653234 +3831656536313531343433376264643261396634633965370a303931336236323766313065636535 +31343338616364386132623232373266663665353563666535383232666262663062323864303932 +3439373734316639650a633732333064633335383833663261326464336538636666323063646331 +66653464636534663935616461636132613162303530663237376661343338396133323431623561 +30303563303465633231306339326561313533333536376433393130376634303932643339656336 +30616130343733626162366561393939393766623138616537393032376665396139333561623830 +33623130336235633034666561383165616439316334323664623661343438663733373833306335 +30396165313731636435636263353133326135346236323030653734353831626238666663656364 +62386164363361373464383730333733626333333736306535336635613634646535383237326332 +37636532663933663739303634333235343137333363316430643263326231613635643633636437 +30336462633233396430633939616661643261666136656363363461343230653738613166346434 +63376363666462346539343063343562623962303130616535306439653134633630353963373337 +64313239353162363862643432666438383330636638333464323731643163643236313535313030 +65656231383738356537326661663163376634613031396436646633376139313237343534653065 +64323532346133313265346334353530393633396339316330366536646565353836396662303866 +65373030393463306664623235626437613965303837313365373632643935656630303936373837 +64663539623764656636613263346638376665626262663430333231633735653563643864343835 +62303637383332653366383066326536316336306539623230353066343739316430356437313365 +33323031613130643832636133636263303565623330306135333762363036633737343933623437 +33633436363062643436376632336235666330656265316333323361316430656631383734343238 +65616366666536363866343361666563643336306532666332643230656264303565353032393932 +36613738326239376161383064646634623639356439663965323237303361343838373737636235 +39656465333063633136643463356564636537326339653033633063366265636138656466646264 +32643961343863656464353439346531633036613138353333353631626665396239326465303632 +39336664333038636435366333623336353963323535316436646136313739373537376537353735 +66333265633232346536313933333836643439656666636236626266323333383935383565616139 +38626133356364613562393664643563346539366364636332613162663337356232333735393431 +30343635393462356431646336636663333735343164616261343836363366616162613563613066 +66333034613234626431356536323832393466363135653231336238356334353638613838626430 +63636630313533363563376161663637353362383431663130646666646433353864373736613662 +62343666633862643839666532656132653939663437373164393536653739376265303130353636 +62616538326330333136626238306133363634326365366637363961333635356133613564373131 +64626139656633663065353862343135393233333231386166346532656534623361623061336434 +37616565626134333863376532393861376132323434656433613731613834633963386635353336 +39653263346637303836663763653264396139383363616461393333333362616332663863363834 +61613832326466646133626337663364333566326633313664613236346135653332373035313034 +62356530313065393737393634313166313039616635363365633031336434396463326462316236 +62353838636533303664663132353838636632353733316639643964666139383539333761663531 +37366530373165653063303032383461326535613336343635626538373635356266396666346133 +61383434323565306438356634646338666232363238393932353365363461376130323363363236 +34643862653937353461383265663933646264373365623465313666643662633334366437646434 +34343332613832366235663563353435323738356339383338333361383561336366376238393237 +38656265663164376666616366623062393530366261623361383365376235643035353838313235 +32396430376139313430316439333539343266393965353030636638656661316137313437663832 +64366661323865383262323036613230316233666566386133363633303931323464663238323861 +62333132346165353830653062353763306232366563323430383933653163373562326536393362 +66353931376262383931613931303034356537363137366235643430323465623162363031373034 +64653563666533363062333831336463376530306639616631303266386136666233653638653732 +31313537393838333730303235363661326333646531653935633935373362303732653565656438 +36626330623063373832333739643666396135653838643164376264393863646433353036643731 +34623636623366353466333237373265323936643134613230326565303663306431373634356334 +37613732393262626135306632623336613364323632316636623037656538313236666562633438 +37306361333130376538366232393637646263656239303431623963626265386634313735373162 +64393733303132383464386330626135336237303563326238633437346164333738646431333730 +62616533363638643261306565373835303832313539656564386132393863373039653938616261 +36343031356361336563316230343433633033353130373933313361326164633765316433616162 +62356338393634653330623933313433343630386264396337373532376335316362333164363963 +363663666365336564663232363462613162 diff --git a/ansible/roles/caddy_site/handlers/main.yml b/ansible/roles/caddy_site/handlers/main.yml index fb57280..4bfa7ff 100644 --- a/ansible/roles/caddy_site/handlers/main.yml +++ b/ansible/roles/caddy_site/handlers/main.yml @@ -2,7 +2,11 @@ # Fires once at the end of the play, however many sites notified it. # Anything later in the same play that needs the new config live must be # preceded by `- ansible.builtin.meta: flush_handlers`. +# become is explicit because handlers do not inherit it from the task that +# notified them. headscale's play runs become: no and elevates per task, so +# without this the reload would run unprivileged and fail. - name: Reload caddy + become: true ansible.builtin.systemd: name: caddy state: reloaded diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index 3fa36d4..860b765 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -498,64 +498,29 @@ vars: datum_gateway_subdomain: "{{ subdomains.datum_gateway }}" datum_gateway_domain: "{{ datum_gateway_subdomain }}.{{ root_domain }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: "0755" + - name: Publish the DATUM Gateway dashboard through Caddy + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: datum-gateway + caddy_site_domain: "{{ datum_gateway_domain }}" + caddy_site_upstream: "knots-box:{{ datum_gateway_api_port }}" + caddy_site_resolvers: "100.100.100.100" + caddy_site_basic_auth: + - user: "{{ datum_dashboard_username }}" + hash: "{{ datum_dashboard_password_hash }}" - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: "import sites-enabled/*" - insertafter: EOF - state: present - backup: yes - create: yes - mode: "0644" - - - name: Create Caddy reverse proxy config for DATUM Gateway dashboard - copy: - dest: "{{ caddy_sites_dir }}/datum-gateway.conf" - content: | - {{ datum_gateway_domain }} { - basic_auth { - {{ datum_dashboard_username }} {{ datum_dashboard_password_hash }} - } - reverse_proxy knots-box:{{ datum_gateway_api_port }} { - # Resolve via Tailscale MagicDNS - transport http { - resolvers 100.100.100.100 - } - } - } - owner: root - group: root - mode: "0644" - - - name: Validate Caddy config - command: caddy validate --config /etc/caddy/Caddyfile --adapter caddyfile - register: caddy_validate + # The role validates the site fragment on its own. This re-validates the + # whole assembled Caddyfile, which is the only thing that catches a + # conflict between this site and another. Kept from the hand-rolled + # version; the other nine services never had it. + - name: Validate the assembled Caddyfile + ansible.builtin.command: caddy validate --config /etc/caddy/Caddyfile --adapter caddyfile changed_when: false - - name: Display Caddy validation output - debug: - msg: "{{ caddy_validate.stdout_lines + caddy_validate.stderr_lines }}" - - - name: Reload Caddy - command: systemctl reload caddy - register: caddy_reload - - - name: Display Caddy reload output - debug: - msg: "{{ caddy_reload.stdout_lines + caddy_reload.stderr_lines }}" - - name: Display DATUM Gateway dashboard URL when: uptime_kuma_enabled | default(false) debug: diff --git a/ansible/services/headscale/deploy_headscale_playbook.yml b/ansible/services/headscale/deploy_headscale_playbook.yml index 61ba22e..527e4c2 100644 --- a/ansible/services/headscale/deploy_headscale_playbook.yml +++ b/ansible/services/headscale/deploy_headscale_playbook.yml @@ -8,7 +8,6 @@ - ./headscale_vars.yml vars: headscale_subdomain: "{{ subdomains.headscale }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" headscale_domain: "{{ headscale_subdomain }}.{{ root_domain }}" headscale_base_domain: "tailnet.{{ root_domain }}" headscale_namespace: "{{ service_settings.headscale.namespace }}" @@ -230,39 +229,34 @@ port: '3478' proto: udp - - name: Ensure Caddy sites-enabled directory exists - become: yes - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Ensure Caddyfile includes import directive for sites-enabled - become: yes - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - - - name: Create Caddy reverse proxy configuration for headscale - become: yes - copy: - dest: "{{ caddy_sites_dir }}/headscale.conf" - content: | - {{ headscale_domain }} { - reverse_proxy localhost:{{ headscale_port }} + - name: Publish headscale through Caddy + ansible.builtin.include_role: + name: caddy_site + # This play is become: no and elevates per task. `apply` is how an + # include_role passes become down to the role's tasks - `become:` on + # the include itself is rejected. The role's handler sets its own. + apply: + become: yes + vars: + caddy_site_name: headscale + caddy_site_domain: "{{ headscale_domain }}" + # Raw body, and it must stay raw: the /admin* route in front of + # Headplane is not expressible as a plain reverse_proxy. The previous + # version of this task wrote only `reverse_proxy localhost:8080`, which + # would have deleted the admin route and its auth on the next run. + caddy_site_body: | + @headplane { + path /admin* } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - become: yes - command: systemctl reload caddy + handle @headplane { + basicauth { + {{ headscale_ui_username }} {{ headscale_ui_password_hash }} + } + reverse_proxy http://localhost:{{ headplane_port }} + } + # Headscale API is protected by its own API key authentication + # All API operations require a valid Bearer token in the Authorization header + reverse_proxy * http://localhost:{{ headscale_port }} # ═════════════════════════════════════════════════════════════════════════ # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. diff --git a/ansible/services/headscale/headscale_vars.yml b/ansible/services/headscale/headscale_vars.yml index 0b0ee50..05fc040 100644 --- a/ansible/services/headscale/headscale_vars.yml +++ b/ansible/services/headscale/headscale_vars.yml @@ -22,3 +22,6 @@ remote_port: "{{ hostvars.get(remote_host_name, {}).get('ansible_port', 22) }}" # Local backup local_backup_dir: "{{ lookup('env', 'HOME') }}/headscale-backups" backup_script_path: "{{ lookup('env', 'HOME') }}/.local/bin/headscale_backup.sh" + +# Headplane (headscale admin UI), proxied at /admin* behind Caddy basic auth +headplane_port: 3000 diff --git a/ansible/services/ntfy/deploy_ntfy_playbook.yml b/ansible/services/ntfy/deploy_ntfy_playbook.yml index d030123..61fafe1 100644 --- a/ansible/services/ntfy/deploy_ntfy_playbook.yml +++ b/ansible/services/ntfy/deploy_ntfy_playbook.yml @@ -8,7 +8,6 @@ - ./ntfy_vars.yml vars: ntfy_subdomain: "{{ subdomains.ntfy }}" - caddy_sites_dir: "{{ caddy_sites_dir }}" ntfy_domain: "{{ ntfy_subdomain }}.{{ root_domain }}" tasks: @@ -76,42 +75,23 @@ shell: | (echo "{{ ntfy_password }}"; echo "{{ ntfy_password }}") | ntfy user add --role=admin "{{ ntfy_username }}" - - name: Ensure Caddy sites-enabled directory exists - file: - path: "{{ caddy_sites_dir }}" - state: directory - owner: root - group: root - mode: '0755' + - name: Publish ntfy through Caddy + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: ntfy + caddy_site_domain: "{{ ntfy_domain }}, http://{{ ntfy_domain }}" + # Raw body: ntfy needs a plain-HTTP listener for its CLI/app clients, + # with only GETs to the docs and topic paths redirected to HTTPS. + caddy_site_body: | + reverse_proxy 127.0.0.1:{{ ntfy_port }} - - name: Ensure Caddyfile includes import directive for sites-enabled - lineinfile: - path: /etc/caddy/Caddyfile - line: 'import sites-enabled/*' - insertafter: EOF - state: present - backup: yes - - - name: Create Caddy reverse proxy configuration for ntfy - copy: - dest: "{{ caddy_sites_dir }}/ntfy.conf" - content: | - {{ ntfy_domain }}, http://{{ ntfy_domain }} { - reverse_proxy 127.0.0.1:{{ ntfy_port }} - - @httpget { - protocol http - method GET - path_regexp ^/([-_a-z0-9]{0,64}$|docs/|static/) - } - redir @httpget https://{host}{uri} + @httpget { + protocol http + method GET + path_regexp ^/([-_a-z0-9]{0,64}$|docs/|static/) } - owner: root - group: root - mode: '0644' - - - name: Reload Caddy to apply new config - command: systemctl reload caddy + redir @httpget https://{host}{uri} handlers: - name: Restart ntfy From c4094b692fddde6efc53b37cbadab632b1da1b17 Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 23:50:15 +0200 Subject: [PATCH 39/67] bitcoin-knots, fulcrum, datum-gateway: add and use the socket_proxy role Three near-identical hosts: edge plays become one role plus three short calls. 183 lines removed, 34 added, plus a 111-line role. Verified before touching any playbook: all six live units on vipy reproduced byte-identically. Then --limit edge --check per playbook - bitcoin-knots and fulcrum changed=0; datum-gateway changed=2, both attributable to the already known caddy_site comment line and the Reload caddy handler it triggers. The 6 units and 14 Caddy files on the hosts are byte-identical afterwards. PLAN_4 claimed these three plays had "no behavioural drift at all". That was wrong - it came from a diff truncated by head -60. The live bitcoin-p2p-proxy units carry four settings this playbook never wrote: .socket Documentation=, FreeBind=true .service Documentation=, TimeoutStopSec=5, StandardOutput=journal, StandardError=journal FreeBind is the one that matters: it lets the socket bind to an address that is not up yet, so without it the socket can fail to start on boot. Running the bitcoin-knots playbook would have stripped it. Same class of hazard as headscale. The role expresses all four; bitcoin-p2p is the only caller that passes any. Also: UFW treats the rule comment as part of the rule. datum-stratum's live comment is "DATUM Gateway Stratum public access" but the role's derived default produced "DATUM Stratum public access", which rewrote the rule. Caught in the dry-run; datum now passes the comment explicitly. Two deliberate differences from the original, both documented in the README: ignore_errors: yes on the upstream check became failed_when: false, and the handler restarts the .socket, which drops connections open through it - it fires only when a unit file actually changes. The inert Uptime Kuma TCP monitor blocks stay in the playbooks rather than being pulled into a new role (12/12/18 guarded tasks). Co-Authored-By: Claude Opus 5 (1M context) --- ansible/roles/socket_proxy/README.md | 62 +++++++++++++++ ansible/roles/socket_proxy/defaults/main.yml | 17 +++++ ansible/roles/socket_proxy/handlers/main.yml | 8 ++ ansible/roles/socket_proxy/tasks/main.yml | 54 +++++++++++++ .../socket_proxy/templates/proxy.service.j2 | 18 +++++ .../socket_proxy/templates/proxy.socket.j2 | 14 ++++ .../deploy_bitcoin_knots_playbook.yml | 76 ++++--------------- .../deploy_datum_gateway_playbook.yml | 72 +++--------------- .../fulcrum/deploy_fulcrum_playbook.yml | 69 ++--------------- 9 files changed, 207 insertions(+), 183 deletions(-) create mode 100644 ansible/roles/socket_proxy/README.md create mode 100644 ansible/roles/socket_proxy/defaults/main.yml create mode 100644 ansible/roles/socket_proxy/handlers/main.yml create mode 100644 ansible/roles/socket_proxy/tasks/main.yml create mode 100644 ansible/roles/socket_proxy/templates/proxy.service.j2 create mode 100644 ansible/roles/socket_proxy/templates/proxy.socket.j2 diff --git a/ansible/roles/socket_proxy/README.md b/ansible/roles/socket_proxy/README.md new file mode 100644 index 0000000..cebb95f --- /dev/null +++ b/ansible/roles/socket_proxy/README.md @@ -0,0 +1,62 @@ +# `socket_proxy` + +Exposes a service running on a private Tailscale host through a public TCP port +on an edge machine, using `systemd-socket-proxyd`. Writes a `.socket` and a +`.service` unit, enables the socket, opens the UFW port, and checks the upstream +is reachable. + +## Usage + +```yaml +- ansible.builtin.include_role: + name: socket_proxy + vars: + socket_proxy_name: fulcrum-ssl # -> fulcrum-ssl-proxy.{socket,service} + socket_proxy_description: "Fulcrum SSL" # -> "Fulcrum SSL Proxy Socket" + socket_proxy_listen_port: "{{ fulcrum_ssl_port }}" + socket_proxy_upstream_host: "{{ fulcrum_tailscale_hostname }}" +``` + +`socket_proxy_upstream_port` defaults to `socket_proxy_listen_port`, which is +what all three current callers want. + +## Optional unit settings + +These exist because the **live** `bitcoin-p2p-proxy` units on vipy carried +settings the playbook never wrote. Somebody added them by hand, so running +`deploy_bitcoin_knots_playbook.yml` would have silently removed them: + +| Variable | Emits | Why it matters | +|---|---|---| +| `socket_proxy_free_bind` | `FreeBind=true` in `[Socket]` | Lets the socket bind to an address that is not up yet. Without it the socket can fail to start on boot. | +| `socket_proxy_documentation` | `Documentation=` in both units | Cosmetic. | +| `socket_proxy_timeout_stop_sec` | `TimeoutStopSec=` | Bounds how long a stop can hang. | +| `socket_proxy_log_to_journal` | `StandardOutput=journal` + `StandardError=journal` | Cosmetic on modern systemd, which defaults to the journal anyway. | + +Only `bitcoin-p2p` passes any of them. + +## `socket_proxy_ufw_comment` + +Defaults to `" public access"`, which reproduces the live rule +comment for bitcoin-p2p and fulcrum-ssl. **datum-stratum must pass it +explicitly** — its live comment is `DATUM Gateway Stratum public access` while +the derived default would be `DATUM Stratum public access`, and UFW treats the +comment as part of the rule, so the mismatch rewrites the rule on every run. + +## The upstream check never fails the play + +`wait_for` on the upstream carries `failed_when: false`. The proxy is correctly +configured whether or not the backend happens to be up, and this is the one task +that depends on another machine. The original plays used `ignore_errors: yes`, +which prints a red "ignoring" line; `failed_when: false` is the quieter +equivalent. + +## Restarts + +The handler restarts the `.socket`, not the `.service` — that is what picks up a +changed unit; the service is started by the socket on the next connection. + +**Restarting a socket drops connections that are currently open through it.** +For bitcoin-p2p that means peers reconnect; for datum-stratum it means a mining +client has to reconnect and may lose in-flight shares. The handler only fires +when a unit file actually changes. diff --git a/ansible/roles/socket_proxy/defaults/main.yml b/ansible/roles/socket_proxy/defaults/main.yml new file mode 100644 index 0000000..6a512ea --- /dev/null +++ b/ansible/roles/socket_proxy/defaults/main.yml @@ -0,0 +1,17 @@ +--- +# Required +socket_proxy_name: "" # "bitcoin-p2p" -> bitcoin-p2p-proxy.{socket,service} +socket_proxy_description: "" # "Bitcoin P2P" -> "Bitcoin P2P Proxy Socket" +socket_proxy_listen_port: 0 # public port on the edge host +socket_proxy_upstream_host: "" # Tailscale hostname, e.g. "knots-box" + +# Optional +socket_proxy_upstream_port: "" # defaults to socket_proxy_listen_port +socket_proxy_documentation: "" # Documentation= in both units +socket_proxy_free_bind: false # FreeBind=true: bind before the address is up +socket_proxy_timeout_stop_sec: "" # TimeoutStopSec= +socket_proxy_log_to_journal: false # StandardOutput/StandardError=journal + +# Firewall +socket_proxy_ufw_proto: tcp +socket_proxy_ufw_comment: "" # defaults to " public access" diff --git a/ansible/roles/socket_proxy/handlers/main.yml b/ansible/roles/socket_proxy/handlers/main.yml new file mode 100644 index 0000000..7fdf421 --- /dev/null +++ b/ansible/roles/socket_proxy/handlers/main.yml @@ -0,0 +1,8 @@ +--- +# Restarting the .socket is what picks up a changed unit; the .service is +# started by the socket on the next connection. +- name: Restart socket proxy + ansible.builtin.systemd: + name: "{{ socket_proxy_name }}-proxy.socket" + state: restarted + daemon_reload: yes diff --git a/ansible/roles/socket_proxy/tasks/main.yml b/ansible/roles/socket_proxy/tasks/main.yml new file mode 100644 index 0000000..4315555 --- /dev/null +++ b/ansible/roles/socket_proxy/tasks/main.yml @@ -0,0 +1,54 @@ +--- +- name: Assert socket_proxy parameters are sane + ansible.builtin.assert: + that: + - socket_proxy_name | length > 0 + - socket_proxy_description | length > 0 + - socket_proxy_listen_port | int > 0 + - socket_proxy_upstream_host | length > 0 + fail_msg: >- + socket_proxy: '{{ socket_proxy_name | default("") }}' needs a name, + a description, a listen port and an upstream host. + quiet: true + +- name: "Create the {{ socket_proxy_name }}-proxy socket unit" + ansible.builtin.template: + src: proxy.socket.j2 + dest: "/etc/systemd/system/{{ socket_proxy_name }}-proxy.socket" + owner: root + group: root + mode: '0644' + notify: Restart socket proxy + +- name: "Create the {{ socket_proxy_name }}-proxy service unit" + ansible.builtin.template: + src: proxy.service.j2 + dest: "/etc/systemd/system/{{ socket_proxy_name }}-proxy.service" + owner: root + group: root + mode: '0644' + notify: Restart socket proxy + +- name: "Enable and start the {{ socket_proxy_name }}-proxy socket" + ansible.builtin.systemd: + name: "{{ socket_proxy_name }}-proxy.socket" + enabled: yes + state: started + daemon_reload: yes + +- name: "Allow the {{ socket_proxy_name }} port through UFW" + community.general.ufw: + rule: allow + port: "{{ socket_proxy_listen_port | string }}" + proto: "{{ socket_proxy_ufw_proto }}" + comment: "{{ socket_proxy_ufw_comment | default(socket_proxy_description ~ ' public access', true) }}" + +# Reachability of the upstream over Tailscale. Deliberately non-fatal: the proxy +# is still correctly configured if the backend happens to be down, and this is +# the one check that depends on another machine being up. +- name: "Verify {{ socket_proxy_upstream_host }} is reachable over Tailscale" + ansible.builtin.wait_for: + host: "{{ socket_proxy_upstream_host }}" + port: "{{ socket_proxy_upstream_port | default(socket_proxy_listen_port, true) }}" + timeout: 10 + failed_when: false diff --git a/ansible/roles/socket_proxy/templates/proxy.service.j2 b/ansible/roles/socket_proxy/templates/proxy.service.j2 new file mode 100644 index 0000000..e59d251 --- /dev/null +++ b/ansible/roles/socket_proxy/templates/proxy.service.j2 @@ -0,0 +1,18 @@ +[Unit] +Description={{ socket_proxy_description }} Proxy to {{ socket_proxy_upstream_host }} +{% if socket_proxy_documentation %} +Documentation={{ socket_proxy_documentation }} +{% endif %} +Requires={{ socket_proxy_name }}-proxy.socket +After=network.target + +[Service] +Type=notify +ExecStart=/lib/systemd/systemd-socket-proxyd {{ socket_proxy_upstream_host }}:{{ socket_proxy_upstream_port | default(socket_proxy_listen_port, true) }} +{% if socket_proxy_timeout_stop_sec %} +TimeoutStopSec={{ socket_proxy_timeout_stop_sec }} +{% endif %} +{% if socket_proxy_log_to_journal %} +StandardOutput=journal +StandardError=journal +{% endif %} diff --git a/ansible/roles/socket_proxy/templates/proxy.socket.j2 b/ansible/roles/socket_proxy/templates/proxy.socket.j2 new file mode 100644 index 0000000..0dc1721 --- /dev/null +++ b/ansible/roles/socket_proxy/templates/proxy.socket.j2 @@ -0,0 +1,14 @@ +[Unit] +Description={{ socket_proxy_description }} Proxy Socket +{% if socket_proxy_documentation %} +Documentation={{ socket_proxy_documentation }} +{% endif %} + +[Socket] +ListenStream={{ socket_proxy_listen_port }} +{% if socket_proxy_free_bind %} +FreeBind=true +{% endif %} + +[Install] +WantedBy=sockets.target diff --git a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml index fcc02b4..4e9faa3 100644 --- a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +++ b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml @@ -753,62 +753,21 @@ uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - - name: Create Bitcoin P2P proxy socket unit - copy: - dest: /etc/systemd/system/bitcoin-p2p-proxy.socket - content: | - [Unit] - Description=Bitcoin P2P Proxy Socket - - [Socket] - ListenStream={{ bitcoin_p2p_port }} - - [Install] - WantedBy=sockets.target - owner: root - group: root - mode: '0644' - notify: Restart bitcoin-p2p-proxy socket - - - name: Create Bitcoin P2P proxy service unit - copy: - dest: /etc/systemd/system/bitcoin-p2p-proxy.service - content: | - [Unit] - Description=Bitcoin P2P Proxy to {{ bitcoin_tailscale_hostname }} - Requires=bitcoin-p2p-proxy.socket - After=network.target - - [Service] - Type=notify - ExecStart=/lib/systemd/systemd-socket-proxyd {{ bitcoin_tailscale_hostname }}:{{ bitcoin_p2p_port }} - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start Bitcoin P2P proxy socket - systemd: - name: bitcoin-p2p-proxy.socket - enabled: yes - state: started - - - name: Allow Bitcoin P2P port through UFW - ufw: - rule: allow - port: "{{ bitcoin_p2p_port | string }}" - proto: tcp - comment: "Bitcoin P2P public access" - - - name: Verify connectivity to knots-box via Tailscale - wait_for: - host: "{{ bitcoin_tailscale_hostname }}" - port: "{{ bitcoin_p2p_port }}" - timeout: 10 - ignore_errors: yes + - name: Expose Bitcoin P2P through a socket proxy + ansible.builtin.include_role: + name: socket_proxy + vars: + socket_proxy_name: bitcoin-p2p + socket_proxy_description: "Bitcoin P2P" + socket_proxy_listen_port: "{{ bitcoin_p2p_port }}" + socket_proxy_upstream_host: "{{ bitcoin_tailscale_hostname }}" + # These four were added by hand on vipy and were NOT in this playbook; + # writing the unit without them would have dropped FreeBind, which lets + # the socket bind before the address is up. + socket_proxy_documentation: "https://github.com/bitcoin/bitcoin" + socket_proxy_free_bind: true + socket_proxy_timeout_stop_sec: 5 + socket_proxy_log_to_journal: true - name: Display public endpoint when: uptime_kuma_enabled | default(false) @@ -931,8 +890,3 @@ - /tmp/setup_bitcoin_p2p_tcp_monitor.py - /tmp/ansible_bitcoin_p2p_config.yml - handlers: - - name: Restart bitcoin-p2p-proxy socket - systemd: - name: bitcoin-p2p-proxy.socket - state: restarted diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index 860b765..1ddc75c 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -663,62 +663,17 @@ uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - - name: Create Stratum proxy socket unit - copy: - dest: /etc/systemd/system/datum-stratum-proxy.socket - content: | - [Unit] - Description=DATUM Stratum Proxy Socket - - [Socket] - ListenStream={{ datum_gateway_stratum_port }} - - [Install] - WantedBy=sockets.target - owner: root - group: root - mode: "0644" - notify: Restart datum-stratum-proxy socket - - - name: Create Stratum proxy service unit - copy: - dest: /etc/systemd/system/datum-stratum-proxy.service - content: | - [Unit] - Description=DATUM Stratum Proxy to {{ datum_tailscale_hostname }} - Requires=datum-stratum-proxy.socket - After=network.target - - [Service] - Type=notify - ExecStart=/lib/systemd/systemd-socket-proxyd {{ datum_tailscale_hostname }}:{{ datum_gateway_stratum_port }} - owner: root - group: root - mode: "0644" - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start Stratum proxy socket - systemd: - name: datum-stratum-proxy.socket - enabled: yes - state: started - - - name: Allow Stratum port through UFW - ufw: - rule: allow - port: "{{ datum_gateway_stratum_port | string }}" - proto: tcp - comment: "DATUM Gateway Stratum public access" - - - name: Verify connectivity to knots-box Stratum via Tailscale - wait_for: - host: "{{ datum_tailscale_hostname }}" - port: "{{ datum_gateway_stratum_port }}" - timeout: 10 - ignore_errors: yes + - name: Expose the DATUM Stratum port through a socket proxy + ansible.builtin.include_role: + name: socket_proxy + vars: + socket_proxy_name: datum-stratum + socket_proxy_description: "DATUM Stratum" + socket_proxy_listen_port: "{{ datum_gateway_stratum_port }}" + socket_proxy_upstream_host: "{{ datum_tailscale_hostname }}" + # Matches the UFW comment already on vipy; the derived default would + # have said "DATUM Stratum" and rewritten the rule. + socket_proxy_ufw_comment: "DATUM Gateway Stratum public access" - name: Display public Stratum endpoint when: uptime_kuma_enabled | default(false) @@ -845,8 +800,3 @@ - /tmp/setup_datum_stratum_tcp_monitor.py - /tmp/ansible_datum_stratum_config.yml - handlers: - - name: Restart datum-stratum-proxy socket - systemd: - name: datum-stratum-proxy.socket - state: restarted diff --git a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml index a1548a7..6ffe254 100644 --- a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml +++ b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml @@ -552,62 +552,14 @@ uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - - name: Create Fulcrum SSL proxy socket unit - copy: - dest: /etc/systemd/system/fulcrum-ssl-proxy.socket - content: | - [Unit] - Description=Fulcrum SSL Proxy Socket - - [Socket] - ListenStream={{ fulcrum_ssl_port }} - - [Install] - WantedBy=sockets.target - owner: root - group: root - mode: '0644' - notify: Restart fulcrum-ssl-proxy socket - - - name: Create Fulcrum SSL proxy service unit - copy: - dest: /etc/systemd/system/fulcrum-ssl-proxy.service - content: | - [Unit] - Description=Fulcrum SSL Proxy to {{ fulcrum_tailscale_hostname }} - Requires=fulcrum-ssl-proxy.socket - After=network.target - - [Service] - Type=notify - ExecStart=/lib/systemd/systemd-socket-proxyd {{ fulcrum_tailscale_hostname }}:{{ fulcrum_ssl_port }} - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start Fulcrum SSL proxy socket - systemd: - name: fulcrum-ssl-proxy.socket - enabled: yes - state: started - - - name: Allow Fulcrum SSL port through UFW - ufw: - rule: allow - port: "{{ fulcrum_ssl_port | string }}" - proto: tcp - comment: "Fulcrum SSL public access" - - - name: Verify connectivity to fulcrum-box via Tailscale - wait_for: - host: "{{ fulcrum_tailscale_hostname }}" - port: "{{ fulcrum_ssl_port }}" - timeout: 10 - ignore_errors: yes + - name: Expose Fulcrum SSL through a socket proxy + ansible.builtin.include_role: + name: socket_proxy + vars: + socket_proxy_name: fulcrum-ssl + socket_proxy_description: "Fulcrum SSL" + socket_proxy_listen_port: "{{ fulcrum_ssl_port }}" + socket_proxy_upstream_host: "{{ fulcrum_tailscale_hostname }}" - name: Display public endpoint when: uptime_kuma_enabled | default(false) @@ -730,9 +682,4 @@ - /tmp/setup_fulcrum_ssl_tcp_monitor.py - /tmp/ansible_fulcrum_ssl_config.yml - handlers: - - name: Restart fulcrum-ssl-proxy socket - systemd: - name: fulcrum-ssl-proxy.socket - state: restarted From cea2523e15c656cc04b960c8eb9b6a324a18115d Mon Sep 17 00:00:00 2001 From: counterweight Date: Fri, 11 Sep 2026 23:53:09 +0200 Subject: [PATCH 40/67] caddy: close out Plan 4 All five close-out greps return nothing: no sites-enabled handling outside roles/, no `systemctl reload caddy`, no caddy_sites_dir self-reference, no inline proxy unit writes. 37 playbooks syntax clean. The 14 Caddy site files and 6 proxy units on the hosts are byte-identical to the Stage 0 baseline. Seven play names still said "on vipy" while the play targeted a group. Renamed to "on the edge host" - the last place a play claimed a hostname after Plan 2. Documented the four vhosts in /etc/caddy/sites-enabled that no playbook writes (uptime-kuma, arbretstaging, bitcoininfra, scriberr) in the caddy_site README. None deleted. uptime-kuma.conf was going to be deleted as dead config. It is not dead: the louislam/uptime-kuma container is STILL RUNNING on watchtower - created 2026-02-07, restart=unless-stopped, healthy - and uptime.contrapeso.xyz returns 302, not the 502 a dead backend would give. The "decommissioning" retired the Ansible code and the vault credentials, not the service. PLAN_3 claimed "the tokens died with the server"; that is corrected there. The Caddyfile.* backups are kept: one per host, Nov-Dec 2025, not churning, and the only record of each Caddyfile before the import line was added. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/roles/caddy_site/README.md | 24 +++++++++++++++++++ .../deploy_bitcoin_knots_playbook.yml | 2 +- .../deploy_datum_gateway_playbook.yml | 4 ++-- .../fulcrum/deploy_fulcrum_playbook.yml | 2 +- .../services/memos/deploy_memos_playbook.yml | 2 +- .../mempool/deploy_mempool_playbook.yml | 2 +- .../phoenixd/deploy_phoenixd_playbook.yml | 2 +- 7 files changed, 31 insertions(+), 7 deletions(-) diff --git a/ansible/roles/caddy_site/README.md b/ansible/roles/caddy_site/README.md index 6891070..c8e32af 100644 --- a/ansible/roles/caddy_site/README.md +++ b/ansible/roles/caddy_site/README.md @@ -92,3 +92,27 @@ upstream hostname` in every case. `datum-gateway` previously said `# Resolve via Tailscale MagicDNS`. Migrating it therefore rewrites one comment line, which Caddy ignores. Every other site renders byte-identical to what its playbook produced. + +## Sites on the hosts that this role does NOT manage + +Four vhosts exist in `/etc/caddy/sites-enabled/` that no playbook writes. They +were made by hand. The role only ever writes the one file it is told to, so it +leaves them alone — but nothing in the repo records them, and that is why they +are listed here. Checked 2026-09-11: + +| File | Host | Serves | State | +|---|---|---|---| +| `uptime-kuma.conf` | watchtower | `localhost:3001` | **HTTP 302 — still live**, see below | +| `arbretstaging.conf` | vipy | `arbret-staging-box:80` via MagicDNS | HTTP 200 | +| `bitcoininfra.conf` | vipy | static `file_server` from `/var/www/bitcoin-services-home` | HTTP 200 | +| `scriberr.conf` | vipy | `scriberr-box:8080` via MagicDNS | HTTP 502 — upstream down | + +**`uptime-kuma.conf` must not be deleted as dead config.** Uptime Kuma was +"decommissioned" in the repo — its playbooks archived and its credentials pulled +from the vault — but the container is **still running** on watchtower +(`louislam/uptime-kuma:latest`, created 2026-02-07, `restart=unless-stopped`) +and is still reachable at its public subdomain. Only the Ansible code was +retired; the service was not. See `archive/uptime_kuma/`. + +`scriberr` returning 502 is the one that looks like genuine rot: it proxies to a +`scriberr-box` that is not answering, and `scriberr-box` is not in the inventory. diff --git a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml index 4e9faa3..1ae5823 100644 --- a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +++ b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml @@ -740,7 +740,7 @@ state: restarted -- name: Setup public Bitcoin P2P forwarding on vipy via systemd-socket-proxyd +- name: Setup public Bitcoin P2P forwarding on the edge host hosts: edge become: yes vars_files: diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index 1ddc75c..0c53e9d 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -487,7 +487,7 @@ # =========================================== # Caddy Reverse Proxy for DATUM Dashboard (on vipy) # =========================================== -- name: Configure Caddy reverse proxy for DATUM Gateway dashboard on vipy +- name: Configure Caddy reverse proxy for DATUM Gateway dashboard on the edge host hosts: edge become: yes vars_files: @@ -650,7 +650,7 @@ # Miners connect to vipy:23334; traffic is forwarded to knots-box:23334 # over the Tailscale network, matching the Bitcoin P2P proxy pattern. # =========================================== -- name: Setup public Stratum port forwarding on vipy via systemd-socket-proxyd +- name: Setup public Stratum port forwarding on the edge host hosts: edge become: yes vars_files: diff --git a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml index 6ffe254..01e9f17 100644 --- a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml +++ b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml @@ -540,7 +540,7 @@ state: restarted -- name: Setup public Fulcrum SSL forwarding on vipy via systemd-socket-proxyd +- name: Setup public Fulcrum SSL forwarding on the edge host hosts: edge become: yes vars_files: diff --git a/ansible/services/memos/deploy_memos_playbook.yml b/ansible/services/memos/deploy_memos_playbook.yml index 83c187e..04e99cf 100644 --- a/ansible/services/memos/deploy_memos_playbook.yml +++ b/ansible/services/memos/deploy_memos_playbook.yml @@ -156,7 +156,7 @@ state: restarted -- name: Configure Caddy reverse proxy for Memos on vipy (proxying via Tailscale) +- name: Configure Caddy reverse proxy for Memos on the edge host (via Tailscale) hosts: edge become: yes vars_files: diff --git a/ansible/services/mempool/deploy_mempool_playbook.yml b/ansible/services/mempool/deploy_mempool_playbook.yml index ba3e9e9..e9cab0e 100644 --- a/ansible/services/mempool/deploy_mempool_playbook.yml +++ b/ansible/services/mempool/deploy_mempool_playbook.yml @@ -603,7 +603,7 @@ - /tmp/ansible_mempool_config.yml - /tmp/mempool_push_urls.yml -- name: Configure Caddy reverse proxy for Mempool on vipy +- name: Configure Caddy reverse proxy for Mempool on the edge host hosts: edge become: yes vars_files: diff --git a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml index f20ecdc..8b30a1f 100644 --- a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml +++ b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml @@ -19,7 +19,7 @@ # ⚠️ After the first run, back up {{ phoenixd_data_dir }}/seed.dat. Losing it # means losing the funds. See setup_backup_phoenixd_to_lapy.yml. -- name: Deploy phoenixd on vipy +- name: Deploy phoenixd on the edge host hosts: edge become: yes vars_files: From a8a2815c5f9099cf2d8a37baf5f273920dd315fd Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 15:14:47 +0200 Subject: [PATCH 41/67] inventory --- ansible/inventory.ini | 22 +++++++++++++--------- 1 file changed, 13 insertions(+), 9 deletions(-) diff --git a/ansible/inventory.ini b/ansible/inventory.ini index 4643dbb..b9b2906 100644 --- a/ansible/inventory.ini +++ b/ansible/inventory.ini @@ -2,19 +2,20 @@ vipy ansible_host=167.172.107.33 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua watchtower ansible_host=164.92.239.72 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua spacey ansible_host=64.227.112.128 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua + [nodito_host] nodito ansible_host=192.168.1.139 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua - +# Requires the tailnet to be up on the control node. [nodito_vms] -knots_box_local ansible_host=192.168.1.135 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -fulcrum_box_local ansible_host=192.168.1.142 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -mempool_box_local ansible_host=192.168.1.140 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -memos_box_local ansible_host=192.168.1.130 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -forgejo_runner_local ansible_host=192.168.1.147 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -arbret_staging_local ansible_host=192.168.1.142 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -small_backups_local ansible_host=192.168.1.148 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -nonkeiwaisi_local ansible_host=192.168.1.151 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +knots_box_local ansible_host=knots-box lan_ip=192.168.1.135 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +fulcrum_box_local ansible_host=fulcrum-box lan_ip=192.168.1.140 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +mempool_box_local ansible_host=mempool-box lan_ip=192.168.1.142 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +memos_box_local ansible_host=memos-box lan_ip=192.168.1.145 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +forgejo_runner_local ansible_host=forgejo-runner-box lan_ip=192.168.1.132 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +arbret_staging_local ansible_host=arbret-staging-box lan_ip=192.168.1.147 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +small_backups_local ansible_host=small-backups-box lan_ip=192.168.1.131 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +nonkeiwaisi_local ansible_host=nonkeiwaisi-box lan_ip=192.168.1.151 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua # Local connection to laptop: this assumes you're running ansible commands from your personal laptop [lapy] @@ -65,3 +66,6 @@ nodito_vms edge monitoring vpn_control + +[backup_store] +small_backups_local From 01b83a80ec80b44289a0637fa4b1f1aaaf9b43d8 Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 15:32:17 +0200 Subject: [PATCH 42/67] ip tricks --- ansible/group_vars/nodito_vms.yml | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 ansible/group_vars/nodito_vms.yml diff --git a/ansible/group_vars/nodito_vms.yml b/ansible/group_vars/nodito_vms.yml new file mode 100644 index 0000000..2ff6994 --- /dev/null +++ b/ansible/group_vars/nodito_vms.yml @@ -0,0 +1,17 @@ +--- +# Reach the VMs over Tailscale, and fall back to the LAN if the tailnet is down. +# +# ansible_host is a MagicDNS name. If tailscaled is not running on the control +# node that name does not resolve, so `nc %h %p` fails fast and the second nc +# takes over on the LAN address recorded as lan_ip in inventory.ini. +# +# This is safe against the LAN addresses drifting again (which is how +# fulcrum/mempool came to be transposed): known_hosts is keyed to the MagicDNS +# NAME, so if lan_ip ever points at a different machine the host key will not +# match and ssh aborts. Verified 2026-09-12 by pointing fulcrum-box at +# mempool-box's address: "Host key verification failed." +# +# lan_ip is a convenience, not an identity. If it goes stale the fallback stops +# working; it will never connect you to the wrong box. +ansible_ssh_common_args: >- + -o ProxyCommand="sh -c 'nc -w2 %h %p 2>/dev/null || nc -w4 {{ lan_ip }} %p'" From 27e036eccd1d5fa88f57ad22ed521cb1fe46275e Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 16:02:00 +0200 Subject: [PATCH 43/67] backup stuff --- ansible/playbooks/backups.yml | 16 ++++ ansible/roles/backup_source/README.md | 76 ++++++++++++++++ ansible/roles/backup_source/defaults/main.yml | 27 ++++++ ansible/roles/backup_source/handlers/main.yml | 4 + ansible/roles/backup_source/tasks/main.yml | 89 +++++++++++++++++++ .../backup_source/templates/backup.service.j2 | 14 +++ .../backup_source/templates/backup.sh.j2 | 69 ++++++++++++++ .../backup_source/templates/backup.timer.j2 | 11 +++ ansible/roles/backup_store/README.md | 55 ++++++++++++ ansible/roles/backup_store/defaults/main.yml | 11 +++ ansible/roles/backup_store/handlers/main.yml | 5 ++ ansible/roles/backup_store/tasks/main.yml | 53 +++++++++++ .../templates/pull-backups.service.j2 | 10 +++ .../backup_store/templates/pull-backups.sh.j2 | 43 +++++++++ .../templates/pull-backups.timer.j2 | 9 ++ .../headscale/setup_backup_headscale.yml | 21 +++++ 16 files changed, 513 insertions(+) create mode 100644 ansible/playbooks/backups.yml create mode 100644 ansible/roles/backup_source/README.md create mode 100644 ansible/roles/backup_source/defaults/main.yml create mode 100644 ansible/roles/backup_source/handlers/main.yml create mode 100644 ansible/roles/backup_source/tasks/main.yml create mode 100644 ansible/roles/backup_source/templates/backup.service.j2 create mode 100644 ansible/roles/backup_source/templates/backup.sh.j2 create mode 100644 ansible/roles/backup_source/templates/backup.timer.j2 create mode 100644 ansible/roles/backup_store/README.md create mode 100644 ansible/roles/backup_store/defaults/main.yml create mode 100644 ansible/roles/backup_store/handlers/main.yml create mode 100644 ansible/roles/backup_store/tasks/main.yml create mode 100644 ansible/roles/backup_store/templates/pull-backups.service.j2 create mode 100644 ansible/roles/backup_store/templates/pull-backups.sh.j2 create mode 100644 ansible/roles/backup_store/templates/pull-backups.timer.j2 create mode 100644 ansible/services/headscale/setup_backup_headscale.yml diff --git a/ansible/playbooks/backups.yml b/ansible/playbooks/backups.yml new file mode 100644 index 0000000..842942e --- /dev/null +++ b/ansible/playbooks/backups.yml @@ -0,0 +1,16 @@ +- name: Configure the offsite backup pull + hosts: backup_store + gather_facts: yes + + tasks: + - name: Ensure the box pulls every source on a timer + ansible.builtin.include_role: + name: backup_store + vars: + backup_store_sources: + - name: arbret + source: "arbret@prd-arbret:/opt/arbret/backups/" + retention_days: 90 + - name: headscale + source: "backup-pull@headscale.contrapeso.xyz:/opt/backups/headscale/" + retention_days: 90 diff --git a/ansible/roles/backup_source/README.md b/ansible/roles/backup_source/README.md new file mode 100644 index 0000000..065e1fc --- /dev/null +++ b/ansible/roles/backup_source/README.md @@ -0,0 +1,76 @@ +# `backup_source` + +Makes a host back **itself** up: dump to stdout, encrypt with `age`, write to a +local directory, prune, on a systemd timer. `small-backups-box` pulls the +directory later (see `backup_store`). + +Modelled on `prd-arbret`, which has been doing exactly this correctly since +before the rest of the estate was migrated. + +## Usage + +```yaml +- ansible.builtin.include_role: + name: backup_source + vars: + backup_source_name: headscale + backup_source_description: "Headscale" + backup_source_dump_command: "tar -czf - -C / var/lib/headscale etc/headscale" + backup_source_stop_service: headscale + backup_source_retention_days: 7 +``` + +Produces `/opt/backups/headscale/headscale_.tar.gz.age`, +`headscale-backup.{service,timer}`, and `/usr/local/bin/headscale-backup.sh`. + +## Why the source encrypts, not the destination + +`age -r ` is asymmetric and the host holds only the **public** key, so +a compromised host cannot read its own backups — or anyone else's. The scripts +this replaces encrypted with GPG *on the laptop, after the data had already +crossed the network*, which protects the artefact at rest but not in transit. + +The matching identity lives only on lapy and is escrowed. **Lose it and every +artefact everywhere becomes noise**, including arbret's. + +## `backup_source_dump_command` writes to STDOUT + +The role pipes it into `age`, so plaintext never touches the disk. Use `-C /` +with relative paths in `tar` rather than absolute ones: it avoids tar's "removing +leading /" and makes the restore target explicit. + +## The trap is the reason this role exists + +When `backup_source_stop_service` is set, the script stops the unit and installs +an EXIT trap that starts it again. Without it, a failed dump leaves the service +down until the next timer fires — **every hand-written script this replaced had +that bug**, and it was only ever masked because their `systemctl stop` failed +first, before anything was stopped. + +Verified on spacey: with the dump forced to fail, the log shows +`Stopping → Writing → Restarting`, the script exits 1 (so systemd marks the unit +failed rather than hiding it), and headscale is `active` afterwards. + +If `systemctl stop` itself fails, `set -e` exits *before* the trap is installed — +which is correct, because nothing was stopped. + +## `.partial` + +The dump writes `.partial` and only `mv`s it into place on success, so +a truncated file is never mistaken for a backup. A failure inside the pipeline +does leave one behind, and the prune glob cannot match it (it ends `.partial`, +not `.tar.gz.age`), so the script clears stale partials at the **start** of each +run. Tested by failing mid-pipeline: 1 partial left, 0 after the next run. + +## `backup_source_stop_service` may be a bare name + +`headscale` and `headscale.service` both work. The unit template normalises it, +because systemd rejects a bare name in `After=` with +`Failed to add dependency ... Invalid argument` — which it logs and then ignores, +so the unit appears to work while carrying no ordering at all. + +## Retention is two-tier + +`backup_source_retention_days` is **local** and short — these hosts are +disk-constrained. The long tail lives on `small-backups-box`, which decides its +own retention per source. Losing the local copy is expected and fine. diff --git a/ansible/roles/backup_source/defaults/main.yml b/ansible/roles/backup_source/defaults/main.yml new file mode 100644 index 0000000..77d17e0 --- /dev/null +++ b/ansible/roles/backup_source/defaults/main.yml @@ -0,0 +1,27 @@ +--- +# Required +backup_source_name: "" # "headscale" -> headscale_.tar.gz.age +backup_source_description: "" # "Headscale" +backup_source_dump_command: "" # must write the payload to STDOUT + +# Placement +backup_source_dir: "/opt/backups/{{ backup_source_name }}" +backup_source_artifact_suffix: "tar.gz.age" + +# Encryption. Asymmetric: the host holds only the public key and cannot decrypt +# what it produces. +backup_source_recipient: "{{ age_backup_recipient }}" + +# The unprivileged account small-backups-box pulls as. It owns the dump +# directory and nothing else; it deliberately has no sudo. +backup_source_pull_user: backup-pull +backup_source_pull_key: "{{ backup_pull_public_key }}" + +# Safety +backup_source_stop_service: "" # local unit stopped for the dump, restored by a trap + +# Retention here is LOCAL and short; small-backups-box keeps the long tail. +backup_source_retention_days: 7 + +# Schedule. The box pulls at 04:00, so dumps must land before that. +backup_source_on_calendar: "*-*-* 02:00:00" diff --git a/ansible/roles/backup_source/handlers/main.yml b/ansible/roles/backup_source/handlers/main.yml new file mode 100644 index 0000000..37a7f2b --- /dev/null +++ b/ansible/roles/backup_source/handlers/main.yml @@ -0,0 +1,4 @@ +--- +- name: Reload systemd for backup units + ansible.builtin.systemd: + daemon_reload: yes diff --git a/ansible/roles/backup_source/tasks/main.yml b/ansible/roles/backup_source/tasks/main.yml new file mode 100644 index 0000000..54d6ac1 --- /dev/null +++ b/ansible/roles/backup_source/tasks/main.yml @@ -0,0 +1,89 @@ +--- +- name: Assert backup_source parameters are sane + ansible.builtin.assert: + that: + - backup_source_name | length > 0 + - backup_source_description | length > 0 + - backup_source_dump_command | length > 0 + - backup_source_recipient | length > 0 + - backup_source_recipient is match('^age1[0-9a-z]{58}$') + fail_msg: >- + backup_source: '{{ backup_source_name | default("") }}' needs a name, + description, dump command and a valid age recipient (age1... 62 chars). + quiet: true + +# Declared here rather than assumed. Stage 1 installed it by hand; this is what +# makes a rebuilt host get it too. +- name: Ensure age is installed + ansible.builtin.apt: + name: age + state: present + update_cache: yes + cache_valid_time: 3600 + +# The pull account: unprivileged, no sudo, exists only so small-backups-box can +# read the dump directory. Trust points one way — the box can read backups, and +# can do nothing else on this host. +- name: "Ensure the {{ backup_source_pull_user }} account exists" + ansible.builtin.user: + name: "{{ backup_source_pull_user }}" + system: yes + shell: /bin/sh # rsync-over-ssh needs a shell; nologin breaks it + home: "/var/lib/{{ backup_source_pull_user }}" + create_home: yes + password: '!' # no password login, ever + when: backup_source_pull_user | length > 0 + +- name: "Authorise the backup box's key for {{ backup_source_pull_user }}" + ansible.posix.authorized_key: + user: "{{ backup_source_pull_user }}" + key: "{{ backup_source_pull_key }}" + key_options: "restrict" # no pty, no forwarding, no user rc + exclusive: yes + state: present + when: backup_source_pull_user | length > 0 + +# The shared container above the per-service directories. It must be traversable +# or the pull account cannot reach its own directory. The script's `mkdir -p` +# runs under `umask 077` and would otherwise create this 0700. +- name: "Ensure {{ backup_source_dir | dirname }} is traversable" + ansible.builtin.file: + path: "{{ backup_source_dir | dirname }}" + state: directory + owner: root + group: root + mode: '0755' + +- name: "Ensure {{ backup_source_dir }} exists" + ansible.builtin.file: + path: "{{ backup_source_dir }}" + state: directory + owner: root + group: "{{ backup_source_pull_user | default('root', true) }}" + mode: '0750' + +- name: "Install the {{ backup_source_name }} backup script" + ansible.builtin.template: + src: backup.sh.j2 + dest: "/usr/local/bin/{{ backup_source_name }}-backup.sh" + owner: root + group: root + mode: '0750' + validate: "bash -n %s" + +- name: "Install the {{ backup_source_name }}-backup systemd units" + ansible.builtin.template: + src: "backup.{{ item }}.j2" + dest: "/etc/systemd/system/{{ backup_source_name }}-backup.{{ item }}" + owner: root + group: root + mode: '0644' + loop: [service, timer] + notify: Reload systemd for backup units + +- name: "Enable the {{ backup_source_name }}-backup timer" + ansible.builtin.systemd: + name: "{{ backup_source_name }}-backup.timer" + enabled: yes + state: started + daemon_reload: yes diff --git a/ansible/roles/backup_source/templates/backup.service.j2 b/ansible/roles/backup_source/templates/backup.service.j2 new file mode 100644 index 0000000..91ad063 --- /dev/null +++ b/ansible/roles/backup_source/templates/backup.service.j2 @@ -0,0 +1,14 @@ +[Unit] +Description={{ backup_source_description }} backup +{% if backup_source_stop_service %} +{# systemd rejects a bare name here ("Failed to add dependency ... Invalid + argument"), so normalise to a full unit name. #} +After={{ backup_source_stop_service if '.' in backup_source_stop_service else backup_source_stop_service ~ '.service' }} +{% endif %} + +[Service] +Type=oneshot +ExecStart=/usr/local/bin/{{ backup_source_name }}-backup.sh +StandardOutput=journal +StandardError=journal +SyslogIdentifier={{ backup_source_name }}-backup diff --git a/ansible/roles/backup_source/templates/backup.sh.j2 b/ansible/roles/backup_source/templates/backup.sh.j2 new file mode 100644 index 0000000..35e4398 --- /dev/null +++ b/ansible/roles/backup_source/templates/backup.sh.j2 @@ -0,0 +1,69 @@ +#!/usr/bin/env bash +# {{ backup_source_description }} backup — managed by Ansible (roles/backup_source) +# +# Dumps to stdout, encrypts with age, writes {{ backup_source_dir }}. +# The host holds only the age PUBLIC key, so it cannot read its own backups. +set -euo pipefail +umask 077 + +BACKUP_DIR="{{ backup_source_dir }}" +RETENTION_DAYS={{ backup_source_retention_days }} +RECIPIENT="{{ backup_source_recipient }}" +SUFFIX="{{ backup_source_artifact_suffix }}" +NAME="{{ backup_source_name }}" +{% if backup_source_stop_service %} +SERVICE="{{ backup_source_stop_service }}" +{% endif %} + +TIMESTAMP=$(date +%Y%m%d_%H%M%S) +ARTIFACT="${BACKUP_DIR}/${NAME}_${TIMESTAMP}.${SUFFIX}" + +die() { echo "FATAL: $*" >&2; exit 1; } +log() { echo "$(date '+%Y-%m-%d %H:%M:%S') $*"; } + +# --- Pre-flight --- +[[ -n "$RECIPIENT" ]] || die "no age recipient configured" +command -v age >/dev/null || die "age is not installed" + +# Mode must agree with what the role sets, or each undoes the other every run. +mkdir -p "$BACKUP_DIR" +{% if backup_source_pull_user %} +chown root:{{ backup_source_pull_user }} "$BACKUP_DIR" +chmod 750 "$BACKUP_DIR" +{% else %} +chmod 700 "$BACKUP_DIR" +{% endif %} + +# A run that died mid-dump leaves a .partial. It is not a backup, and the prune +# glob below cannot match it (it ends .partial, not .${SUFFIX}), so clear them +# here or they accumulate forever. +rm -f "${BACKUP_DIR}/${NAME}_"*.partial + +{% if backup_source_stop_service %} +# --- Stop the service, and guarantee it comes back --- +# The trap is the point: without it a failed dump leaves the service down until +# the next timer fires. Every hand-written script this replaced had that bug. +log "Stopping ${SERVICE}..." +systemctl stop "$SERVICE" +trap 'log "Restarting ${SERVICE}..."; systemctl start "${SERVICE}" || true' EXIT +{% endif %} + +# --- Dump straight into age; plaintext never touches the disk --- +log "Writing ${ARTIFACT}..." +{{ backup_source_dump_command }} | age -r "$RECIPIENT" -o "${ARTIFACT}.partial" +mv "${ARTIFACT}.partial" "$ARTIFACT" +{% if backup_source_pull_user %} +# Readable by the pull account and nobody else. The contents are age-encrypted +# regardless, so this is depth rather than the actual protection. +chown root:{{ backup_source_pull_user }} "$ARTIFACT" +chmod 640 "$ARTIFACT" +{% else %} +chmod 600 "$ARTIFACT" +{% endif %} +log "Wrote ${ARTIFACT} ($(du -h "$ARTIFACT" | cut -f1))" + +# --- Prune --- +log "Pruning local artefacts older than ${RETENTION_DAYS} days..." +find "$BACKUP_DIR" -maxdepth 1 -type f -name "${NAME}_*.${SUFFIX}" -mtime +"${RETENTION_DAYS}" -delete + +log "Done." diff --git a/ansible/roles/backup_source/templates/backup.timer.j2 b/ansible/roles/backup_source/templates/backup.timer.j2 new file mode 100644 index 0000000..d9a7e6a --- /dev/null +++ b/ansible/roles/backup_source/templates/backup.timer.j2 @@ -0,0 +1,11 @@ +[Unit] +Description={{ backup_source_description }} backup + +[Timer] +OnCalendar={{ backup_source_on_calendar }} +# Persistent: a window missed while the host was down runs on next boot. cron on +# a laptop had no equivalent, which is how two backups went unnoticed for months. +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/roles/backup_store/README.md b/ansible/roles/backup_store/README.md new file mode 100644 index 0000000..13d9f5a --- /dev/null +++ b/ansible/roles/backup_store/README.md @@ -0,0 +1,55 @@ +# `backup_store` + +Pulls already-encrypted backup artefacts from every source host onto +`small-backups-box`, on a timer, and expires them per source. + +Generalises the hand-written `pull-backups.sh` that had one hardcoded source +(`arbret`). That job's behaviour is preserved exactly: same source path, same +90 days, same destination directory. + +## This host holds no key + +Everything pulled here is ciphertext produced by `backup_source` on the source +host. The box cannot read any of it — the age identity lives only on lapy. That +is deliberate: the machine holding every backup should not also be able to open +them. + +## One failing source must not stop the others + +The script is `set -uo pipefail`, **not** `-e`. Each source runs in its own +function, failures are counted, and the script exits non-zero at the end so +systemd marks the unit failed. A dead host costs you that one source, not the +whole run. + +This is the specific failure the whole plan exists to prevent: the laptop jobs +aborted on first error and then silently produced empty directories for nine +months. + +## Trust points one way + +The box authenticates with `~/.ssh/id_pull` to an unprivileged, dedicated +account on each source (`backup-pull`, or `arbret` on prd-arbret), authorised +with `restrict`. That account can read one directory and do nothing else — no +sudo, no pty, no forwarding. A compromised backup box cannot reach into +production. + +## Addressing: names, never IPs + +Sources are addressed by name. The job this replaced hardcoded spacey's IP; the +droplet was later rebuilt, the address was recycled to a stranger, and the +backup failed silently from 2025-12-01 while the directory listing still looked +healthy. + +Two kinds of name are in play: + +- **Tailnet members** (vipy, memos-box, …) → MagicDNS names. These require a + headscale ACL grant from `tag:small-backups-box` to the source's `:22`; without + it the box cannot even resolve the peer, let alone reach it. +- **spacey** is *not* a tailnet member — it is the headscale control server — so + its backup is pulled over the public internet via `headscale.contrapeso.xyz`, + which follows the host if the droplet is rebuilt. + +## Retention here is the long tail + +Sources keep a few days locally; this box keeps 90 (or whatever the source entry +says). Losing the source's local copy is expected. diff --git a/ansible/roles/backup_store/defaults/main.yml b/ansible/roles/backup_store/defaults/main.yml new file mode 100644 index 0000000..c0c128b --- /dev/null +++ b/ansible/roles/backup_store/defaults/main.yml @@ -0,0 +1,11 @@ +--- +backup_store_dir: "{{ ansible_env.HOME }}/backups" +backup_store_ssh_key: "{{ ansible_env.HOME }}/.ssh/id_pull" +backup_store_on_calendar: "*-*-* 04:00:00" + +# One entry per source. `retention_days` is the LONG tail; the source keeps its +# own short local retention. +# - name: headscale +# source: "backup-pull@headscale.contrapeso.xyz:/opt/backups/headscale/" +# retention_days: 90 +backup_store_sources: [] diff --git a/ansible/roles/backup_store/handlers/main.yml b/ansible/roles/backup_store/handlers/main.yml new file mode 100644 index 0000000..632be35 --- /dev/null +++ b/ansible/roles/backup_store/handlers/main.yml @@ -0,0 +1,5 @@ +--- +- name: Reload systemd for pull-backups + ansible.builtin.systemd: + daemon_reload: yes + become: yes diff --git a/ansible/roles/backup_store/tasks/main.yml b/ansible/roles/backup_store/tasks/main.yml new file mode 100644 index 0000000..d138830 --- /dev/null +++ b/ansible/roles/backup_store/tasks/main.yml @@ -0,0 +1,53 @@ +--- +- name: Assert backup_store sources are sane + ansible.builtin.assert: + that: + - backup_store_sources | length > 0 + - backup_store_sources | map(attribute='name') | list | length == backup_store_sources | length + - backup_store_sources | map(attribute='source') | list | length == backup_store_sources | length + - backup_store_sources | map(attribute='retention_days') | list | length == backup_store_sources | length + fail_msg: "backup_store: every source needs name, source and retention_days" + quiet: true + +- name: Ensure rsync is installed + ansible.builtin.apt: + name: rsync + state: present + update_cache: yes + cache_valid_time: 3600 + become: yes + +- name: Ensure the backup store directory exists + ansible.builtin.file: + path: "{{ backup_store_dir }}" + state: directory + mode: '0700' + +- name: Install the pull-backups script + ansible.builtin.template: + src: pull-backups.sh.j2 + dest: /usr/local/bin/pull-backups.sh + owner: root + group: root + mode: '0755' + validate: "bash -n %s" + become: yes + +- name: Install the pull-backups systemd units + ansible.builtin.template: + src: "pull-backups.{{ item }}.j2" + dest: "/etc/systemd/system/pull-backups.{{ item }}" + owner: root + group: root + mode: '0644' + loop: [service, timer] + become: yes + notify: Reload systemd for pull-backups + +- name: Enable the pull-backups timer + ansible.builtin.systemd: + name: pull-backups.timer + enabled: yes + state: started + daemon_reload: yes + become: yes diff --git a/ansible/roles/backup_store/templates/pull-backups.service.j2 b/ansible/roles/backup_store/templates/pull-backups.service.j2 new file mode 100644 index 0000000..c65516c --- /dev/null +++ b/ansible/roles/backup_store/templates/pull-backups.service.j2 @@ -0,0 +1,10 @@ +[Unit] +Description=Pull encrypted backups from production + +[Service] +Type=oneshot +User={{ ansible_user_id }} +ExecStart=/usr/local/bin/pull-backups.sh +StandardOutput=journal +StandardError=journal +SyslogIdentifier=pull-backups diff --git a/ansible/roles/backup_store/templates/pull-backups.sh.j2 b/ansible/roles/backup_store/templates/pull-backups.sh.j2 new file mode 100644 index 0000000..267900c --- /dev/null +++ b/ansible/roles/backup_store/templates/pull-backups.sh.j2 @@ -0,0 +1,43 @@ +#!/usr/bin/env bash +# Pull encrypted backups from production — managed by Ansible (roles/backup_store) +# +# Everything here is already ciphertext: this host only moves and expires files, +# and holds no key that can read them. +set -uo pipefail # deliberately NOT -e; see the loop below + +SSH_KEY="{{ backup_store_ssh_key }}" +STORE="{{ backup_store_dir }}" + +log() { echo "$(date '+%Y-%m-%d %H:%M:%S') $*"; } +fail() { echo "$(date '+%Y-%m-%d %H:%M:%S') ERROR: $*" >&2; failures=$((failures + 1)); } + +failures=0 + +# One source failing must not stop the others. The whole point of this box is +# that a single dead host cannot silently take the rest of the backups with it — +# which is exactly how the laptop-based jobs failed unnoticed for nine months. +{% for src in backup_store_sources %} +# --- {{ src.name }} --- +pull_{{ src.name | replace('-', '_') }}() { + local dir="${STORE}/{{ src.name }}" + mkdir -p "$dir" + log "Pulling {{ src.name }} from {{ src.source }}..." + if rsync -az --timeout=120 \ + -e "ssh -i $SSH_KEY -o StrictHostKeyChecking=accept-new -o ConnectTimeout=15" \ + "{{ src.source }}" "$dir/"; then + log " {{ src.name }}: ok ($(find "$dir" -maxdepth 1 -type f | wc -l) artefacts, $(du -sh "$dir" | cut -f1))" + else + fail "{{ src.name }}: rsync failed" + return 1 + fi + log " {{ src.name }}: pruning older than {{ src.retention_days }} days" + find "$dir" -maxdepth 1 -type f -name '{{ src.name }}_*' -mtime +{{ src.retention_days }} -delete +} +pull_{{ src.name | replace('-', '_') }} || true + +{% endfor %} +if [ "$failures" -gt 0 ]; then + log "FAILED: $failures source(s) did not pull" + exit 1 +fi +log "All sources pulled." diff --git a/ansible/roles/backup_store/templates/pull-backups.timer.j2 b/ansible/roles/backup_store/templates/pull-backups.timer.j2 new file mode 100644 index 0000000..336e1ca --- /dev/null +++ b/ansible/roles/backup_store/templates/pull-backups.timer.j2 @@ -0,0 +1,9 @@ +[Unit] +Description=Daily offsite backup pull + +[Timer] +OnCalendar={{ backup_store_on_calendar }} +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/services/headscale/setup_backup_headscale.yml b/ansible/services/headscale/setup_backup_headscale.yml new file mode 100644 index 0000000..9c72d1d --- /dev/null +++ b/ansible/services/headscale/setup_backup_headscale.yml @@ -0,0 +1,21 @@ +--- +- name: Configure the Headscale backup on the vpn_control host + hosts: vpn_control + become: yes + vars_files: + - ../../group_vars/all/main.yml + - ./headscale_vars.yml + + tasks: + - name: Ensure Headscale dumps itself, encrypted, on a timer + ansible.builtin.include_role: + name: backup_source + vars: + backup_source_name: headscale + backup_source_description: "Headscale" + # -C / with relative paths: avoids tar's "removing leading /" and makes + # the restore target explicit. + backup_source_dump_command: "tar -czf - -C / var/lib/headscale etc/headscale" + backup_source_stop_service: headscale + backup_source_retention_days: 7 + backup_source_on_calendar: "*-*-* 02:00:00" From 394f2519ff261f518473f6df171c736d51a164a2 Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 16:20:42 +0200 Subject: [PATCH 44/67] age backups everywhere --- ansible/group_vars/all/main.yml | 7 +++ ansible/group_vars/nodito_vms.yml | 23 ++++++--- ansible/playbooks/backups.yml | 12 +++++ ansible/roles/backup_source/README.md | 50 +++++++++++++++++++ ansible/roles/backup_source/defaults/main.yml | 8 ++- ansible/roles/backup_source/tasks/main.yml | 2 + .../backup_source/templates/backup.sh.j2 | 12 +++-- .../services/forgejo/setup_backup_forgejo.yml | 25 ++++++++++ .../services/lnbits/setup_backup_lnbits.yml | 25 ++++++++++ ansible/services/memos/setup_backup_memos.yml | 26 ++++++++++ .../vaultwarden/setup_backup_vaultwarden.yml | 25 ++++++++++ 11 files changed, 200 insertions(+), 15 deletions(-) create mode 100644 ansible/services/forgejo/setup_backup_forgejo.yml create mode 100644 ansible/services/lnbits/setup_backup_lnbits.yml create mode 100644 ansible/services/memos/setup_backup_memos.yml create mode 100644 ansible/services/vaultwarden/setup_backup_vaultwarden.yml diff --git a/ansible/group_vars/all/main.yml b/ansible/group_vars/all/main.yml index 36d35f8..0be0162 100644 --- a/ansible/group_vars/all/main.yml +++ b/ansible/group_vars/all/main.yml @@ -7,3 +7,10 @@ root_domain: contrapeso.xyz # playbooks are kept deliberately — the check logic is meant to be rewired to # whatever replaces it. This flag keeps them inert until then. See archive/uptime_kuma/. uptime_kuma_enabled: false + +# age recipient for all backup artefacts +age_backup_recipient: "age192wwdaseqej2ggwyp884gtm05c396anp7chr0vr8m47g50fahpyqr9fsza" + +# Public key small-backups-box pulls with +# Authorised on each source host for an unprivileged, dedicated user only +backup_pull_public_key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIOfIixKMhA9z+Nvyx6ToZIniC8aEgyiInRiboaTTemgX offsite-backup-pull" diff --git a/ansible/group_vars/nodito_vms.yml b/ansible/group_vars/nodito_vms.yml index 2ff6994..1072b51 100644 --- a/ansible/group_vars/nodito_vms.yml +++ b/ansible/group_vars/nodito_vms.yml @@ -2,16 +2,23 @@ # Reach the VMs over Tailscale, and fall back to the LAN if the tailnet is down. # # ansible_host is a MagicDNS name. If tailscaled is not running on the control -# node that name does not resolve, so `nc %h %p` fails fast and the second nc -# takes over on the LAN address recorded as lan_ip in inventory.ini. +# node that name does not resolve, the probe fails, and the LAN address recorded +# as lan_ip in inventory.ini takes over. # -# This is safe against the LAN addresses drifting again (which is how -# fulcrum/mempool came to be transposed): known_hosts is keyed to the MagicDNS -# NAME, so if lan_ip ever points at a different machine the host key will not -# match and ssh aborts. Verified 2026-09-12 by pointing fulcrum-box at -# mempool-box's address: "Host key verification failed." +# Why the probe-then-connect shape rather than a plain `nc -w5 %h %p`: +# netcat-openbsd's -w is an IDLE timeout as well as a connect timeout, so a +# single `nc -w5` silently tears down the SSH session after five quiet seconds. +# That produced intermittent "Data could not be sent to remote host" failures on +# exactly the long, quiet operations (apt) where a dropped connection costs most. +# `nc -z` probes, then `exec nc` carries the session with no timeout at all. +# +# Safe against the LAN addresses drifting again (which is how fulcrum/mempool +# came to be transposed): known_hosts is keyed to the MagicDNS NAME, so if +# lan_ip ever points at a different machine the host key will not match and ssh +# aborts. Verified by pointing fulcrum-box at mempool-box's address: +# "Host key verification failed." # # lan_ip is a convenience, not an identity. If it goes stale the fallback stops # working; it will never connect you to the wrong box. ansible_ssh_common_args: >- - -o ProxyCommand="sh -c 'nc -w2 %h %p 2>/dev/null || nc -w4 {{ lan_ip }} %p'" + -o ProxyCommand="sh -c 'nc -z -w5 %h %p 2>/dev/null && exec nc %h %p || exec nc {{ lan_ip }} %p'" diff --git a/ansible/playbooks/backups.yml b/ansible/playbooks/backups.yml index 842942e..0686ab6 100644 --- a/ansible/playbooks/backups.yml +++ b/ansible/playbooks/backups.yml @@ -14,3 +14,15 @@ - name: headscale source: "backup-pull@headscale.contrapeso.xyz:/opt/backups/headscale/" retention_days: 90 + - name: memos + source: "backup-pull@memos-box:/opt/backups/memos/" + retention_days: 90 + - name: vaultwarden + source: "backup-pull@prd-vipy:/opt/backups/vaultwarden/" + retention_days: 90 + - name: lnbits + source: "backup-pull@prd-vipy:/opt/backups/lnbits/" + retention_days: 90 + - name: forgejo + source: "backup-pull@prd-vipy:/opt/backups/forgejo/" + retention_days: 14 diff --git a/ansible/roles/backup_source/README.md b/ansible/roles/backup_source/README.md index 065e1fc..a7794ab 100644 --- a/ansible/roles/backup_source/README.md +++ b/ansible/roles/backup_source/README.md @@ -39,6 +39,56 @@ The role pipes it into `age`, so plaintext never touches the disk. Use `-C /` with relative paths in `tar` rather than absolute ones: it avoids tar's "removing leading /" and makes the restore target explicit. +## Services that are not systemd + +`backup_source_stop_service` runs `systemctl stop/start`. For anything else, +give the pair explicitly — vaultwarden is a docker compose stack, so +`systemctl stop vaultwarden` silently does nothing: + +```yaml +backup_source_stop_command: "docker compose -f /opt/vaultwarden/docker-compose.yml stop" +backup_source_start_command: "docker compose -f /opt/vaultwarden/docker-compose.yml start" +``` + +The same EXIT trap wraps both forms. The assert refuses a stop command without a +matching start command, because that combination fails in the one way you would +not notice: the service stops and never comes back. + +## More than one thing to back up + +`tar` takes several paths, so multiple files or directories are normally **one** +artefact — headscale captures `/var/lib/headscale` and `/etc/headscale` together, +lnbits captures its data directory and its `.env`. + +Prefer one artefact. A backup should be a consistent snapshot, and two artefacts +written by two runs can drift — you can end up restoring an `.env` that does not +match the database it configures. Pulling a single file back out needs no +unpacking: + +```bash +age -d -i | tar -xzO opt/lnbits/lnbits/.env +``` + +If you genuinely need separate artefacts, call the role twice with different +`backup_source_name`s rather than extending it — but only one call may set +`backup_source_stop_service`, or the service is stopped twice per night. + +The case this shape cannot express is a **database dump plus a file tree** +(`pg_dump` and a media directory, say): you cannot merge those into one stream +without staging plaintext on disk, which is exactly what this design avoids. +None of the current services need it — all are file trees, all stopped for the +dump. A future one that does should use two role calls. + +## Everything here is sqlite, so everything stops + +All five services are sqlite-backed, several in WAL mode (`-wal`/`-shm` files +present). A live copy of a WAL-mode database can be torn or stale, so each is +stopped for the duration. Measured downtime: under a second for headscale and +memos, ~6 s vaultwarden, ~11 s lnbits, and **2m36s for forgejo** — 2.7 G of repos +and database. That last one is the real cost of a consistent snapshot; if it +becomes unacceptable the answer is `sqlite3 .backup` plus an online repo copy, +not skipping the stop. + ## The trap is the reason this role exists When `backup_source_stop_service` is set, the script stops the unit and installs diff --git a/ansible/roles/backup_source/defaults/main.yml b/ansible/roles/backup_source/defaults/main.yml index 77d17e0..7de6006 100644 --- a/ansible/roles/backup_source/defaults/main.yml +++ b/ansible/roles/backup_source/defaults/main.yml @@ -17,8 +17,12 @@ backup_source_recipient: "{{ age_backup_recipient }}" backup_source_pull_user: backup-pull backup_source_pull_key: "{{ backup_pull_public_key }}" -# Safety -backup_source_stop_service: "" # local unit stopped for the dump, restored by a trap +# Safety. Give either a systemd unit, or an explicit pair of commands for +# services that are not systemd-managed (vaultwarden is a docker compose stack). +# Whichever is used, a trap guarantees the restart. +backup_source_stop_service: "" # systemd unit stopped for the dump +backup_source_stop_command: "" # overrides stop_service when set +backup_source_start_command: "" # required alongside stop_command # Retention here is LOCAL and short; small-backups-box keeps the long tail. backup_source_retention_days: 7 diff --git a/ansible/roles/backup_source/tasks/main.yml b/ansible/roles/backup_source/tasks/main.yml index 54d6ac1..4283b84 100644 --- a/ansible/roles/backup_source/tasks/main.yml +++ b/ansible/roles/backup_source/tasks/main.yml @@ -7,9 +7,11 @@ - backup_source_dump_command | length > 0 - backup_source_recipient | length > 0 - backup_source_recipient is match('^age1[0-9a-z]{58}$') + - not (backup_source_stop_command | length > 0 and backup_source_start_command | length == 0) fail_msg: >- backup_source: '{{ backup_source_name | default("") }}' needs a name, description, dump command and a valid age recipient (age1... 62 chars). + backup_source_stop_command must be paired with backup_source_start_command. quiet: true # Declared here rather than assumed. Stage 1 installed it by hand; this is what diff --git a/ansible/roles/backup_source/templates/backup.sh.j2 b/ansible/roles/backup_source/templates/backup.sh.j2 index 35e4398..18fabd6 100644 --- a/ansible/roles/backup_source/templates/backup.sh.j2 +++ b/ansible/roles/backup_source/templates/backup.sh.j2 @@ -11,8 +11,10 @@ RETENTION_DAYS={{ backup_source_retention_days }} RECIPIENT="{{ backup_source_recipient }}" SUFFIX="{{ backup_source_artifact_suffix }}" NAME="{{ backup_source_name }}" -{% if backup_source_stop_service %} -SERVICE="{{ backup_source_stop_service }}" +{% if backup_source_stop_service or backup_source_stop_command %} +STOP_CMD={{ (backup_source_stop_command or ('systemctl stop ' ~ backup_source_stop_service)) | quote }} +START_CMD={{ (backup_source_start_command or ('systemctl start ' ~ backup_source_stop_service)) | quote }} +SERVICE="{{ backup_source_stop_service or backup_source_description }}" # label for the log only {% endif %} TIMESTAMP=$(date +%Y%m%d_%H%M%S) @@ -39,13 +41,13 @@ chmod 700 "$BACKUP_DIR" # here or they accumulate forever. rm -f "${BACKUP_DIR}/${NAME}_"*.partial -{% if backup_source_stop_service %} +{% if backup_source_stop_service or backup_source_stop_command %} # --- Stop the service, and guarantee it comes back --- # The trap is the point: without it a failed dump leaves the service down until # the next timer fires. Every hand-written script this replaced had that bug. log "Stopping ${SERVICE}..." -systemctl stop "$SERVICE" -trap 'log "Restarting ${SERVICE}..."; systemctl start "${SERVICE}" || true' EXIT +eval "$STOP_CMD" +trap 'log "Restarting ${SERVICE}..."; eval "$START_CMD" || true' EXIT {% endif %} # --- Dump straight into age; plaintext never touches the disk --- diff --git a/ansible/services/forgejo/setup_backup_forgejo.yml b/ansible/services/forgejo/setup_backup_forgejo.yml new file mode 100644 index 0000000..d6aafef --- /dev/null +++ b/ansible/services/forgejo/setup_backup_forgejo.yml @@ -0,0 +1,25 @@ +--- +# Forgejo backup: dumps locally on vipy, encrypted with age. +# +# The biggest artefact in the estate (~2.7 G) and the reason retention here is +# short: 7 days locally would be 19 G of vipy's 36 G free. The box keeps 14. +# Forgejo is sqlite3 (DB_TYPE in app.ini), so it is stopped for the dump — the +# old job did the same. +- name: Configure the Forgejo backup on the edge host + hosts: edge + become: yes + vars_files: + - ../../group_vars/all/main.yml + - ./forgejo_vars.yml + + tasks: + - name: Ensure Forgejo dumps itself, encrypted, on a timer + ansible.builtin.include_role: + name: backup_source + vars: + backup_source_name: forgejo + backup_source_description: "Forgejo" + backup_source_dump_command: "tar -czf - -C / var/lib/forgejo etc/forgejo" + backup_source_stop_service: forgejo + backup_source_retention_days: 2 + backup_source_on_calendar: "*-*-* 02:30:00" diff --git a/ansible/services/lnbits/setup_backup_lnbits.yml b/ansible/services/lnbits/setup_backup_lnbits.yml new file mode 100644 index 0000000..46abf31 --- /dev/null +++ b/ansible/services/lnbits/setup_backup_lnbits.yml @@ -0,0 +1,25 @@ +--- +# LNBits backup: dumps locally on vipy, encrypted with age. +# The old job produced TWO gpg artefacts (data, then .env separately). They are +# folded into one tar here so the wallet database and the .env that configures +# it are always the same point in time; two artefacts written by two runs can +# drift. Pulling one file back out needs no unpacking: +# age -d -i | tar -xzO opt/lnbits/lnbits/.env +- name: Configure the LNBits backup on the edge host + hosts: edge + become: yes + vars_files: + - ../../group_vars/all/main.yml + - ./lnbits_vars.yml + + tasks: + - name: Ensure LNBits dumps itself, encrypted, on a timer + ansible.builtin.include_role: + name: backup_source + vars: + backup_source_name: lnbits + backup_source_description: "LNBits" + backup_source_dump_command: "tar -czf - -C / opt/lnbits/data opt/lnbits/lnbits/.env" + backup_source_stop_service: lnbits + backup_source_retention_days: 7 + backup_source_on_calendar: "*-*-* 02:20:00" diff --git a/ansible/services/memos/setup_backup_memos.yml b/ansible/services/memos/setup_backup_memos.yml new file mode 100644 index 0000000..2131924 --- /dev/null +++ b/ansible/services/memos/setup_backup_memos.yml @@ -0,0 +1,26 @@ +--- +# Memos backup: dumps locally on memos-box, encrypted with age. +# Replaces the lapy pull, which had been writing EMPTY directories since +# 2025-12-27 — its script hardcoded 192.168.1.130, which DHCP later reassigned +# to a different machine that has no rsync. +- name: Configure the Memos backup on its own host + hosts: memos + become: yes + vars_files: + - ../../group_vars/all/main.yml + - ./memos_vars.yml + + tasks: + - name: Ensure Memos dumps itself, encrypted, on a timer + ansible.builtin.include_role: + name: backup_source + vars: + backup_source_name: memos + backup_source_description: "Memos" + backup_source_dump_command: "tar -czf - -C / var/lib/memos" + # sqlite in WAL mode: stopping checkpoints the WAL, so the artefact is a + # consistent database rather than a torn mid-write copy. The old rsync + # job did not stop it. + backup_source_stop_service: memos + backup_source_retention_days: 7 + backup_source_on_calendar: "*-*-* 02:00:00" diff --git a/ansible/services/vaultwarden/setup_backup_vaultwarden.yml b/ansible/services/vaultwarden/setup_backup_vaultwarden.yml new file mode 100644 index 0000000..8430475 --- /dev/null +++ b/ansible/services/vaultwarden/setup_backup_vaultwarden.yml @@ -0,0 +1,25 @@ +--- +# Vaultwarden backup: dumps locally on vipy, encrypted with age. +# Previously rsynced to lapy in the CLEAR; the artefact now never exists +# unencrypted, on disk or on the wire. +- name: Configure the Vaultwarden backup on the edge host + hosts: edge + become: yes + vars_files: + - ../../group_vars/all/main.yml + - ./vaultwarden_vars.yml + + tasks: + - name: Ensure Vaultwarden dumps itself, encrypted, on a timer + ansible.builtin.include_role: + name: backup_source + vars: + backup_source_name: vaultwarden + backup_source_description: "Vaultwarden" + backup_source_dump_command: "tar -czf - -C / opt/vaultwarden/data" + # Not systemd — a docker compose stack — so stop/start explicitly. + # sqlite in WAL mode, hence stopping at all. + backup_source_stop_command: "docker compose -f /opt/vaultwarden/docker-compose.yml stop" + backup_source_start_command: "docker compose -f /opt/vaultwarden/docker-compose.yml start" + backup_source_retention_days: 7 + backup_source_on_calendar: "*-*-* 02:10:00" From e9bb90f8f8f35d5eabf3361596ee02e065181222 Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 17:16:26 +0200 Subject: [PATCH 45/67] clean up old backing up --- ansible/inventory.ini | 2 +- ansible/services/forgejo/forgejo_vars.yml | 3 - .../forgejo/setup_backup_forgejo_to_lapy.yml | 122 ---------------- ansible/services/headscale/headscale_vars.yml | 4 - .../setup_backup_headscale_to_lapy.yml | 75 ---------- ansible/services/lnbits/lnbits_vars.yml | 3 - .../lnbits/setup_backup_lnbits_to_lapy.yml | 104 -------------- ansible/services/memos/memos_vars.yml | 8 -- .../memos/setup_backup_memos_to_lapy.yml | 106 -------------- ansible/services/phoenixd/phoenixd_vars.yml | 3 - .../setup_backup_phoenixd_to_lapy.yml | 136 ------------------ .../setup_backup_vaultwarden_to_lapy.yml | 105 -------------- .../services/vaultwarden/vaultwarden_vars.yml | 3 - 13 files changed, 1 insertion(+), 673 deletions(-) delete mode 100644 ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml delete mode 100644 ansible/services/headscale/setup_backup_headscale_to_lapy.yml delete mode 100644 ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml delete mode 100644 ansible/services/memos/setup_backup_memos_to_lapy.yml delete mode 100644 ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml delete mode 100644 ansible/services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml diff --git a/ansible/inventory.ini b/ansible/inventory.ini index b9b2906..3d4a352 100644 --- a/ansible/inventory.ini +++ b/ansible/inventory.ini @@ -19,7 +19,7 @@ nonkeiwaisi_local ansible_host=nonkeiwaisi-box lan_ip=192.168.1.151 ansible_user # Local connection to laptop: this assumes you're running ansible commands from your personal laptop [lapy] -localhost ansible_connection=local ansible_user=counterweight gpg_recipient=counterweightoperator@protonmail.com gpg_key_id=883EDBAA726BD96C +localhost ansible_connection=local ansible_user=counterweight [arbret] prd-arbret ansible_host=167.99.242.62 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua diff --git a/ansible/services/forgejo/forgejo_vars.yml b/ansible/services/forgejo/forgejo_vars.yml index 2a24133..9fb4cc9 100644 --- a/ansible/services/forgejo/forgejo_vars.yml +++ b/ansible/services/forgejo/forgejo_vars.yml @@ -18,6 +18,3 @@ remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counter remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" remote_port: "{{ hostvars.get(remote_host_name, {}).get('ansible_port', 22) }}" -# Local backup -local_backup_dir: "{{ lookup('env', 'HOME') }}/forgejo-backups" -backup_script_path: "{{ lookup('env', 'HOME') }}/.local/bin/forgejo_backup.sh" diff --git a/ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml b/ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml deleted file mode 100644 index c3ff8d6..0000000 --- a/ansible/services/forgejo/setup_backup_forgejo_to_lapy.yml +++ /dev/null @@ -1,122 +0,0 @@ ---- -- name: Configure local backup for Forgejo from remote - hosts: control - gather_facts: no - vars_files: - - ../../infra_vars.yml - - ./forgejo_vars.yml - vars: - remote_data_path: "{{ forgejo_data_dir }}" - remote_config_path: "{{ forgejo_config_dir }}" - forgejo_service_name: "forgejo" - gpg_recipient: "{{ hostvars['localhost']['gpg_recipient'] | default('') }}" - gpg_key_id: "{{ hostvars['localhost']['gpg_key_id'] | default('') }}" - - tasks: - - name: Debug Forgejo backup vars - debug: - msg: - - "remote_host={{ remote_host }}" - - "remote_user={{ remote_user }}" - - "remote_data_path='{{ remote_data_path }}'" - - "remote_config_path='{{ remote_config_path }}'" - - "local_backup_dir={{ local_backup_dir }}" - - "gpg_recipient={{ gpg_recipient }}" - - "gpg_key_id={{ gpg_key_id }}" - - - name: Ensure local backup directory exists - ansible.builtin.file: - path: "{{ local_backup_dir }}" - state: directory - mode: "0755" - - - name: Ensure ~/.local/bin exists - ansible.builtin.file: - path: "{{ lookup('env', 'HOME') }}/.local/bin" - state: directory - mode: "0755" - - - name: Create Forgejo backup script - ansible.builtin.copy: - dest: "{{ backup_script_path }}" - mode: "0750" - content: | - #!/bin/bash - set -euo pipefail - - if [ -z "{{ gpg_recipient }}" ]; then - echo "GPG recipient is not configured. Aborting." - exit 1 - fi - - TIMESTAMP=$(date +'%Y-%m-%d') - ENCRYPTED_BACKUP="{{ local_backup_dir }}/forgejo-backup-$TIMESTAMP.tar.gz.gpg" - - {% if remote_key_file %} - SSH_CMD="ssh -i {{ remote_key_file }} -p {{ remote_port }}" - {% else %} - SSH_CMD="ssh -p {{ remote_port }}" - {% endif %} - - echo "Stopping Forgejo service..." - $SSH_CMD {{ remote_user }}@{{ remote_host }} "sudo systemctl stop {{ forgejo_service_name }}" - - echo "Creating encrypted backup archive..." - $SSH_CMD {{ remote_user }}@{{ remote_host }} "sudo tar -czf - {{ remote_data_path }} {{ remote_config_path }}" | \ - gpg --batch --yes --encrypt --recipient "{{ gpg_recipient }}" --output "$ENCRYPTED_BACKUP" - - echo "Starting Forgejo service..." - $SSH_CMD {{ remote_user }}@{{ remote_host }} "sudo systemctl start {{ forgejo_service_name }}" - - # Rotate old backups (keep 3 days) - # Calculate cutoff date (3 days ago) and delete backups older than that - CUTOFF_DATE=$(date -d '3 days ago' +'%Y-%m-%d') - for backup_file in "{{ local_backup_dir }}"/forgejo-backup-*.tar.gz.gpg; do - if [ -f "$backup_file" ]; then - # Extract date from filename: forgejo-backup-YYYY-MM-DD.tar.gz.gpg - file_date=$(basename "$backup_file" | sed -n 's/forgejo-backup-\([0-9]\{4\}-[0-9]\{2\}-[0-9]\{2\}\)\.tar\.gz\.gpg/\1/p') - if [ -n "$file_date" ] && [ "$file_date" != "$TIMESTAMP" ] && [ "$file_date" \< "$CUTOFF_DATE" ]; then - rm -f "$backup_file" - fi - fi - done - - echo "Backup completed successfully" - - - name: Ensure cronjob for Forgejo backup exists - ansible.builtin.cron: - name: "Forgejo backup" - user: "{{ lookup('env', 'USER') }}" - job: "{{ backup_script_path }}" - minute: 5 - hour: "9,12,15,18" - - - name: Run Forgejo backup script to create initial backup - ansible.builtin.command: "{{ backup_script_path }}" - - - name: Verify backup was created - block: - - name: Get today's date - command: date +'%Y-%m-%d' - register: today_date - changed_when: false - - - name: Check if backup file exists - stat: - path: "{{ local_backup_dir }}/forgejo-backup-{{ today_date.stdout }}.tar.gz.gpg" - register: backup_file_stat - - - name: Verify backup file exists - assert: - that: - - backup_file_stat.stat.exists - - backup_file_stat.stat.isreg - fail_msg: "Backup file {{ local_backup_dir }}/forgejo-backup-{{ today_date.stdout }}.tar.gz.gpg was not created" - success_msg: "Backup file {{ local_backup_dir }}/forgejo-backup-{{ today_date.stdout }}.tar.gz.gpg exists" - - - name: Verify backup file is not empty - assert: - that: - - backup_file_stat.stat.size > 0 - fail_msg: "Backup file {{ local_backup_dir }}/forgejo-backup-{{ today_date.stdout }}.tar.gz.gpg exists but is empty" - success_msg: "Backup file size is {{ backup_file_stat.stat.size }} bytes" diff --git a/ansible/services/headscale/headscale_vars.yml b/ansible/services/headscale/headscale_vars.yml index 05fc040..ab12a82 100644 --- a/ansible/services/headscale/headscale_vars.yml +++ b/ansible/services/headscale/headscale_vars.yml @@ -19,9 +19,5 @@ remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counter remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" remote_port: "{{ hostvars.get(remote_host_name, {}).get('ansible_port', 22) }}" -# Local backup -local_backup_dir: "{{ lookup('env', 'HOME') }}/headscale-backups" -backup_script_path: "{{ lookup('env', 'HOME') }}/.local/bin/headscale_backup.sh" - # Headplane (headscale admin UI), proxied at /admin* behind Caddy basic auth headplane_port: 3000 diff --git a/ansible/services/headscale/setup_backup_headscale_to_lapy.yml b/ansible/services/headscale/setup_backup_headscale_to_lapy.yml deleted file mode 100644 index acc82ce..0000000 --- a/ansible/services/headscale/setup_backup_headscale_to_lapy.yml +++ /dev/null @@ -1,75 +0,0 @@ -- name: Configure local backup for Headscale from remote - hosts: control - gather_facts: no - vars_files: - - ../../infra_vars.yml - - ./headscale_vars.yml - vars: - remote_data_path: "{{ headscale_data_dir }}" - remote_config_path: "/etc/headscale" - - tasks: - - name: Debug remote backup vars - debug: - msg: - - "remote_host={{ remote_host }}" - - "remote_user={{ remote_user }}" - - "remote_data_path='{{ remote_data_path }}'" - - "remote_config_path='{{ remote_config_path }}'" - - "local_backup_dir={{ local_backup_dir }}" - - - name: Ensure local backup directory exists - file: - path: "{{ local_backup_dir }}" - state: directory - mode: '0755' - - - name: Ensure ~/.local/bin exists - file: - path: "{{ lookup('env', 'HOME') }}/.local/bin" - state: directory - mode: '0755' - - - name: Create backup script - copy: - dest: "{{ backup_script_path }}" - mode: '0750' - content: | - #!/bin/bash - set -euo pipefail - - TIMESTAMP=$(date +'%Y-%m-%d') - BACKUP_DIR="{{ local_backup_dir }}/$TIMESTAMP" - mkdir -p "$BACKUP_DIR" - - {% if remote_key_file %} - SSH_CMD="ssh -i {{ remote_key_file }} -p {{ remote_port }}" - {% else %} - SSH_CMD="ssh -p {{ remote_port }}" - {% endif %} - - # Stop headscale service for consistent backup - $SSH_CMD {{ remote_user }}@{{ remote_host }} "sudo systemctl stop headscale" - - # Backup data directory - rsync -az -e "$SSH_CMD" --delete {{ remote_user }}@{{ remote_host }}:{{ remote_data_path }}/ "$BACKUP_DIR/data/" - - # Backup config directory - rsync -az -e "$SSH_CMD" --delete {{ remote_user }}@{{ remote_host }}:{{ remote_config_path }}/ "$BACKUP_DIR/config/" - - # Start headscale service again - $SSH_CMD {{ remote_user }}@{{ remote_host }} "sudo systemctl start headscale" - - # Rotate old backups (keep 14 days) - find "{{ local_backup_dir }}" -maxdepth 1 -type d -name '20*' -mtime +13 -exec rm -rf {} \; - - - name: Ensure cronjob for backup exists - cron: - name: "Headscale backup" - user: "{{ lookup('env', 'USER') }}" - job: "{{ backup_script_path }}" - minute: 5 - hour: "9,12,15,18" - - - name: Run the backup script to make the first backup - command: "{{ backup_script_path }}" diff --git a/ansible/services/lnbits/lnbits_vars.yml b/ansible/services/lnbits/lnbits_vars.yml index eabd6fc..e855d57 100644 --- a/ansible/services/lnbits/lnbits_vars.yml +++ b/ansible/services/lnbits/lnbits_vars.yml @@ -12,6 +12,3 @@ remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counter remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" remote_port: "{{ hostvars.get(remote_host_name, {}).get('ansible_port', 22) }}" -# Local backup -local_backup_dir: "{{ lookup('env', 'HOME') }}/lnbits-backups" -backup_script_path: "{{ lookup('env', 'HOME') }}/.local/bin/lnbits_backup.sh" diff --git a/ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml b/ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml deleted file mode 100644 index 48296cb..0000000 --- a/ansible/services/lnbits/setup_backup_lnbits_to_lapy.yml +++ /dev/null @@ -1,104 +0,0 @@ -- name: Configure local backup for LNBits from remote - hosts: control - gather_facts: no - vars_files: - - ../../infra_vars.yml - - ./lnbits_vars.yml - vars: - remote_data_path: "{{ lnbits_data_dir }}" - remote_lnbits_dir: "{{ lnbits_dir }}/lnbits" - gpg_recipient: "{{ hostvars['localhost']['gpg_recipient'] | default('') }}" - gpg_key_id: "{{ hostvars['localhost']['gpg_key_id'] | default('') }}" - - tasks: - - name: Debug remote backup vars - debug: - msg: - - "remote_host={{ remote_host }}" - - "remote_user={{ remote_user }}" - - "remote_data_path='{{ remote_data_path }}'" - - "local_backup_dir={{ local_backup_dir }}" - - "gpg_recipient={{ gpg_recipient }}" - - "gpg_key_id={{ gpg_key_id }}" - - - name: Ensure local backup directory exists - file: - path: "{{ local_backup_dir }}" - state: directory - mode: '0755' - - - name: Ensure ~/.local/bin exists - file: - path: "{{ lookup('env', 'HOME') }}/.local/bin" - state: directory - mode: '0755' - - - name: Create backup script - copy: - dest: "{{ backup_script_path }}" - mode: '0750' - content: | - #!/bin/bash - set -euo pipefail - - TIMESTAMP=$(date +'%Y-%m-%d') - ENCRYPTED_BACKUP="{{ local_backup_dir }}/lnbits-backup-$TIMESTAMP.tar.gz.gpg" - - {% if remote_key_file %} - SSH_CMD="ssh -i {{ remote_key_file }} -p {{ remote_port }}" - {% else %} - SSH_CMD="ssh -p {{ remote_port }}" - {% endif %} - - # Stop LNBits service before backup - echo "Stopping LNBits service..." - $SSH_CMD {{ remote_user }}@{{ remote_host }} "sudo systemctl stop lnbits.service" - - # Create encrypted backup on the fly - # First, create a tar archive of the data directory and pipe it through gpg - echo "Creating backup..." - $SSH_CMD {{ remote_user }}@{{ remote_host }} "cd {{ remote_data_path }} && tar -czf - ." | \ - gpg --batch --yes --encrypt --recipient "{{ gpg_recipient }}" --output "$ENCRYPTED_BACKUP" - - # Also backup the .env file separately (smaller, might need quick access) - $SSH_CMD {{ remote_user }}@{{ remote_host }} "cat {{ remote_lnbits_dir }}/.env" | \ - gpg --batch --yes --encrypt --recipient "{{ gpg_recipient }}" --output "{{ local_backup_dir }}/lnbits-env-$TIMESTAMP.gpg" - - # Start LNBits service after backup - echo "Starting LNBits service..." - $SSH_CMD {{ remote_user }}@{{ remote_host }} "sudo systemctl start lnbits.service" - - # Rotate old backups (keep 14 days) - # Calculate cutoff date (14 days ago) and delete backups older than that - CUTOFF_DATE=$(date -d '14 days ago' +'%Y-%m-%d') - for backup_file in "{{ local_backup_dir }}"/lnbits-backup-*.tar.gz.gpg; do - if [ -f "$backup_file" ]; then - # Extract date from filename: lnbits-backup-YYYY-MM-DD.tar.gz.gpg - file_date=$(basename "$backup_file" | sed -n 's/lnbits-backup-\([0-9]\{4\}-[0-9]\{2\}-[0-9]\{2\}\)\.tar\.gz\.gpg/\1/p') - if [ -n "$file_date" ] && [ "$file_date" != "$TIMESTAMP" ] && [ "$file_date" \< "$CUTOFF_DATE" ]; then - rm -f "$backup_file" - fi - fi - done - for env_file in "{{ local_backup_dir }}"/lnbits-env-*.gpg; do - if [ -f "$env_file" ]; then - # Extract date from filename: lnbits-env-YYYY-MM-DD.gpg - file_date=$(basename "$env_file" | sed -n 's/lnbits-env-\([0-9]\{4\}-[0-9]\{2\}-[0-9]\{2\}\)\.gpg/\1/p') - if [ -n "$file_date" ] && [ "$file_date" != "$TIMESTAMP" ] && [ "$file_date" \< "$CUTOFF_DATE" ]; then - rm -f "$env_file" - fi - fi - done - - echo "Backup completed successfully" - - - name: Ensure cronjob for backup exists - cron: - name: "LNBits backup" - user: "{{ lookup('env', 'USER') }}" - job: "{{ backup_script_path }}" - minute: 5 - hour: "9,12,15,18" - - - name: Run the backup script to make the first backup - command: "{{ backup_script_path }}" diff --git a/ansible/services/memos/memos_vars.yml b/ansible/services/memos/memos_vars.yml index 99618db..e1c42a3 100644 --- a/ansible/services/memos/memos_vars.yml +++ b/ansible/services/memos/memos_vars.yml @@ -15,12 +15,4 @@ memos_tailscale_ip: "100.64.0.4" # (caddy_sites_dir and subdomain in services_config.yml) # Remote access (for backup from lapy via Tailscale) -backup_host: "{{ memos_tailscale_hostname }}" -backup_user: "counterweight" -backup_key_file: "~/.ssh/counterganzua" -backup_port: 22 - -# Local backup -local_backup_dir: "{{ lookup('env', 'HOME') }}/memos-backups" -backup_script_path: "{{ lookup('env', 'HOME') }}/.local/bin/memos_backup.sh" diff --git a/ansible/services/memos/setup_backup_memos_to_lapy.yml b/ansible/services/memos/setup_backup_memos_to_lapy.yml deleted file mode 100644 index c32c332..0000000 --- a/ansible/services/memos/setup_backup_memos_to_lapy.yml +++ /dev/null @@ -1,106 +0,0 @@ -- name: Configure local backup for Memos from memos-box - hosts: control - gather_facts: no - vars_files: - - ../../infra_vars.yml - - ./memos_vars.yml - vars: - backup_data_path: "{{ memos_data_dir }}" - - tasks: - - name: Debug remote backup vars - debug: - msg: - - "backup_host={{ backup_host }}" - - "backup_user={{ backup_user }}" - - "backup_data_path='{{ backup_data_path }}'" - - "local_backup_dir={{ local_backup_dir }}" - - - name: Ensure local backup directory exists - file: - path: "{{ local_backup_dir }}" - state: directory - mode: '0755' - - - name: Ensure ~/.local/bin exists - file: - path: "{{ lookup('env', 'HOME') }}/.local/bin" - state: directory - mode: '0755' - - - name: Create backup script - copy: - dest: "{{ backup_script_path }}" - mode: '0750' - content: | - #!/bin/bash - set -euo pipefail - - TIMESTAMP=$(date +'%Y-%m-%d') - BACKUP_DIR="{{ local_backup_dir }}/$TIMESTAMP" - mkdir -p "$BACKUP_DIR" - - {% if backup_key_file %} - SSH_CMD="ssh -i {{ backup_key_file }} -p {{ backup_port }}" - {% else %} - SSH_CMD="ssh -p {{ backup_port }}" - {% endif %} - - rsync -az -e "$SSH_CMD" --rsync-path="sudo rsync" --delete {{ backup_user }}@{{ backup_host }}:{{ backup_data_path }}/ "$BACKUP_DIR/" - - # Rotate old backups (keep 14 days) - # Calculate cutoff date (14 days ago) and delete backups older than that - CUTOFF_DATE=$(date -d '14 days ago' +'%Y-%m-%d') - for dir in "{{ local_backup_dir }}"/20*; do - if [ -d "$dir" ]; then - dir_date=$(basename "$dir") - if [ "$dir_date" != "$TIMESTAMP" ] && [ "$dir_date" \< "$CUTOFF_DATE" ]; then - rm -rf "$dir" - fi - fi - done - - - name: Ensure cronjob for backup exists - cron: - name: "Memos backup" - user: "{{ lookup('env', 'USER') }}" - job: "{{ backup_script_path }}" - minute: 15 - hour: "9,12,15,18" - - - name: Run the backup script to make the first backup - command: "{{ backup_script_path }}" - - - name: Verify backup was created - block: - - name: Get today's date - command: date +'%Y-%m-%d' - register: today_date - changed_when: false - - - name: Check backup directory exists and contains files - stat: - path: "{{ local_backup_dir }}/{{ today_date.stdout }}" - register: backup_dir_stat - - - name: Verify backup directory exists - assert: - that: - - backup_dir_stat.stat.exists - - backup_dir_stat.stat.isdir - fail_msg: "Backup directory {{ local_backup_dir }}/{{ today_date.stdout }} was not created" - success_msg: "Backup directory {{ local_backup_dir }}/{{ today_date.stdout }} exists" - - - name: Check if backup directory contains files - find: - paths: "{{ local_backup_dir }}/{{ today_date.stdout }}" - recurse: yes - register: backup_files - - - name: Verify backup directory is not empty - assert: - that: - - backup_files.files | length > 0 - fail_msg: "Backup directory {{ local_backup_dir }}/{{ today_date.stdout }} exists but is empty" - success_msg: "Backup directory contains {{ backup_files.files | length }} file(s)" - diff --git a/ansible/services/phoenixd/phoenixd_vars.yml b/ansible/services/phoenixd/phoenixd_vars.yml index f67c302..93921fd 100644 --- a/ansible/services/phoenixd/phoenixd_vars.yml +++ b/ansible/services/phoenixd/phoenixd_vars.yml @@ -47,6 +47,3 @@ remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counter remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" remote_port: "{{ hostvars.get(remote_host_name, {}).get('ansible_port', 22) }}" -# Local backup -local_backup_dir: "{{ lookup('env', 'HOME') }}/phoenixd-backups" -backup_script_path: "{{ lookup('env', 'HOME') }}/.local/bin/phoenixd_backup.sh" diff --git a/ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml b/ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml deleted file mode 100644 index 7474af1..0000000 --- a/ansible/services/phoenixd/setup_backup_phoenixd_to_lapy.yml +++ /dev/null @@ -1,136 +0,0 @@ ---- -# Backs up the phoenixd seed and config from vipy to Lapy, gpg encrypted. -# -# Only seed.dat and phoenix.conf are backed up, on purpose: -# - seed.dat is what actually recovers the funds. phoenixd keeps its channel -# state with the ACINQ peer, so a wallet is restored from the seed alone. -# - restoring a *stale* channel database to a running Lightning node is -# dangerous (it can trigger a force close and a penalty), so we do not keep -# copies of phoenix.db around to be tempted by. -# - phoenix.conf holds the http api password, which is what LNBits and any -# other local consumer authenticate with. -# -# Because both files are static, phoenixd does not need to be stopped. - -- name: Configure local backup for phoenixd from remote - hosts: control - gather_facts: no - vars_files: - - ../../infra_vars.yml - - ./phoenixd_vars.yml - vars: - remote_data_path: "{{ phoenixd_data_dir }}" - gpg_recipient: "{{ hostvars['localhost']['gpg_recipient'] | default('') }}" - gpg_key_id: "{{ hostvars['localhost']['gpg_key_id'] | default('') }}" - - tasks: - - name: Debug phoenixd backup vars - debug: - msg: - - "remote_host={{ remote_host }}" - - "remote_user={{ remote_user }}" - - "remote_data_path='{{ remote_data_path }}'" - - "local_backup_dir={{ local_backup_dir }}" - - "gpg_recipient={{ gpg_recipient }}" - - "gpg_key_id={{ gpg_key_id }}" - - - name: Ensure local backup directory exists - ansible.builtin.file: - path: "{{ local_backup_dir }}" - state: directory - mode: "0700" - - - name: Ensure ~/.local/bin exists - ansible.builtin.file: - path: "{{ lookup('env', 'HOME') }}/.local/bin" - state: directory - mode: "0755" - - - name: Create phoenixd backup script - ansible.builtin.copy: - dest: "{{ backup_script_path }}" - mode: "0750" - content: | - #!/bin/bash - set -euo pipefail - - if [ -z "{{ gpg_recipient }}" ]; then - echo "GPG recipient is not configured. Aborting." - exit 1 - fi - - TIMESTAMP=$(date +'%Y-%m-%d') - ENCRYPTED_BACKUP="{{ local_backup_dir }}/phoenixd-backup-$TIMESTAMP.tar.gz.gpg" - - {% if remote_key_file %} - SSH_CMD="ssh -i {{ remote_key_file }} -p {{ remote_port }}" - {% else %} - SSH_CMD="ssh -p {{ remote_port }}" - {% endif %} - - # seed.dat + phoenix.conf only, see the header of the playbook. - echo "Creating encrypted backup archive..." - $SSH_CMD {{ remote_user }}@{{ remote_host }} \ - "sudo tar -czf - -C {{ remote_data_path }} seed.dat phoenix.conf" | \ - gpg --batch --yes --encrypt --recipient "{{ gpg_recipient }}" --output "$ENCRYPTED_BACKUP" - - chmod 600 "$ENCRYPTED_BACKUP" - - # Rotate old backups (keep 14 days) - CUTOFF_DATE=$(date -d '14 days ago' +'%Y-%m-%d') - for backup_file in "{{ local_backup_dir }}"/phoenixd-backup-*.tar.gz.gpg; do - if [ -f "$backup_file" ]; then - # Extract date from filename: phoenixd-backup-YYYY-MM-DD.tar.gz.gpg - file_date=$(basename "$backup_file" | sed -n 's/phoenixd-backup-\([0-9]\{4\}-[0-9]\{2\}-[0-9]\{2\}\)\.tar\.gz\.gpg/\1/p') - if [ -n "$file_date" ] && [ "$file_date" != "$TIMESTAMP" ] && [ "$file_date" \< "$CUTOFF_DATE" ]; then - rm -f "$backup_file" - fi - fi - done - - echo "Backup completed successfully" - - - name: Ensure cronjob for phoenixd backup exists - ansible.builtin.cron: - name: "phoenixd backup" - user: "{{ lookup('env', 'USER') }}" - job: "{{ backup_script_path }}" - minute: 15 - hour: "9" - - - name: Run phoenixd backup script to create initial backup - ansible.builtin.command: "{{ backup_script_path }}" - - - name: Verify backup was created - block: - - name: Get today's date - command: date +'%Y-%m-%d' - register: today_date - changed_when: false - - - name: Check if backup file exists - stat: - path: "{{ local_backup_dir }}/phoenixd-backup-{{ today_date.stdout }}.tar.gz.gpg" - register: backup_file_stat - - - name: Verify backup file exists - assert: - that: - - backup_file_stat.stat.exists - - backup_file_stat.stat.isreg - fail_msg: "Backup file {{ local_backup_dir }}/phoenixd-backup-{{ today_date.stdout }}.tar.gz.gpg was not created" - success_msg: "Backup file {{ local_backup_dir }}/phoenixd-backup-{{ today_date.stdout }}.tar.gz.gpg exists" - - - name: Verify backup file is not empty - assert: - that: - - backup_file_stat.stat.size > 0 - fail_msg: "Backup file {{ local_backup_dir }}/phoenixd-backup-{{ today_date.stdout }}.tar.gz.gpg exists but is empty" - success_msg: "Backup file size is {{ backup_file_stat.stat.size }} bytes" - - - name: Remind about the offline seed copy - debug: - msg: | - These encrypted backups are only as safe as your GPG key. - Also write the 12 words down offline once: - ssh {{ remote_user }}@{{ remote_host }} "sudo cat {{ remote_data_path }}/seed.dat" diff --git a/ansible/services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml b/ansible/services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml deleted file mode 100644 index 096f9d3..0000000 --- a/ansible/services/vaultwarden/setup_backup_vaultwarden_to_lapy.yml +++ /dev/null @@ -1,105 +0,0 @@ -- name: Configure local backup for Vaultwarden from remote - hosts: control - gather_facts: no - vars_files: - - ../../infra_vars.yml - - ./vaultwarden_vars.yml - vars: - remote_data_path: "{{ vaultwarden_data_dir }}" - - tasks: - - name: Debug remote backup vars - debug: - msg: - - "remote_host={{ remote_host }}" - - "remote_user={{ remote_user }}" - - "remote_data_path='{{ remote_data_path }}'" - - "local_backup_dir={{ local_backup_dir }}" - - - name: Ensure local backup directory exists - file: - path: "{{ local_backup_dir }}" - state: directory - mode: '0755' - - - name: Ensure ~/.local/bin exists - file: - path: "{{ lookup('env', 'HOME') }}/.local/bin" - state: directory - mode: '0755' - - - name: Create backup script - copy: - dest: "{{ backup_script_path }}" - mode: '0750' - content: | - #!/bin/bash - set -euo pipefail - - TIMESTAMP=$(date +'%Y-%m-%d') - BACKUP_DIR="{{ local_backup_dir }}/$TIMESTAMP" - mkdir -p "$BACKUP_DIR" - - {% if remote_key_file %} - SSH_CMD="ssh -i {{ remote_key_file }} -p {{ remote_port }}" - {% else %} - SSH_CMD="ssh -p {{ remote_port }}" - {% endif %} - - rsync -az -e "$SSH_CMD" --delete {{ remote_user }}@{{ remote_host }}:{{ remote_data_path }}/ "$BACKUP_DIR/" - - # Rotate old backups (keep 14 days) - # Calculate cutoff date (14 days ago) and delete backups older than that - CUTOFF_DATE=$(date -d '14 days ago' +'%Y-%m-%d') - for dir in "{{ local_backup_dir }}"/20*; do - if [ -d "$dir" ]; then - dir_date=$(basename "$dir") - if [ "$dir_date" != "$TIMESTAMP" ] && [ "$dir_date" \< "$CUTOFF_DATE" ]; then - rm -rf "$dir" - fi - fi - done - - - name: Ensure cronjob for backup exists - cron: - name: "Vaultwarden backup" - user: "{{ lookup('env', 'USER') }}" - job: "{{ backup_script_path }}" - minute: 5 - hour: "9,12,15,18" - - - name: Run the backup script to make the first backup - command: "{{ backup_script_path }}" - - - name: Verify backup was created - block: - - name: Get today's date - command: date +'%Y-%m-%d' - register: today_date - changed_when: false - - - name: Check backup directory exists and contains files - stat: - path: "{{ local_backup_dir }}/{{ today_date.stdout }}" - register: backup_dir_stat - - - name: Verify backup directory exists - assert: - that: - - backup_dir_stat.stat.exists - - backup_dir_stat.stat.isdir - fail_msg: "Backup directory {{ local_backup_dir }}/{{ today_date.stdout }} was not created" - success_msg: "Backup directory {{ local_backup_dir }}/{{ today_date.stdout }} exists" - - - name: Check if backup directory contains files - find: - paths: "{{ local_backup_dir }}/{{ today_date.stdout }}" - recurse: yes - register: backup_files - - - name: Verify backup directory is not empty - assert: - that: - - backup_files.files | length > 0 - fail_msg: "Backup directory {{ local_backup_dir }}/{{ today_date.stdout }} exists but is empty" - success_msg: "Backup directory contains {{ backup_files.files | length }} file(s)" diff --git a/ansible/services/vaultwarden/vaultwarden_vars.yml b/ansible/services/vaultwarden/vaultwarden_vars.yml index e60f1d4..0605d27 100644 --- a/ansible/services/vaultwarden/vaultwarden_vars.yml +++ b/ansible/services/vaultwarden/vaultwarden_vars.yml @@ -12,6 +12,3 @@ remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counter remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" remote_port: "{{ hostvars.get(remote_host_name, {}).get('ansible_port', 22) }}" -# Local backup -local_backup_dir: "{{ lookup('env', 'HOME') }}/vaultwarden-backups" -backup_script_path: "{{ lookup('env', 'HOME') }}/.local/bin/vaultwarden_backup.sh" From 2ebb2f9a64c81ac1a2ae161ba74dd81d26c58acb Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 17:57:43 +0200 Subject: [PATCH 46/67] brushing up backups --- ansible/roles/backup_source/tasks/main.yml | 14 +- .../backup_source/templates/backup.sh.j2 | 6 + ansible/roles/backup_store/tasks/main.yml | 13 ++ .../templates/check-backups.sh.j2 | 148 ++++++++++++++++++ .../backup_store/templates/pull-backups.sh.j2 | 7 +- 5 files changed, 185 insertions(+), 3 deletions(-) create mode 100644 ansible/roles/backup_store/templates/check-backups.sh.j2 diff --git a/ansible/roles/backup_source/tasks/main.yml b/ansible/roles/backup_source/tasks/main.yml index 4283b84..ec2d6cf 100644 --- a/ansible/roles/backup_source/tasks/main.yml +++ b/ansible/roles/backup_source/tasks/main.yml @@ -16,12 +16,22 @@ # Declared here rather than assumed. Stage 1 installed it by hand; this is what # makes a rebuilt host get it too. +# Cache refresh is best-effort on purpose. An unrelated third-party repo with a +# bad signing key (spacey had two: an expired Caddy subkey and a SHA1 nodesource +# key) makes `apt-get update` return warnings, which the apt module treats as a +# hard failure — and that must not stop backups being configured. Installing the +# package is NOT best-effort: if age is genuinely unavailable, the next task fails. +- name: Refresh the apt cache (best effort) + ansible.builtin.apt: + update_cache: yes + cache_valid_time: 3600 + failed_when: false + changed_when: false + - name: Ensure age is installed ansible.builtin.apt: name: age state: present - update_cache: yes - cache_valid_time: 3600 # The pull account: unprivileged, no sudo, exists only so small-backups-box can # read the dump directory. Trust points one way — the box can read backups, and diff --git a/ansible/roles/backup_source/templates/backup.sh.j2 b/ansible/roles/backup_source/templates/backup.sh.j2 index 18fabd6..abdadfe 100644 --- a/ansible/roles/backup_source/templates/backup.sh.j2 +++ b/ansible/roles/backup_source/templates/backup.sh.j2 @@ -53,6 +53,12 @@ trap 'log "Restarting ${SERVICE}..."; eval "$START_CMD" || true' EXIT # --- Dump straight into age; plaintext never touches the disk --- log "Writing ${ARTIFACT}..." {{ backup_source_dump_command }} | age -r "$RECIPIENT" -o "${ARTIFACT}.partial" +{% if backup_source_pull_user %} +# Match the final ownership immediately, so even a partial left by a later +# failure is not an unreadable obstacle to the pull. +chown root:{{ backup_source_pull_user }} "${ARTIFACT}.partial" +chmod 640 "${ARTIFACT}.partial" +{% endif %} mv "${ARTIFACT}.partial" "$ARTIFACT" {% if backup_source_pull_user %} # Readable by the pull account and nobody else. The contents are age-encrypted diff --git a/ansible/roles/backup_store/tasks/main.yml b/ansible/roles/backup_store/tasks/main.yml index d138830..2596dbc 100644 --- a/ansible/roles/backup_store/tasks/main.yml +++ b/ansible/roles/backup_store/tasks/main.yml @@ -33,6 +33,19 @@ validate: "bash -n %s" become: yes +# A human-run assertion that last night actually worked. Generated from the same +# source list as the puller, so it can never drift out of sync with what is +# supposed to be arriving. +- name: Install the backup check script + ansible.builtin.template: + src: check-backups.sh.j2 + dest: /usr/local/bin/check-backups.sh + owner: root + group: root + mode: '0755' + validate: "bash -n %s" + become: yes + - name: Install the pull-backups systemd units ansible.builtin.template: src: "pull-backups.{{ item }}.j2" diff --git a/ansible/roles/backup_store/templates/check-backups.sh.j2 b/ansible/roles/backup_store/templates/check-backups.sh.j2 new file mode 100644 index 0000000..f7a8830 --- /dev/null +++ b/ansible/roles/backup_store/templates/check-backups.sh.j2 @@ -0,0 +1,148 @@ +#!/usr/bin/env bash +# Assert the nightly backups actually worked. +# +# Run as {{ ansible_user_id }} on this host. Needs no sudo. +# +# What it CANNOT do: verify contents. The age identity lives only on lapy, so +# this host cannot decrypt anything it holds — by design. These are freshness, +# completeness and integrity checks. To verify content, decrypt on lapy: +# ssh {{ ansible_user_id }}@$(hostname) "cat ~/backups//" \ +# | age -d -i ~/.age/counterweight_age | tar -tzf - | head +# +# Exit 0 = everything passed (warnings allowed), 1 = at least one FAIL. +set -uo pipefail + +STORE="{{ backup_store_dir }}" +MAX_AGE_H="${1:-26}" # an artefact older than this is stale +NOW=$(date +%s) +fails=0; warns=0 + +# Colour only when attached to a terminal: this gets piped into files and, later, +# probably into a notification. +if [ -t 1 ]; then R=$'\033[31m'; Y=$'\033[33m'; G=$'\033[32m'; N=$'\033[0m' +else R=''; Y=''; G=''; N=''; fi + +red() { printf ' %sFAIL%s %s\n' "$R" "$N" "$*"; fails=$((fails+1)); } +yell() { printf ' %sWARN%s %s\n' "$Y" "$N" "$*"; warns=$((warns+1)); } +ok() { printf ' %sok%s %s\n' "$G" "$N" "$*"; } + +hours_since() { echo $(( (NOW - $1) / 3600 )); } + +# Pull the dump timestamp out of _YYYYmmdd_HHMMSS.. This is when +# the SOURCE produced it, which is the thing that actually matters: a source +# whose timer died still pulls "ok" forever, because yesterday's artefact is +# still sitting there. Checking only the pull would miss exactly that. +dump_epoch() { + local base ts + base=$(basename "$1") + ts=$(echo "$base" | grep -oE '[0-9]{8}_[0-9]{6}' | head -1) || return 1 + [ -n "$ts" ] || return 1 + date -d "${ts:0:4}-${ts:4:2}-${ts:6:2} ${ts:9:2}:${ts:11:2}:${ts:13:2}" +%s 2>/dev/null +} + +check_source() { + local name="$1" keep="$2" dir="$STORE/$1" + printf '\n%s\n' "== $name" + + [ -d "$dir" ] || { red "$name: no directory $dir"; return; } + + local n; n=$(find "$dir" -maxdepth 1 -type f -name "${name}_*" | wc -l) + [ "$n" -gt 0 ] || { red "$name: no artefacts at all"; return; } + + local partials; partials=$(find "$dir" -maxdepth 1 -name '*.partial' | wc -l) + [ "$partials" -eq 0 ] || red "$name: $partials .partial file(s) pulled — the pull should exclude these" + + local newest; newest=$(ls -t "$dir"/${name}_* 2>/dev/null | head -1) + local prev; prev=$(ls -t "$dir"/${name}_* 2>/dev/null | sed -n 2p) + + # 1. Did the SOURCE dump recently? + local de; de=$(dump_epoch "$newest") + if [ -z "${de:-}" ]; then + yell "$name: cannot parse a dump timestamp from $(basename "$newest")" + else + local dh; dh=$(hours_since "$de") + if [ "$dh" -lt 0 ]; then + # A future-dated artefact would otherwise stay "fresh" forever and the + # staleness check would never fire again — the exact silent failure this + # script exists to catch. + red "$name: newest dump is dated ${dh#-}h in the FUTURE — clock skew on the source?" + elif [ "$dh" -gt "$MAX_AGE_H" ]; then + red "$name: newest dump is ${dh}h old (>${MAX_AGE_H}h) — the source timer did not run" + else + ok "$name: dumped ${dh}h ago" + fi + fi + + # 2. Did the PULL bring it over recently? + local ph; ph=$(hours_since "$(stat -c %Y "$newest")") + if [ "$ph" -gt "$MAX_AGE_H" ]; then + red "$name: newest artefact was pulled ${ph}h ago (>${MAX_AGE_H}h)" + else + ok "$name: pulled ${ph}h ago" + fi + + # 3. Is it plausibly a real backup? + local sz; sz=$(stat -c %s "$newest") + if [ "$sz" -eq 0 ]; then + red "$name: newest artefact is ZERO bytes" + elif [ -n "$prev" ]; then + local psz; psz=$(stat -c %s "$prev") + if [ "$psz" -gt 0 ] && [ "$sz" -lt $(( psz / 2 )) ]; then + # Not automatically wrong: headscale legitimately shrank 297K -> 20K when + # a clean stop checkpointed its write-ahead log into the database. + yell "$name: $(numfmt --to=iec "$sz") is less than half the previous $(numfmt --to=iec "$psz") — check it decrypts to what you expect" + else + ok "$name: $(numfmt --to=iec "$sz") ($n artefacts)" + fi + else + ok "$name: $(numfmt --to=iec "$sz") (first artefact)" + fi + + # 4. Is retention pruning? Allow generous slack for multiple dumps per day. + if [ "$n" -gt $(( keep * 3 + 10 )) ]; then + yell "$name: $n artefacts for a ${keep}-day retention — pruning may not be working" + fi +} + +echo "Backup check on $(hostname) at $(date '+%Y-%m-%d %H:%M:%S %Z')" +echo "Artefacts older than ${MAX_AGE_H}h are treated as stale." + +# --- the pull job itself --- +printf '\n%s\n' "== pull-backups.service" +result=$(systemctl show pull-backups.service -p Result --value 2>/dev/null) +status=$(systemctl show pull-backups.service -p ExecMainStatus --value 2>/dev/null) +when=$(systemctl show pull-backups.service -p ExecMainExitTimestamp --value 2>/dev/null) +[ "$result" = "success" ] && ok "last run result: success" || red "last run result: ${result:-unknown} (exit ${status:-?})" +if [ -n "$when" ]; then + wh=$(hours_since "$(date -d "$when" +%s)") + [ "$wh" -le "$MAX_AGE_H" ] && ok "last ran ${wh}h ago" || red "last ran ${wh}h ago (>${MAX_AGE_H}h) — did the timer fire?" +fi +systemctl is-enabled pull-backups.timer >/dev/null 2>&1 \ + && ok "timer enabled, next $(systemctl show pull-backups.timer -p NextElapseUSecRealtime --value 2>/dev/null)" \ + || red "pull-backups.timer is NOT enabled" + +# --- each source --- +{% for src in backup_store_sources %} +check_source "{{ src.name }}" {{ src.retention_days }} +{% endfor %} + +# --- capacity --- +printf '\n%s\n' "== disk" +use=$(df --output=pcent "$STORE" | tail -1 | tr -dc '0-9') +avail=$(df -h --output=avail "$STORE" | tail -1 | tr -d ' ') +if [ "$use" -ge 90 ]; then red "store is ${use}% full, ${avail} free" +elif [ "$use" -ge 75 ]; then yell "store is ${use}% full, ${avail} free" +else ok "store is ${use}% full, ${avail} free"; fi + +printf '\n%s\n' "-----" +if [ "$fails" -gt 0 ]; then + echo "RESULT: $fails failure(s), $warns warning(s)" + echo "Investigate with: journalctl -u pull-backups -n 50 --no-pager" + exit 1 +fi +if [ "$warns" -gt 0 ]; then + echo "RESULT: all checks passed, $warns warning(s)" +else + echo "RESULT: all checks passed" +fi +exit 0 diff --git a/ansible/roles/backup_store/templates/pull-backups.sh.j2 b/ansible/roles/backup_store/templates/pull-backups.sh.j2 index 267900c..1db7d9c 100644 --- a/ansible/roles/backup_store/templates/pull-backups.sh.j2 +++ b/ansible/roles/backup_store/templates/pull-backups.sh.j2 @@ -22,7 +22,12 @@ pull_{{ src.name | replace('-', '_') }}() { local dir="${STORE}/{{ src.name }}" mkdir -p "$dir" log "Pulling {{ src.name }} from {{ src.source }}..." - if rsync -az --timeout=120 \ + # --exclude '*.partial': a dump that died mid-write leaves one behind, owned + # root:root 0600 because the chown only happens after a successful mv. Without + # this exclude the pull account cannot read it and rsync fails for the WHOLE + # source — so one failed dump would silently block every subsequent pull of + # that service. An incomplete artefact is never worth transferring anyway. + if rsync -az --timeout=120 --exclude '*.partial' \ -e "ssh -i $SSH_KEY -o StrictHostKeyChecking=accept-new -o ConnectTimeout=15" \ "{{ src.source }}" "$dir/"; then log " {{ src.name }}: ok ($(find "$dir" -maxdepth 1 -type f | wc -l) artefacts, $(du -sh "$dir" | cut -f1))" From 73340d5fbe5bd5eee260409a16255dfb5e8a7825 Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 18:15:29 +0200 Subject: [PATCH 47/67] forgejo-runner: convert to a role, de-Uptime-Kuma the health check 409-line playbook becomes a 16-line playbook plus a 318-line role with phases split across tasks/{prerequisites,install,configure,service,healthcheck}.yml and four templates. forgejo_runner_vars.yml is deleted; its content is the role's defaults. Applies the Plan 6 Stage 0 decision: keep whatever determines whether the service is healthy, drop the Uptime Kuma specifics, make the reporting point pluggable. Gone from the role: the embedded Python that created monitors over the Kuma API, the /tmp credentials file, token extraction, the systemd Environment= rewrite, and 8 `when: uptime_kuma_enabled` guards. What remains is the check itself, its log, the systemd unit and timer, and an honest exit code - `systemctl is-failed forgejo-runner-healthcheck.service` now answers the question with no monitoring system involved at all. Reporting is one variable, healthcheck_push_url, empty by default. Any endpoint that accepts an HTTP ping plugs in there. A pull-based monitor wants it left empty and reads unit state instead. PREMISE CORRECTION: Uptime Kuma is NOT dead. Plan 3 recorded "48 push timers curling an endpoint that no longer answers" and Plan 6 said the check had "nowhere to report to". Both wrong - 24+ push scripts across 11 hosts are pushing successfully right now (HTTP 200). Only the Ansible code and the vault credentials were decommissioned; the service never stopped. So the existing push URLs were harvested into a vaulted healthcheck_push_urls dict and are preserved, keeping this refactor behaviour-neutral. Retiring Kuma stays a deliberate act rather than a side effect. PLAN_3 and PLAN_6 are corrected. Verified: - task-list diff vs the old playbook shows ONLY the five Kuma tasks removed, everything else identical and in the same order - first run ok=22 changed=1 (the rewritten health script); both systemd units and forgejo-runner.service came back ok, so the templates reproduce the previous files byte-for-byte - second run ok=22 changed=0, fully idempotent - still reports "Ping sent successfully (HTTP 200)" from a script containing zero Uptime Kuma references - the 4 skipped tasks are genuine already-configured guards, checked not assumed Two things for the next service: - import_tasks, not include_tasks. Dynamic includes are opaque to --list-tasks, which is the primary verification tool here; the first attempt produced a useless diff. - `Assert runner is running` was guarded by uptime_kuma_enabled and so had not run since the decommissioning. It is not monitoring, it is the deployment checking its own work - the deprecation banner swept it up with the Kuma plumbing, and a runner that failed to start was deploying "successfully" in silence. Ungated now. The banner was applied to contiguous blocks, so read every uptime_kuma_enabled guard and ask whether it is monitoring or deployment. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 220 +++++++--- ansible/infra_secrets.yml | 220 +++++++--- ansible/roles/forgejo_runner/README.md | 59 +++ .../roles/forgejo_runner/defaults/main.yml | 40 ++ .../roles/forgejo_runner/tasks/configure.yml | 43 ++ .../forgejo_runner/tasks/healthcheck.yml | 73 ++++ .../roles/forgejo_runner/tasks/install.yml | 27 ++ ansible/roles/forgejo_runner/tasks/main.yml | 9 + .../forgejo_runner/tasks/prerequisites.yml | 12 + .../roles/forgejo_runner/tasks/service.yml | 33 ++ .../templates/forgejo-runner.service.j2 | 17 + .../templates/healthcheck.service.j2 | 13 + .../templates/healthcheck.sh.j2 | 43 ++ .../templates/healthcheck.timer.j2 | 11 + .../deploy_forgejo_runner_playbook.yml | 409 +----------------- .../forgejo-runner/forgejo_runner_vars.yml | 9 - 16 files changed, 710 insertions(+), 528 deletions(-) create mode 100644 ansible/roles/forgejo_runner/README.md create mode 100644 ansible/roles/forgejo_runner/defaults/main.yml create mode 100644 ansible/roles/forgejo_runner/tasks/configure.yml create mode 100644 ansible/roles/forgejo_runner/tasks/healthcheck.yml create mode 100644 ansible/roles/forgejo_runner/tasks/install.yml create mode 100644 ansible/roles/forgejo_runner/tasks/main.yml create mode 100644 ansible/roles/forgejo_runner/tasks/prerequisites.yml create mode 100644 ansible/roles/forgejo_runner/tasks/service.yml create mode 100644 ansible/roles/forgejo_runner/templates/forgejo-runner.service.j2 create mode 100644 ansible/roles/forgejo_runner/templates/healthcheck.service.j2 create mode 100644 ansible/roles/forgejo_runner/templates/healthcheck.sh.j2 create mode 100644 ansible/roles/forgejo_runner/templates/healthcheck.timer.j2 delete mode 100644 ansible/services/forgejo-runner/forgejo_runner_vars.yml diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index d1f18ec..6c01d31 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,60 +1,162 @@ $ANSIBLE_VAULT;1.1;AES256 -61356165613635386631393135656434646436303665313031346566323336313138353433316463 -3363323534613064643132663335623238366431393062340a346538396662306537663163623366 -38626166383933616331623231373137306562623637313263333237633661663436666266616433 -3862346438643638650a306634333535653633613534646630386131333236366538333765323333 -38656163303837303732663561373232393132343331663164656262393730326434373731333636 -36613538646431396536363936336562616431656665653965373864633366663836353434626434 -34383932373461333564303439623565383661646365386665393831383463663662356536356236 -36373964666236626465366161636135393734356536633466383262326537343833636630343738 -62353066316131613737373162643363653662656261363465386364323962656537373061373032 -33343763353464383438363438343965653532393831343930393562633630383932653862623637 -38386239353237356631646436356166373961333464396639383538383662326534343339313330 -65373339356364636634616532363832386631323062363530313861336238353261353334306235 -62616338396431316537346638656365356564346666366366343638356261623664393263323937 -30343261363562383332323462336435376664386134646562643836363834313237373631353731 -35653535663864643266313332356635363262363533663232656531373130336539633066376139 -32316665393831663035623962656364363831333563366135636164346335383738363336663566 -63356532643563393939383635386462663561386434323939303431653438653131363538383034 -33623933333464363032656636643033326162626163353633343062633966343332383138363963 -36383831663562616533316436366566323061386535343538393861383462333166343562316633 -30623665623035393537393965626363323132656433313339396233356666346634316332616336 -37363635363330323230373332326565343530653335383437373230366563366237633665626331 -66623336626230663361636439316337393865383035326136653264666438666566646132353036 -38616264313833316536623238633339373466613866626366383835656638623863323838653030 -63303938376164653966356435386333363731656666313234663535666165646233313137343563 -36386437393139656438333262383437656666343831313239323961373637653163643664356565 -62346431393133656530316262303763646165643336396661666431383436323562336137653031 -62626538373839613734396366653065306534636630346338316237616161613037616364356431 -38643965376133616161336633383664326230383435363334353137303162663738313331346238 -39356161616533616134356231323530306338333162343363353531303263636632613036386638 -63633136386232306234323936303563646466313935326631396565383432386130656638616266 -32383334363237336539396665336366643764633131643663643137376438323666326435626461 -64346633636431393137633537306431646564386565303933636434386462346630626537346438 -32623333666133303061646564366366326665363163396262633164323631636337346130303239 -38373936663337356134666132303165393365663763396362623434633737373538653566646134 -62316164396438303532616266313062326666633130656338653139376634306664633031333037 -31636166306565353334633435656233336233363664306264626237623366336161303134353433 -63383739666462623336386537346662633666626466393039653439346436653937633537396436 -35393339383066326630353066623132333034656539363561346462626265363263303535343961 -39316461616630326539613731303039613736393633373338646266323938326162373831346336 -30656130343463366534323030646238313465306266383034623065636665623366333063383736 -38373063393837306462303564643962373334343139626338623935336435643730646532633630 -66363730386636633639346463363365343239373265303738353732653633653437636130636664 -39386365353334653765343335303263363461313965383664326563333734626533376436626530 -63353163323637303730353564383733653365613635353764333266393532653663326533646132 -39313763373735323835626437306435373238653432393936643165663663656665316132653330 -35383038353532656434366336346235363563636264303734633138323963396562306232646236 -38366536306561653937336333373434336164663336613839353439356435333833396363636437 -38323934613735643363656233333037336465336564313966623063376566663030303230323262 -65643564666534326234306164343365383632333061316238623565353538313538396364313337 -35336230313339643736653238386231343661623337306236383665356632366236356335323530 -66373831616239376231636361333430343433303233393066323865663434643433303832373262 -36366534646161323130393931626362626139663139643263366639656531613436313533363130 -65643736363833333939613566663339623964333262643863333030623138633464386238613934 -31323639333234336264313663376465323737353766643839303665313737336534386665363034 -30326439666433373232306136306365343764643434306561343339353132346430646436343362 -32333765616262363930353435616563333736313533653339656231316230346166363335356638 -37333561366339613136306438306130343230663732333862663838396463623661303961336433 -65646332373139303462303633346432366530643130366133363937653739653036366136373434 -343864373734666431326239373866313734 +32333164303734353564643266316239636365313631613866306666643538353537323939373066 +3439396330326266396531303737363131313831396365650a343832363564356337396530383438 +36356465623634623361643436383262393339663134373630666363613464653437666164393731 +3462326463346535310a396531393761643062643563613964613234666531643139666535323734 +63653630616138373539343434636434326466353134643864396466373435613738653536356532 +36313432373138316532663464353638386530666231376636643931663338303663346665363039 +39323734343934393037633766626132333835363265653538373266323236336137356630353834 +63306164313434363130306339346435643463376137366534643033366537383861363531353861 +34363966303234333133336466646635623136396138663637613133613861313866646634643132 +34383461366133333430633631353561653037346339626165346564353630373363643065323535 +61386237353135636665363538623538313039343965363535323734653763366238373334323065 +64626135383736316135303731656231313263306266636563343465653666326663333536383435 +32643537616161623830656161633763303566656462356235633866303165383663386435363133 +61353430656337656364646231383332316534316233393333646436353062316461366439613030 +62346366383566356661376463616137373062303363346333636134633465396238343761363139 +38616336633532366165376237626337333933353935613066303865303536303464633834643533 +31376236333133336166373635626130316632396561393930353065616465663336393938323539 +35343531306261343733626539346265386436643135326461353734346164633233383237393731 +32303030623564666539383835333064323630393539393062313435383663303334383134666436 +63336564353464313534323866383533323365393761643565633430346263346666623638303030 +38653461616661346661316537306666313165613163303835363636616561343335623131636633 +36636263356530366663616461316635316365393666313066326265306335363038366561356339 +37366231656330643764613664653735633963666638623961653134316136666336396438333864 +65653738623638373636366666663463373035363862396565396166643332343934626261353938 +34666638383531366532663963323533323965373439623735353661373862393830623934313535 +62363237663764356338383133336463303234393262623061363062373938613262383836323431 +35336133306139386336373965393132393431343535306162383337643961323039373530646330 +35383962303263353234376164633530666264386338376335653465616161383532613830636562 +36613837326564313565633632353964346537303337316233623033383961363737313861393234 +64313130306264636134626638396661353362346439373463663965653165613436363633323238 +39356430613033643334363731346333346639643563633162333636653066386233373063653138 +38623630353265663332366630316362633135633362313735306533333962373433376237366266 +64643266323862383363633033656465303032336134623036646530323264323532653234306363 +64326663343064643865373164306661613463366561383737363535303861646634353139636666 +66626466303037363064303865313830356531313834353165303839326238613962313261353536 +63626133663765373763623930623963653038313661656131666261356236366565626663323831 +32313130366135616662616639386436656265346635613762353832626466323261346333373661 +33656338326631393064313762343161363832643030303737663639356261633131353937396135 +34666361353266373661626534353134663235343662636435383032636261636637636361613631 +30373965373638386664313432386264303761376266363161363633343831633032356639343836 +36306163303363353534313466353863633834393337303161313431343165346162653537346130 +36393439386232613865613837346563643031393530646433353936383463303562326331313738 +30646537616264333332323562363237323530313333386531323066343335323133366566383935 +34613232323965626339653132323162626234356135323436353263306137343130346337326531 +62396164386661303366363566393833353130643636613865616433633166666532653937646262 +63323530616533373962626264633236313064306336633063356536343862316237383166346264 +65666466366435653134303164613632666336626630373764333534393164303132656530343031 +32386137633135376531373662336131313030303436333833663234343938323232343832633234 +35623361393533316232393431646561616535643638383533353266646235313736356366343231 +30383962353833326161663534396565316139393439366631313731663737656332646361326331 +35623038343931666231653731346435653162326237376339663936343933346264343564666132 +62393230363262323366313963373737646138336163313434376462383964396337313030383831 +35643865303738393736323032353239633631306262663536663732623933383431383563646664 +65326230323239353831623232633935636632636438636165376633313364643637373233323231 +30323534333131623138636266396565373964633963626536383930316663656634636638316363 +36326438363463386138633133346666383163323936653131666339333430316363393966316130 +66333732333565646564303633373464666461373437656634306564336432323465386631323034 +32376635666139313465393539353438626534646332323537363163653233353030366231306234 +31663065393062633438343438303762633931653564366666336662323430366466333334303763 +64356138373035323862326137643963323865666539316439333064343961303838636265623735 +36636233373065303239363461373239646662346162326164613332373761326561656234636438 +39636231386531353236303933343661393333383730636366626263393534656363366337316432 +66393162313731616165343362643162376334346262353730653738323138646463666433383963 +31626131373736633662366366376566656438343330383330616235376239353661663663313233 +63373836363638366432376130633930343862613436633263613538336536616163656364656561 +35353163323663663833343834373036366531623536303138613035353064303065623761376462 +32656363613765316635666639383865636538326130386535316132623238393730353631346136 +65353864393863376632356437303565626262343039636436636335383337623465336263623439 +61393030636466303836663166613766326164626639313965373734643466373663343565343631 +35356431303139396131346335346664373830663361616136306535643431353037636431343132 +32613164333232346335663236343238653033643133316564666534323130373861623962393461 +62386336326461326239383534303631346639343765393839656364353762356266653733303238 +37393533366164393863373134633439383765616435346239333065623438353431663035306132 +30363663383765356365333234323061326163653463646566373764363037373932623032366133 +64626633643737323631306539333531343239326639643166393435323731353932336262633963 +62363561616635356230626431626232383439633738306532636361336334616238643665323266 +35666165396566323639303733623364353436613966336337656363633762393664303939656537 +33656431663431396237643536613233376561313261323635613634613439633533393361333435 +33616136633936386135623563616530663833643339623439366430646639646162613863643565 +32343732613134656332653163623437366637656537666334653239366638653537393639396234 +63613362623037663465626438646136396362663033376261376166306138363436666566393838 +63363262386261316432313235323166333334346539343663303265303535636439653130366566 +38383632326433323036356266646138366233656131663633613236303137336437383333653430 +35653738623635656661393232333334643937303631663464383239353635393735353833333265 +38303831623130303634666464316465653639383230623662326534333136616561666166613930 +37643632393265326364376230643634356538626337633638663634326538396536383633303633 +62613532336435326161626263376462363162303762613835663831623539623562336361333031 +37346438343230363433633738643064336235323438346534643463333930626430653131623538 +61363735336332633663383136613633376430303133366634643839636562656431663737336431 +65653164393331626532356566623830623931373431393330633638306663376566346136643863 +61653461633637643663303761326464636339336635383938633039613132336139663762363734 +34313831366232383562363834653232613665616266323033336635653933626630346132643164 +66336663363161616431316666383430326266636134396239656561643763386438306434373166 +65616637356366353336346539396665396364383162383533323837303134313664326338366131 +64336333316531353661616561656430386462666565373535613134336634383136613638323565 +34303130316162376137303137396330343466383263363638306334386161613535303735623537 +30613065383963666166336135663632303038326163346630376466393761663738653532333466 +35346161373237383766393063623939363664356639653862353132353930383261643761363135 +66323364376432363437656239626362313835333537663030386435623764373937636130663331 +31663636616261636663306230653264353566313966333665383938393731336538363665636665 +33383334613565393139613138633636656234386263663932323364653361623433326438326132 +34636633663838636634323236636564333630666362356333333230346333373833653365363135 +37326630386439356139653865643039666237313833306161643263663363393965633163366436 +65373132623265363134343233653366383665633163363766353066653863373937613631373332 +34653835636565353731396331633133393435373739306663343632323234633462356630666238 +36316634303031346138393835363565366131663534323933623166653662323931386135363432 +36626434323536326636653165626661386663393134653061326634633162653666393661353762 +31343766383933633261366437396332356362663539623032663435356231656430366532623535 +35636361343330333263636230383030326139323065373032376238333162343462383566333264 +66386439303163616638353635303736376461326533623234633264353865393065363765366636 +63383034373631656331386439633863613731633930646330613931303430366236643735633434 +33346638356330643534396236633630616134626237613363306334383639336161306235626133 +32616339633033343431633131666131383036373335303739663634653831376563363034363365 +63366532393761323862373434656264323938313334333234323162353663306263636435653636 +65343531663234636362623139653439386132316430346536623864383539313037666439643638 +35323235303164396361383832303362613235386530303538376265373964623363653365383961 +61306365346532336164303763396364363038646563323562653865303461353339366334376534 +61346230343662336633316133666631613136316537393464666635643336393566613730333837 +62643232623831333730313737666536353137613138333566323934353236616464656238303339 +63346665326661653832646466636165366338376638353431346130323230646534303137386634 +66646362663831623234333335376561336664343232323361326430313138633266343065366535 +39396463636330663738333033616334623366643636623735396435643865396630656234333433 +63346437663331653233313430666633366264343565373838656536323231633966663262346434 +35386631333063666233303364346139336461333736366438396239353136373339636638323534 +61663564353733363131363139663761646561313337656232336164653234646563666266656362 +36613666303165383530386530646331346661323633393232376533386661376533663166346337 +63316431393964333731373332313732643765386662656437373136663532663531383436393761 +36323939643464343035313633313436373038646234343366356264386131623462343161313261 +62383839343863386365366432396666303065616137396631373330643339656433363463313131 +30653431393630653331633634363233323632306434626566393662323566613565633639313938 +65626434353530306630636430646261373365386538613232383663336339303838626533306164 +62386166363861613863393331343362303665613239626265316366386165633264636163343366 +30643965633239333162633433366132323162396564383530373839333331643632333930643035 +33313663623165613033353130616666643136303137356462613237653430366663653061653035 +64653961343163646530623863313663303231633764626264363366613162313535653631656365 +39633561643833313334363465626263656132623463633335333465306266373634386238666165 +37373134646438303134653734323665666531626533666663363534316238666232316438393939 +32633531313565636531646361356566663133316164623133653935313335323835616238336431 +61663664393239343837353039303334353035323861323533663161376361353933313032306136 +36623463363235303166616162373665616333346366336333333531623932376664383861616232 +38393263336364343566653735396561363836373064346131336366316231303539386531643835 +34363463636232653137656339643630336333653163383332663763663633393837343963336666 +61316236366436363064396464383138373639373864636261383533333139306237303030373037 +36333563653036393734393439636262303837356133636233663064623136356133613932656365 +62356630323134373063663664316133663761383035633065383838333364663732353866356230 +33333263316666323133323433653330646161386462303163303235343031343761633234306166 +63363138376531353130663431636637646262366436393161376633306332346465316333656432 +34333238306263666537626231376665376435623466643131363736656432353938653236656132 +37313633323863656463633030336566343265666534663737396530316234343433383931336462 +35306236626564373766393438653134353332616532373132336663626130666532373933323863 +39656463386136646131356462653933306532383232313334326334613738356339623566636661 +62373139326138326663666237663436656631343635373733626439643466326439663361313266 +61323966336532653263633530646363366439326663356664313065306166313863396439396564 +30303138326261303661333831306536333363663536323235356263623533346233386161396131 +64356265363964316231333933613463623163316538393830393539306533313035303765656437 +38623262333662613665396537323331393937366533303039386138643639383664663263643734 +65313531343636393232343230613639363466616235633162353430646364363934393530643263 +39633539333237653733363132303163613564623232313464376336306431666534653338323765 +35646564653736383962343664383666333936363864396334333463313637343336396434633766 +35396466306235323338393937643362373764613161333462303665646239636562306635663330 +3630 diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml index 4803d86..c597ce2 100644 --- a/ansible/infra_secrets.yml +++ b/ansible/infra_secrets.yml @@ -1,60 +1,162 @@ $ANSIBLE_VAULT;1.1;AES256 -34616531626537376463326332306434383163393761363536633133363161373631376437653234 -3831656536313531343433376264643261396634633965370a303931336236323766313065636535 -31343338616364386132623232373266663665353563666535383232666262663062323864303932 -3439373734316639650a633732333064633335383833663261326464336538636666323063646331 -66653464636534663935616461636132613162303530663237376661343338396133323431623561 -30303563303465633231306339326561313533333536376433393130376634303932643339656336 -30616130343733626162366561393939393766623138616537393032376665396139333561623830 -33623130336235633034666561383165616439316334323664623661343438663733373833306335 -30396165313731636435636263353133326135346236323030653734353831626238666663656364 -62386164363361373464383730333733626333333736306535336635613634646535383237326332 -37636532663933663739303634333235343137333363316430643263326231613635643633636437 -30336462633233396430633939616661643261666136656363363461343230653738613166346434 -63376363666462346539343063343562623962303130616535306439653134633630353963373337 -64313239353162363862643432666438383330636638333464323731643163643236313535313030 -65656231383738356537326661663163376634613031396436646633376139313237343534653065 -64323532346133313265346334353530393633396339316330366536646565353836396662303866 -65373030393463306664623235626437613965303837313365373632643935656630303936373837 -64663539623764656636613263346638376665626262663430333231633735653563643864343835 -62303637383332653366383066326536316336306539623230353066343739316430356437313365 -33323031613130643832636133636263303565623330306135333762363036633737343933623437 -33633436363062643436376632336235666330656265316333323361316430656631383734343238 -65616366666536363866343361666563643336306532666332643230656264303565353032393932 -36613738326239376161383064646634623639356439663965323237303361343838373737636235 -39656465333063633136643463356564636537326339653033633063366265636138656466646264 -32643961343863656464353439346531633036613138353333353631626665396239326465303632 -39336664333038636435366333623336353963323535316436646136313739373537376537353735 -66333265633232346536313933333836643439656666636236626266323333383935383565616139 -38626133356364613562393664643563346539366364636332613162663337356232333735393431 -30343635393462356431646336636663333735343164616261343836363366616162613563613066 -66333034613234626431356536323832393466363135653231336238356334353638613838626430 -63636630313533363563376161663637353362383431663130646666646433353864373736613662 -62343666633862643839666532656132653939663437373164393536653739376265303130353636 -62616538326330333136626238306133363634326365366637363961333635356133613564373131 -64626139656633663065353862343135393233333231386166346532656534623361623061336434 -37616565626134333863376532393861376132323434656433613731613834633963386635353336 -39653263346637303836663763653264396139383363616461393333333362616332663863363834 -61613832326466646133626337663364333566326633313664613236346135653332373035313034 -62356530313065393737393634313166313039616635363365633031336434396463326462316236 -62353838636533303664663132353838636632353733316639643964666139383539333761663531 -37366530373165653063303032383461326535613336343635626538373635356266396666346133 -61383434323565306438356634646338666232363238393932353365363461376130323363363236 -34643862653937353461383265663933646264373365623465313666643662633334366437646434 -34343332613832366235663563353435323738356339383338333361383561336366376238393237 -38656265663164376666616366623062393530366261623361383365376235643035353838313235 -32396430376139313430316439333539343266393965353030636638656661316137313437663832 -64366661323865383262323036613230316233666566386133363633303931323464663238323861 -62333132346165353830653062353763306232366563323430383933653163373562326536393362 -66353931376262383931613931303034356537363137366235643430323465623162363031373034 -64653563666533363062333831336463376530306639616631303266386136666233653638653732 -31313537393838333730303235363661326333646531653935633935373362303732653565656438 -36626330623063373832333739643666396135653838643164376264393863646433353036643731 -34623636623366353466333237373265323936643134613230326565303663306431373634356334 -37613732393262626135306632623336613364323632316636623037656538313236666562633438 -37306361333130376538366232393637646263656239303431623963626265386634313735373162 -64393733303132383464386330626135336237303563326238633437346164333738646431333730 -62616533363638643261306565373835303832313539656564386132393863373039653938616261 -36343031356361336563316230343433633033353130373933313361326164633765316433616162 -62356338393634653330623933313433343630386264396337373532376335316362333164363963 -363663666365336564663232363462613162 +34313539653264336537326165616161313931303664633032373566626264643439393136386230 +6330353962336634623963336535333130626662336561650a303864333966333336376661356139 +35303531343330356331356166666666343739636564353039313637663066356135623630396236 +6434663137386232300a323334313332373630366133393530313438623733666333623437666539 +65376166643436303661343763383166616662653137636539623435653632623433396465616363 +33336431363266653138366564343862626634373838316439313339313961323437656435623635 +37653534376262656130373734353335663764386633646436396335616437633636376462343861 +38343430363733316166353032343164313333663865356335393634656335373636346565653233 +63386535366533303736623463623830363166646530666631643730303265396163353538656662 +34363964356665666464336636373535613639343535633361383064623065643230303066303661 +35333666353331363031353732643233613439666539373539343437336362656132373634663636 +30383039303266663961666662613730646664386266356338366335633663666162343338363263 +63323531613735363333613637666530323830346135653839353733306561396239373734313530 +61616233313264323264396230393163303766666665656330636636306661303838333265383535 +63383330363331663038643735653266333233613666393161636634396365663832653736306437 +39636230336336613863623562376134616234323434653061656534383132353038333966663631 +36666562623234303130373661636136336538386163353230383830303438336435346432623761 +36636263306564323039356165623835363730643463343065366436323235616639376363363037 +38623165623635363239366361616337663932653734383162376363396533373563643439633632 +36306564336165363035386663323366343636383366393366373732393634356236626535356333 +38383633316362396133346631303234363366666664343137633765353231306534653837323035 +65333762383164663433356438333264323631373561353762646265386630616536633833306565 +32336635623036666131613634333332393037643063373561303938313762666262353564303039 +31333731396661373865313862633732313864306131303936333638323965613831323034303130 +39393864663136383430626137323736313132303030326131616463653635383262353731653034 +65643437323434626637313333666538613039633635333231386433363963653432376164383364 +34376437386166363335373964666631393230383038333137636439643936393866356537393535 +35306135313532363462653734633162363236356436386161656164666261393236393162353865 +38613964613434636334663361396362396163623533623166306261366130633962653131386365 +65626464663736326365376166633538303439326561643362633163626330643265393630303163 +62383835623639663838663934663030383363646339663632666364613830373263313437303964 +37343664356135663361323539663562323265373163303035633463326162363462636131363436 +31343764313537666661656631643263353239616563636361633561313966663932616163313932 +66363837656538373337653737653061356237383436643964653230333263313063333033663866 +61613634363463363232663765646438653266386166623431363932383931396339633665633430 +62306437633438393136643930653939623734643162353961313831613337646435656436353436 +30396265643732636337303233623931643361663032383335333431623063393731633764636166 +38663465366537636539376661643237623936333565383666333163363066366532633965346463 +38653265373262663039353433663134373361663535336633383566303462393663623432653836 +64323838343432343061323338353633613930613463383465633465313436343932366163313361 +32356431353434353434306462306134643332326363353462666561663035356530376263653831 +30616533323666363161653231663835663532336236663034316364333438336130383537346633 +65656565643363303036396439373963623338626461366234623233626231323534616130623537 +66323666346439333763646635643635663538633361356637646565353234316264306339373765 +37356236323763383761323637306163613933623661353262643362636330393362646236386563 +38323236666633306336393839396439303536643933393331333363626533373636356339623235 +66643436343736383136656630356139653537356632343838636364306461326539373932383430 +66343736333261316363316230623336313366303963366134666533626363353937373434383965 +65363737626534363030386461333261363736373230323437303636353030373335366266643566 +34313863363864306331306130626135623265643935376130303566636432646532356438613635 +36353766373434323638626330313030333737346262653833643735306432623836323834663638 +34646130623363613763346163363435643561336332663865623563343666616338616134336361 +31313731646434666263303136613338613637613530303932646561616138393032343838623766 +35633163366463346663656331386539376263646264343466366466663664643661383238326138 +38646231616133373038343636383733643661366131646335623564623663326537373039353535 +36643062623063663131363765666265303231333332333535316634336262306437633236323961 +66653263613836373235363166326337393762346534636330633931356265613765363336386162 +61393630636136313630613363383138333939313231396332353132313139323362386162613364 +63633261313965323966396635623334333633353963663263666433616632613463353231386238 +61333439613839353431613637633465616338363334333962666464396563356266323538373761 +64343436636138663934306130336663633665356465303064363637316332656432666639633330 +39383664616536383265303265663437303438393833663066343735616261653736346230663663 +34663563653862316335326661663264333138613430643934303334343838623934356431313264 +61393166616239613131636131643762313563623137633966326362326664346438366564306130 +30663161306462383836646436316138373431636339633438343439633633663735336433356232 +36373437323031303637636231646262343133646632633561393339303835326336656533613562 +36616530623530386238623262393537626632376637663232336138353232396333383837303362 +36623538333432396439306261356663383163653665393530313434616137626230356463333434 +30383861363164343866326264366135346166633936313834636162303764613735383565663264 +30616264643535353730396435633863646634616332303337363230366666373563656534383661 +34613835346565393730353165663264656661623930623361383465366336383933626162393530 +39353535613237313064613537383564323164323337313039393637396534383563373637373730 +66653539303963623034333862613539613663323238333765646138656464386535336334626437 +35626465623864376539333964316533326235626265346438333737633738393532323162306430 +66353633386337626462643466343333353130356663363766353161333662323830643261376139 +32616465326135373037636265306436653630646164323365636166356362353634343036623035 +35303136373235313431613932666230353034396334336561333732353762656434633037666664 +31306438386635613038623662336137336430653434383730653161383239376466326232323163 +66653738363030626231383930303335333663623837366464633764333461636461303131393966 +32303035376661623738323032346638613261313963626130386464383032303334383735396564 +63326633376139316436393236643066656631396134393962366363346233666232393434626134 +31653933383631613064303563656665383834623662306137303861623063663437636335326430 +62306630393938323832613830353230313136383834333832343366646632366634393632383539 +61373531653862623261633766613531346263343030373733383465353233316162656566386264 +35313834333739393064303730306262663032376335663032613233343131336662623065346231 +37653164633632396635363434623936346135616531353536323961336662646334666335306137 +38343037303337363966663934663733303364323134613633313566346161376661383032393832 +35363531316431303937323761323234666436313661623833323235373539393134346261666533 +32396665383961316166633836656562623665656134353762303238373633376333356664666539 +33356430613663366162316663623735646434656261613835363835343935323765626335326139 +34636439313466363438343862313636303163376635383361633338363664623830613837616464 +34306636346438313437613430306466383166373431353435656166643638626236373533393939 +62366538353262383239396236653664613664626236326531386536643433313361393065316562 +37303435333231313434653939313264613031343266616537383464356334333137646132396536 +30623238346662323163656365323933373366323530306530613032363861643037393332626531 +30353838393263306137633564376430323038313162626633623238353136653032653936303863 +33373135656434316466663363616138623734313063666632393265343532333433303731373239 +30393635333465313735363030386634323635393335383532326138346538633264386163623463 +34396336366638306633633732363162633336353962303261346439396231366337653633653161 +34383732333530366136366334396630303533336165653333326465306439313233373335326237 +30663331643839663534666338616466323633636364363663353462613632353764353931663164 +39306462623434653035396534663232326533643032343139663237366235653265393164373735 +30626435626235343730366265633935663463663435633137646138393765323738633162623239 +65666666353339656636633765313164306339383135623633643763396530643061383764373364 +38616233613562353831343838343830323130646662363061316534346564336631343639623161 +38666338346330643537366631313165366632643932323131613836303362383733633165653762 +64383539303139386337323931623531373461343737303236396534346566363338306138393334 +62363961386234326539663433646530663233306365613662633737376237393838306361343365 +31623538353362663830646436376236333161656238303332303339353531613461313839643239 +65613230373035336434363632633735643063646363656133326630336461313035633662613763 +65333665323764306535656137343034656165366530386166353964636338653532613332633437 +64323665333335386661376437303539336534393865353262623366323762326435313262626234 +37323433333138363837613034616662383362636462616134393431303737306166646233346362 +33633536636537633230353039623562636635396138366131366337613131323334356139346433 +62393564656635663836383164343562303637303533376665383638343238653939663962303535 +61383232316264323434366530623635623637623062333235646436646531643261633264393461 +36333766356434633532363536353734383866656461613338633234363966313661333537613439 +37343236396535666661656634326439633539303339616563643465613732343933393131623038 +63353231633633356563373438396335636439326136363463366663646565313532303230313836 +33613734323538316534386531616233646431623435326338366339386431376363656439326535 +39373065386463353435396466323433343436306639303266323431653037336463316532376533 +39656365303166306430646664346539376366343332633837303030376561623966393732393466 +65353563346234633965623036636264386239336537656237313433633932323063636239616561 +34323261373938333462623862393063333963306333616530656232646461363132323932663231 +65656361326139323030326332613166303461326236383432323761663061353134633736353665 +39643932363538376163613063316533346530646139626536356663633464636235353035653362 +61313838363065623365666538643562356336373539343237366161316130373461306362353133 +37343965666436643633626461613738303738313838326265373164373464643037363861643730 +30326431393865373362363836333631393039313938343834366237653961353163326665306262 +34643836653434313838303035623961613036656630666266366330313031386338326563663833 +64323535663438313865306539343936393864376263393532336538316530663839643161346233 +65383831316566373734636462396161323630376434666430653235373238303932616662666437 +31373532353339366534316463623237313835366164663630343561623861383263626365663566 +66383034613939363736356262336336356363313332363466653534653336353964313831363139 +34626436623231306434393464376639303462356132636630373762366135613935363033623137 +39643036366263353139623539666130333163616562353239353731623534346635653239396636 +39363636346234356465383365356462333430366265306464666666393033656638393930626661 +36616365393637316231656239336135333664346139373163303231363266393562626364346561 +36333738373738643136653465353331316561646162383230666133636135646330636363653061 +35623234303233326531356537616131356533386334623938303462613337323930623566343664 +34323831616636393264643765653262316434623635373733623139396131636366353033653263 +66393361653766623161343935626234336163363765313339653266343837313731653839643066 +36633861663130346635343863646166393238333165316365333230376433333439613731333461 +37353035323931306462646465323066333463333236353665303664393461353030343963346132 +32396237626335313831616662383165333439393739353865623631333630363432633366646263 +32633435366461656535323230613739323634386536666132633935313532663763353963343065 +61616134333630633333616639333062663636326237366265326264613863353537323232373165 +32663964313736313031376662663631663635393731333264613566336661623030616235333163 +31343865333735633933373933303030393632656435333032373730663866343736626663623136 +62643566356665356234306134653037383638303064316138313638653336616364393934623862 +38613864313962633335653165363437383631366531613731633339313562366364306164386430 +34326133643436616537336132306634613461353963306664313531633564653134383330336537 +64393331346566313966626238663733646565343761353563623461663662336331656634323964 +34323965326638363038316233623431656233346165373136343532316635326561356265336232 +63396563383765333736356638363261396634303730383136663566383135303331373534313766 +66363335356265396164326438393862613333663936666230316133396563376535663365633061 +35323665373233656631633038343631356233633934666161633766306331373537646231306437 +32623965656139373835333238343565643635306437656262623334316633646361623262386435 +32663031323338313339386663656363316164353666346237316137313562623838326436383862 +35616663633433613064613264363637396632333734326231343830633537336364346336316461 +3833 diff --git a/ansible/roles/forgejo_runner/README.md b/ansible/roles/forgejo_runner/README.md new file mode 100644 index 0000000..e7f761b --- /dev/null +++ b/ansible/roles/forgejo_runner/README.md @@ -0,0 +1,59 @@ +# `forgejo_runner` + +Installs and runs a Forgejo Actions runner, registers it with the Forgejo +instance, and keeps a health check on a systemd timer. + +Converted from `deploy_forgejo_runner_playbook.yml` (409 lines) under Plan 6. +The playbook is now 16 lines. + +## Phases + +`tasks/main.yml` imports five files in order: + +| | | +|---|---| +| `prerequisites.yml` | Docker must be present | +| `install.yml` | binary, system user, working directory | +| `configure.yml` | config file, registration with the instance | +| `service.yml` | systemd unit, start, assert it came up | +| `healthcheck.yml` | check script, unit, timer | + +`import_tasks`, not `include_tasks` — static imports are visible to +`--list-tasks`, which is how the conversion was verified against the playbook it +replaced. + +## Monitoring: one variable, no product knowledge + +This role contains **nothing specific to any monitoring system**. What used to +be here — an ~80-line embedded Python script creating monitors over the Uptime +Kuma API, a `/tmp` credentials file, token extraction, a systemd `Environment=` +rewrite, and 8 `when: uptime_kuma_enabled` guards — is gone. + +What remains answers the actual question, *is this service healthy*, and records +it two ways: + +- **the exit code**, which systemd keeps: `systemctl is-failed + forgejo-runner-healthcheck.service` is a complete answer with no monitoring + system involved at all; +- **a log file** at `{{ healthcheck_log_file }}`. + +To report health somewhere, set one variable: + +```yaml +healthcheck_push_url: "https://example/api/push/TOKEN" +``` + +Any endpoint accepting an HTTP ping works. Empty (the default) means check, log, +exit honestly, report nowhere — which is also the right setting for a *pull*-based +monitor like Prometheus' textfile collector, since that reads unit state instead. + +The push URL is a credential (anyone holding it can forge an "up"), so callers +pass it from the vault rather than committing it. + +## One behaviour change, deliberate + +`Assert runner is running` used to be guarded by `uptime_kuma_enabled`, so it +never ran. It is not a monitoring task — it is the deployment checking its own +work — and the deprecation banner swept it up by mistake. It is ungated here, +which means a runner that fails to start now fails the play instead of +deploying "successfully" in silence. diff --git a/ansible/roles/forgejo_runner/defaults/main.yml b/ansible/roles/forgejo_runner/defaults/main.yml new file mode 100644 index 0000000..e0bac24 --- /dev/null +++ b/ansible/roles/forgejo_runner/defaults/main.yml @@ -0,0 +1,40 @@ +--- +# Binary +forgejo_runner_version: "6.3.1" +forgejo_runner_arch: "linux-amd64" +forgejo_runner_url: "https://code.forgejo.org/forgejo/runner/releases/download/v{{ forgejo_runner_version }}/forgejo-runner-{{ forgejo_runner_version }}-{{ forgejo_runner_arch }}" +forgejo_runner_bin_path: "/usr/local/bin/forgejo-runner" + +# Runtime +forgejo_runner_user: "runner" +forgejo_runner_dir: "/opt/forgejo-runner" +forgejo_runner_config_path: "{{ forgejo_runner_dir }}/config.yml" +forgejo_runner_labels: "docker:docker://node:20-bookworm,ubuntu-latest:docker://node:20-bookworm,ubuntu-22.04:docker://node:20-bookworm,ubuntu-24.04:docker://node:20-bookworm" + +# The Forgejo instance this runner registers with. +forgejo_instance_url: "https://forgejo.contrapeso.xyz" +# forgejo_runner_registration_token comes from the vault. + +# --- Health check ----------------------------------------------------------- +# The check answers "is this service healthy" and records the answer two ways: +# a log file, and its own exit code. The exit code is the durable artefact — +# systemd stores it, so `systemctl is-failed forgejo-runner-healthcheck.service` +# answers the question with no monitoring system involved at all. +healthcheck_interval_seconds: 60 +healthcheck_timeout_seconds: 90 +healthcheck_retries: 1 +healthcheck_script_dir: /opt/forgejo-runner-healthcheck +healthcheck_script_path: "{{ healthcheck_script_dir }}/forgejo_runner_healthcheck.sh" +healthcheck_log_file: "{{ healthcheck_script_dir }}/forgejo_runner_healthcheck.log" +healthcheck_service_name: forgejo-runner-healthcheck + +# WHERE TO REPORT HEALTH — the one place to plug in monitoring. +# +# Empty means "check, log, exit honestly, report nowhere". Set it to any URL +# that accepts an HTTP ping and the check will report there. Nothing in this +# role is specific to a particular monitoring product: the Uptime Kuma API +# calls, monitor creation and token handling that used to live here are gone. +# +# A pull-based monitor (Prometheus node_exporter textfile, say) needs this left +# empty — it reads the systemd unit state instead. +healthcheck_push_url: "" diff --git a/ansible/roles/forgejo_runner/tasks/configure.yml b/ansible/roles/forgejo_runner/tasks/configure.yml new file mode 100644 index 0000000..ab96d52 --- /dev/null +++ b/ansible/roles/forgejo_runner/tasks/configure.yml @@ -0,0 +1,43 @@ +--- +- name: Check if config already exists + stat: + path: "{{ forgejo_runner_config_path }}" + register: config_stat + +- name: Generate default config + shell: "{{ forgejo_runner_bin_path }} generate-config > {{ forgejo_runner_config_path }}" + args: + chdir: "{{ forgejo_runner_dir }}" + when: not config_stat.stat.exists + +- name: Set config file ownership + file: + path: "{{ forgejo_runner_config_path }}" + owner: "{{ forgejo_runner_user }}" + group: "{{ forgejo_runner_user }}" + when: not config_stat.stat.exists + +# ── 6. Register runner ───────────────────────────────────────────── +- name: Check if runner is already registered + stat: + path: "{{ forgejo_runner_dir }}/.runner" + register: runner_stat + +- name: Register runner with Forgejo instance + command: > + {{ forgejo_runner_bin_path }} register --no-interactive + --instance {{ forgejo_instance_url }} + --token {{ forgejo_runner_registration_token }} + --name forgejo-runner-box + --labels "{{ forgejo_runner_labels }}" + args: + chdir: "{{ forgejo_runner_dir }}" + when: not runner_stat.stat.exists + +- name: Set runner registration file ownership + file: + path: "{{ forgejo_runner_dir }}/.runner" + owner: "{{ forgejo_runner_user }}" + group: "{{ forgejo_runner_user }}" + when: not runner_stat.stat.exists + diff --git a/ansible/roles/forgejo_runner/tasks/healthcheck.yml b/ansible/roles/forgejo_runner/tasks/healthcheck.yml new file mode 100644 index 0000000..0abdf1f --- /dev/null +++ b/ansible/roles/forgejo_runner/tasks/healthcheck.yml @@ -0,0 +1,73 @@ +--- +# Everything here answers "is the service healthy" and records the answer. +# The Uptime Kuma specifics that used to surround it — an embedded Python script +# that created monitors over the API, a /tmp credentials file, token extraction, +# and a systemd Environment= rewrite — are gone. What reports where is now one +# variable, healthcheck_push_url. See the role README. +- name: Create healthcheck script directory + ansible.builtin.file: + path: "{{ healthcheck_script_dir }}" + state: directory + owner: root + group: root + mode: '0755' + +- name: Create forgejo-runner healthcheck script + ansible.builtin.template: + src: healthcheck.sh.j2 + dest: "{{ healthcheck_script_path }}" + owner: root + group: root + mode: '0755' + validate: "bash -n %s" + +- name: Create healthcheck systemd service + ansible.builtin.template: + src: healthcheck.service.j2 + dest: "/etc/systemd/system/{{ healthcheck_service_name }}.service" + owner: root + group: root + mode: '0644' + +- name: Create healthcheck systemd timer + ansible.builtin.template: + src: healthcheck.timer.j2 + dest: "/etc/systemd/system/{{ healthcheck_service_name }}.timer" + owner: root + group: root + mode: '0644' + +- name: Reload systemd for healthcheck units + systemd: + daemon_reload: yes + +- name: Enable and start healthcheck timer + systemd: + name: "{{ healthcheck_service_name }}.timer" + enabled: yes + state: started + +- name: Test healthcheck script + command: "{{ healthcheck_script_path }}" + register: healthcheck_test + changed_when: false + +- name: Verify healthcheck script works + assert: + that: + - healthcheck_test.rc == 0 + fail_msg: "Healthcheck script failed to execute properly" + +- name: Display deployment summary + debug: + msg: | + Forgejo Runner deployed successfully! + + Runner Name: forgejo-runner-box + Instance: {{ forgejo_instance_url }} + Working Directory: {{ forgejo_runner_dir }} + Service: forgejo-runner.service ({{ runner_active.stdout }}) + + Healthcheck Monitor: {{ healthcheck_service_name }} + Healthcheck Interval: Every {{ healthcheck_interval_seconds }}s + Reporting to: {{ healthcheck_push_url | default('', true) | regex_replace('/api/push/.*', '/api/push/***') | default('(nowhere - set healthcheck_push_url)', true) }} diff --git a/ansible/roles/forgejo_runner/tasks/install.yml b/ansible/roles/forgejo_runner/tasks/install.yml new file mode 100644 index 0000000..b95ab4c --- /dev/null +++ b/ansible/roles/forgejo_runner/tasks/install.yml @@ -0,0 +1,27 @@ +--- +- name: Download forgejo-runner binary + get_url: + url: "{{ forgejo_runner_url }}" + dest: "{{ forgejo_runner_bin_path }}" + mode: '0755' + +# ── 3. Create runner system user ─────────────────────────────────── +- name: Create runner system user + user: + name: "{{ forgejo_runner_user }}" + system: yes + shell: /usr/sbin/nologin + home: "{{ forgejo_runner_dir }}" + create_home: no + groups: docker + append: yes + comment: 'Forgejo Runner' + +# ── 4. Create working directory ──────────────────────────────────── +- name: Create forgejo-runner working directory + file: + path: "{{ forgejo_runner_dir }}" + state: directory + owner: "{{ forgejo_runner_user }}" + group: "{{ forgejo_runner_user }}" + mode: '0750' diff --git a/ansible/roles/forgejo_runner/tasks/main.yml b/ansible/roles/forgejo_runner/tasks/main.yml new file mode 100644 index 0000000..aa71001 --- /dev/null +++ b/ansible/roles/forgejo_runner/tasks/main.yml @@ -0,0 +1,9 @@ +--- +# import_tasks, not include_tasks: these are unconditional phases, and static +# imports are visible to `--list-tasks`. That matters because the task list is +# how this refactor was verified against the playbook it replaced. +- ansible.builtin.import_tasks: prerequisites.yml +- ansible.builtin.import_tasks: install.yml +- ansible.builtin.import_tasks: configure.yml +- ansible.builtin.import_tasks: service.yml +- ansible.builtin.import_tasks: healthcheck.yml diff --git a/ansible/roles/forgejo_runner/tasks/prerequisites.yml b/ansible/roles/forgejo_runner/tasks/prerequisites.yml new file mode 100644 index 0000000..d441709 --- /dev/null +++ b/ansible/roles/forgejo_runner/tasks/prerequisites.yml @@ -0,0 +1,12 @@ +--- +- name: Check if Docker is installed + command: docker --version + register: docker_check + changed_when: false + failed_when: docker_check.rc != 0 + +- name: Fail if Docker is not available + assert: + that: + - docker_check.rc == 0 + fail_msg: "Docker is required for forgejo-runner but is not installed" diff --git a/ansible/roles/forgejo_runner/tasks/service.yml b/ansible/roles/forgejo_runner/tasks/service.yml new file mode 100644 index 0000000..f971e52 --- /dev/null +++ b/ansible/roles/forgejo_runner/tasks/service.yml @@ -0,0 +1,33 @@ +--- +- name: Create forgejo-runner systemd service + ansible.builtin.template: + src: forgejo-runner.service.j2 + dest: /etc/systemd/system/forgejo-runner.service + owner: root + group: root + mode: '0644' + +- name: Reload systemd + systemd: + daemon_reload: yes + +- name: Enable and start forgejo-runner service + systemd: + name: forgejo-runner + enabled: yes + state: started + +- name: Verify forgejo-runner is active + command: systemctl is-active forgejo-runner + register: runner_active + changed_when: false + +# Ungated on purpose. This was previously guarded by `uptime_kuma_enabled`, but +# it is not a monitoring task — it is the deployment asserting its own success. +# The deprecation banner swept it up along with the Kuma plumbing, which meant a +# broken runner deployed "successfully" and silently. +- name: Assert runner is running + assert: + that: + - runner_active.stdout == "active" + fail_msg: "forgejo-runner service is not active: {{ runner_active.stdout }}" diff --git a/ansible/roles/forgejo_runner/templates/forgejo-runner.service.j2 b/ansible/roles/forgejo_runner/templates/forgejo-runner.service.j2 new file mode 100644 index 0000000..d3db25d --- /dev/null +++ b/ansible/roles/forgejo_runner/templates/forgejo-runner.service.j2 @@ -0,0 +1,17 @@ +[Unit] +Description=Forgejo Runner +Documentation=https://forgejo.org/docs/latest/admin/actions/ +After=docker.service +Requires=docker.service + +[Service] +Type=simple +User={{ forgejo_runner_user }} +Group={{ forgejo_runner_user }} +WorkingDirectory={{ forgejo_runner_dir }} +ExecStart={{ forgejo_runner_bin_path }} daemon --config {{ forgejo_runner_config_path }} +Restart=on-failure +RestartSec=10 + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/forgejo_runner/templates/healthcheck.service.j2 b/ansible/roles/forgejo_runner/templates/healthcheck.service.j2 new file mode 100644 index 0000000..aae9eb5 --- /dev/null +++ b/ansible/roles/forgejo_runner/templates/healthcheck.service.j2 @@ -0,0 +1,13 @@ +[Unit] +Description=Forgejo Runner Healthcheck +After=network.target + +[Service] +Type=oneshot +ExecStart={{ healthcheck_script_path }} +User=root +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/forgejo_runner/templates/healthcheck.sh.j2 b/ansible/roles/forgejo_runner/templates/healthcheck.sh.j2 new file mode 100644 index 0000000..b9d43c3 --- /dev/null +++ b/ansible/roles/forgejo_runner/templates/healthcheck.sh.j2 @@ -0,0 +1,43 @@ +#!/bin/bash +# Forgejo Runner healthcheck — managed by Ansible (roles/forgejo_runner) +# +# Answers "is forgejo-runner healthy" and records it two ways: this log, and the +# exit code. The exit code is the durable artefact — systemd keeps it, so +# systemctl is-failed {{ healthcheck_service_name }}.service +# answers the question with no monitoring system involved. +# +# Reporting is optional and generic: if a push URL is configured it also pings +# it. Nothing here knows or cares which monitoring product is on the other end. + +LOG_FILE="{{ healthcheck_log_file }}" +PUSH_URL="{{ healthcheck_push_url }}" + +log_message() { + echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" +} + +main() { + if ! systemctl is-active --quiet forgejo-runner; then + log_message "ERROR: forgejo-runner is not active" + exit 1 + fi + + if [ -z "$PUSH_URL" ]; then + # Healthy, and nothing to report to. Not an error: the exit code below + # is still a complete answer for anything reading unit state. + log_message "forgejo-runner is active (no push URL configured)" + exit 0 + fi + + log_message "forgejo-runner is active, sending ping" + response=$(curl -s -w "\n%{http_code}" "$PUSH_URL?status=up&msg=forgejo-runner%20is%20active" 2>&1) + http_code=$(echo "$response" | tail -n1) + if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then + log_message "Ping sent successfully (HTTP $http_code)" + else + log_message "ERROR: Failed to send ping (HTTP $http_code)" + exit 1 + fi +} + +main diff --git a/ansible/roles/forgejo_runner/templates/healthcheck.timer.j2 b/ansible/roles/forgejo_runner/templates/healthcheck.timer.j2 new file mode 100644 index 0000000..4cba51a --- /dev/null +++ b/ansible/roles/forgejo_runner/templates/healthcheck.timer.j2 @@ -0,0 +1,11 @@ +[Unit] +Description=Run Forgejo Runner Healthcheck every minute +Requires={{ healthcheck_service_name }}.service + +[Timer] +OnBootSec=30sec +OnUnitActiveSec={{ healthcheck_interval_seconds }}sec +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml index bdc8428..446cdff 100644 --- a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml +++ b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml @@ -1,3 +1,4 @@ +--- - name: Install Forgejo Runner on Debian 13 hosts: ci_runner become: yes @@ -5,405 +6,11 @@ - ../../infra_vars.yml - ../../services_config.yml - ../../infra_secrets.yml - - ./forgejo_runner_vars.yml vars: - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - ntfy_topic: "{{ service_settings.ntfy.topic }}" - healthcheck_interval_seconds: 60 - healthcheck_timeout_seconds: 90 - healthcheck_retries: 1 - healthcheck_script_dir: /opt/forgejo-runner-healthcheck - healthcheck_script_path: "{{ healthcheck_script_dir }}/forgejo_runner_healthcheck.sh" - healthcheck_log_file: "{{ healthcheck_script_dir }}/forgejo_runner_healthcheck.log" - healthcheck_service_name: forgejo-runner-healthcheck - - tasks: - # ── 1. Assert Docker is available ────────────────────────────────── - - name: Check if Docker is installed - command: docker --version - register: docker_check - changed_when: false - failed_when: docker_check.rc != 0 - - - name: Fail if Docker is not available - assert: - that: - - docker_check.rc == 0 - fail_msg: > - Docker is not installed or not in PATH. - Please install Docker before running this playbook. - - # ── 2. Download forgejo-runner binary ────────────────────────────── - - name: Download forgejo-runner binary - get_url: - url: "{{ forgejo_runner_url }}" - dest: "{{ forgejo_runner_bin_path }}" - mode: '0755' - - # ── 3. Create runner system user ─────────────────────────────────── - - name: Create runner system user - user: - name: "{{ forgejo_runner_user }}" - system: yes - shell: /usr/sbin/nologin - home: "{{ forgejo_runner_dir }}" - create_home: no - groups: docker - append: yes - comment: 'Forgejo Runner' - - # ── 4. Create working directory ──────────────────────────────────── - - name: Create forgejo-runner working directory - file: - path: "{{ forgejo_runner_dir }}" - state: directory - owner: "{{ forgejo_runner_user }}" - group: "{{ forgejo_runner_user }}" - mode: '0750' - - # ── 5. Generate default config ───────────────────────────────────── - - name: Check if config already exists - stat: - path: "{{ forgejo_runner_config_path }}" - register: config_stat - - - name: Generate default config - shell: "{{ forgejo_runner_bin_path }} generate-config > {{ forgejo_runner_config_path }}" - args: - chdir: "{{ forgejo_runner_dir }}" - when: not config_stat.stat.exists - - - name: Set config file ownership - file: - path: "{{ forgejo_runner_config_path }}" - owner: "{{ forgejo_runner_user }}" - group: "{{ forgejo_runner_user }}" - when: not config_stat.stat.exists - - # ── 6. Register runner ───────────────────────────────────────────── - - name: Check if runner is already registered - stat: - path: "{{ forgejo_runner_dir }}/.runner" - register: runner_stat - - - name: Register runner with Forgejo instance - command: > - {{ forgejo_runner_bin_path }} register --no-interactive - --instance {{ forgejo_instance_url }} - --token {{ forgejo_runner_registration_token }} - --name forgejo-runner-box - --labels "{{ forgejo_runner_labels }}" - args: - chdir: "{{ forgejo_runner_dir }}" - when: not runner_stat.stat.exists - - - name: Set runner registration file ownership - file: - path: "{{ forgejo_runner_dir }}/.runner" - owner: "{{ forgejo_runner_user }}" - group: "{{ forgejo_runner_user }}" - when: not runner_stat.stat.exists - - # ── 7. Create systemd service ────────────────────────────────────── - - name: Create forgejo-runner systemd service - copy: - dest: /etc/systemd/system/forgejo-runner.service - content: | - [Unit] - Description=Forgejo Runner - Documentation=https://forgejo.org/docs/latest/admin/actions/ - After=docker.service - Requires=docker.service - - [Service] - Type=simple - User={{ forgejo_runner_user }} - Group={{ forgejo_runner_user }} - WorkingDirectory={{ forgejo_runner_dir }} - ExecStart={{ forgejo_runner_bin_path }} daemon --config {{ forgejo_runner_config_path }} - Restart=on-failure - RestartSec=10 - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - # ── 8. Reload systemd, enable and start ──────────────────────────── - - name: Reload systemd - systemd: - daemon_reload: yes - - - name: Enable and start forgejo-runner service - systemd: - name: forgejo-runner - enabled: yes - state: started - - # ── 9. Verify runner is active ───────────────────────────────────── - - name: Verify forgejo-runner is active - command: systemctl is-active forgejo-runner - register: runner_active - changed_when: false - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Assert runner is running - when: uptime_kuma_enabled | default(false) - assert: - that: - - runner_active.stdout == "active" - fail_msg: "forgejo-runner service is not active: {{ runner_active.stdout }}" - - # ── 10. Set up Uptime Kuma push monitor ──────────────────────────── - - name: Create Uptime Kuma push monitor setup script - when: uptime_kuma_enabled | default(false) - copy: - dest: /tmp/setup_forgejo_runner_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - retries = int(sys.argv[8]) - ntfy_topic = sys.argv[9] if len(sys.argv) > 9 else "alerts" - - api = UptimeKumaApi(api_url, timeout=60, wait_events=2.0) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Get all notifications and find ntfy notification - notifications = api.get_notifications() - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - # Find or create group - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name=group_name) - # Refresh to get the full group object with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - # Find or create/update push monitor - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': False, - 'maxretries': retries, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - monitor = api.edit_monitor(existing_monitor['id'], **monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - else: - monitor_result = api.add_monitor(**monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma push monitor setup - when: uptime_kuma_enabled | default(false) - command: > - {{ ansible_playbook_python }} - /tmp/setup_forgejo_runner_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "services" - "forgejo-runner-healthcheck" - "Forgejo Runner healthcheck - ping every {{ healthcheck_interval_seconds }}s" - "{{ healthcheck_timeout_seconds }}" - "{{ healthcheck_retries }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - when: uptime_kuma_enabled | default(false) - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL - when: uptime_kuma_enabled | default(false) - set_fact: - uptime_kuma_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - - - name: Create healthcheck script directory - file: - path: "{{ healthcheck_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create forgejo-runner healthcheck script - when: uptime_kuma_enabled | default(false) - copy: - dest: "{{ healthcheck_script_path }}" - content: | - #!/bin/bash - - # Forgejo Runner Healthcheck Script - # Checks if forgejo-runner is active and pings Uptime Kuma on success - - LOG_FILE="{{ healthcheck_log_file }}" - UPTIME_KUMA_URL="{{ uptime_kuma_push_url }}" - - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - main() { - if systemctl is-active --quiet forgejo-runner; then - log_message "forgejo-runner is active, sending ping" - response=$(curl -s -w "\n%{http_code}" "$UPTIME_KUMA_URL?status=up&msg=forgejo-runner%20is%20active" 2>&1) - http_code=$(echo "$response" | tail -n1) - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Ping sent successfully (HTTP $http_code)" - else - log_message "ERROR: Failed to send ping (HTTP $http_code)" - exit 1 - fi - else - log_message "ERROR: forgejo-runner is not active" - exit 1 - fi - } - - main - owner: root - group: root - mode: '0755' - - - name: Create healthcheck systemd service - copy: - dest: "/etc/systemd/system/{{ healthcheck_service_name }}.service" - content: | - [Unit] - Description=Forgejo Runner Healthcheck - After=network.target - - [Service] - Type=oneshot - ExecStart={{ healthcheck_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Create healthcheck systemd timer - copy: - dest: "/etc/systemd/system/{{ healthcheck_service_name }}.timer" - content: | - [Unit] - Description=Run Forgejo Runner Healthcheck every minute - Requires={{ healthcheck_service_name }}.service - - [Timer] - OnBootSec=30sec - OnUnitActiveSec={{ healthcheck_interval_seconds }}sec - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd for healthcheck units - systemd: - daemon_reload: yes - - - name: Enable and start healthcheck timer - systemd: - name: "{{ healthcheck_service_name }}.timer" - enabled: yes - state: started - - - name: Test healthcheck script - command: "{{ healthcheck_script_path }}" - register: healthcheck_test - changed_when: false - - - name: Verify healthcheck script works - assert: - that: - - healthcheck_test.rc == 0 - fail_msg: "Healthcheck script failed to execute properly" - - - name: Display deployment summary - debug: - msg: | - Forgejo Runner deployed successfully! - - Runner Name: forgejo-runner-box - Instance: {{ forgejo_instance_url }} - Working Directory: {{ forgejo_runner_dir }} - Service: forgejo-runner.service ({{ runner_active.stdout }}) - - Healthcheck Monitor: forgejo-runner-healthcheck - Healthcheck Interval: Every {{ healthcheck_interval_seconds }}s - Timeout: {{ healthcheck_timeout_seconds }}s - - - name: Clean up temporary monitor setup script - when: uptime_kuma_enabled | default(false) - file: - path: /tmp/setup_forgejo_runner_monitor.py - state: absent - delegate_to: localhost - become: no + # Preserves the push URL this host has been reporting to all along, so the + # move to a role changes no behaviour. The role itself knows nothing about + # Uptime Kuma — this is just "a URL that accepts a ping", and whatever + # replaces it sets the same variable. + healthcheck_push_url: "{{ healthcheck_push_urls.forgejo_runner | default('') }}" + roles: + - forgejo_runner diff --git a/ansible/services/forgejo-runner/forgejo_runner_vars.yml b/ansible/services/forgejo-runner/forgejo_runner_vars.yml deleted file mode 100644 index e618fca..0000000 --- a/ansible/services/forgejo-runner/forgejo_runner_vars.yml +++ /dev/null @@ -1,9 +0,0 @@ -forgejo_runner_version: "6.3.1" -forgejo_runner_arch: "linux-amd64" -forgejo_runner_url: "https://code.forgejo.org/forgejo/runner/releases/download/v{{ forgejo_runner_version }}/forgejo-runner-{{ forgejo_runner_version }}-{{ forgejo_runner_arch }}" -forgejo_runner_bin_path: "/usr/local/bin/forgejo-runner" -forgejo_runner_user: "runner" -forgejo_runner_dir: "/opt/forgejo-runner" -forgejo_runner_config_path: "{{ forgejo_runner_dir }}/config.yml" -forgejo_runner_labels: "docker:docker://node:20-bookworm,ubuntu-latest:docker://node:20-bookworm,ubuntu-22.04:docker://node:20-bookworm,ubuntu-24.04:docker://node:20-bookworm" -forgejo_instance_url: "https://forgejo.contrapeso.xyz" From 6c1bcbed95dbfb5c3461ac873bfb7ca83cde44cb Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 18:21:24 +0200 Subject: [PATCH 48/67] phoenixd: convert to a role, de-Uptime-Kuma the health check 552-line playbook becomes 18 lines plus a 411-line role (install/service/healthcheck phases, four templates, two handlers). phoenixd_vars.yml is deleted; its content is the role's defaults. Task-list diff vs the old playbook shows ONLY the eight Uptime Kuma tasks removed - everything else identical and in the same order. phoenixd holds a Lightning node, so the run was checked against a pre-flight: before: channel 6c25fa83..., balanceSat 1550723, capacitySat 3114830, blockHeight 966692, active since 2026-09-02 after: identical, and still active since 2026-09-02 - it did NOT restart `Create phoenixd systemd service` came back unchanged, which is what proves the template reproduces the live unit byte-for-byte. changed=3 was the health check script, its unit (Environment rename), and the timer restart. Second run: changed=0. Two things the conversion fixed, both symptoms of the deprecation banner having been applied to contiguous blocks rather than to individual tasks: - The health check logged "ERROR: UPTIME_KUMA_PUSH_URL not set" on every fire - about 1,400 times a day - because its Environment= was emptied at decommissioning. The exit code was still correct so nothing was broken, but it is exactly the kind of noise that trains you to ignore a log. An unset push URL is now normal and silent. - `Enable and start phoenixd health check timer` was guarded by uptime_kuma_enabled and so had not run since the decommissioning, while the timer itself was still live on the host from before. Ansible had quietly stopped managing something that was still running. Ungated. Noted, not changed: seed.dat is mode 0644 on the host. That is phoenixd's own doing, but it is a Lightning seed and worth tightening. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/roles/phoenixd/README.md | 56 ++ .../phoenixd/defaults/main.yml} | 23 +- ansible/roles/phoenixd/handlers/main.yml | 12 + ansible/roles/phoenixd/tasks/healthcheck.yml | 68 +++ ansible/roles/phoenixd/tasks/install.yml | 125 ++++ ansible/roles/phoenixd/tasks/main.yml | 6 + ansible/roles/phoenixd/tasks/service.yml | 52 ++ .../phoenixd/templates/healthcheck.service.j2 | 14 + .../phoenixd/templates/healthcheck.sh.j2 | 39 ++ .../phoenixd/templates/healthcheck.timer.j2 | 10 + .../phoenixd/templates/phoenixd.service.j2 | 31 + .../phoenixd/deploy_phoenixd_playbook.yml | 552 +----------------- 12 files changed, 436 insertions(+), 552 deletions(-) create mode 100644 ansible/roles/phoenixd/README.md rename ansible/{services/phoenixd/phoenixd_vars.yml => roles/phoenixd/defaults/main.yml} (69%) create mode 100644 ansible/roles/phoenixd/handlers/main.yml create mode 100644 ansible/roles/phoenixd/tasks/healthcheck.yml create mode 100644 ansible/roles/phoenixd/tasks/install.yml create mode 100644 ansible/roles/phoenixd/tasks/main.yml create mode 100644 ansible/roles/phoenixd/tasks/service.yml create mode 100644 ansible/roles/phoenixd/templates/healthcheck.service.j2 create mode 100644 ansible/roles/phoenixd/templates/healthcheck.sh.j2 create mode 100644 ansible/roles/phoenixd/templates/healthcheck.timer.j2 create mode 100644 ansible/roles/phoenixd/templates/phoenixd.service.j2 diff --git a/ansible/roles/phoenixd/README.md b/ansible/roles/phoenixd/README.md new file mode 100644 index 0000000..3c147e3 --- /dev/null +++ b/ansible/roles/phoenixd/README.md @@ -0,0 +1,56 @@ +# `phoenixd` + +Deploys and runs [phoenixd](https://phoenix.acinq.co/server), an ACINQ Lightning +node, on the edge host. LNBits uses it as a wallet backend. The HTTP API stays on +loopback — phoenixd is never published through Caddy. + +Converted from `deploy_phoenixd_playbook.yml` (552 lines) under Plan 6. The +playbook is now 18 lines. + +## Phases + +| | | +|---|---| +| `install.yml` | packages, system user, directories, versioned download and install | +| `service.yml` | systemd unit, start, then first-boot checks (config written, seed created) | +| `healthcheck.yml` | check script, unit, timer | + +## The seed + +`{{ phoenixd_data_dir }}/seed.dat` **is** the funds. phoenixd is deliberately +excluded from the automated backups (Plan 5, Model C): the seed is twelve fixed +words that never change, so an automated job would only manufacture more copies +of a static secret on more machines. Write them down offline, once. + +Note the live file is mode `0644`. That is phoenixd's own doing, not this role's, +and it is worth tightening. + +## Monitoring: one variable, no product knowledge + +The check asks the node itself — the service must be active **and** +`phoenix-cli getinfo` must return a `nodeId` — and records the answer in its exit +code, which systemd keeps: + +```bash +systemctl is-failed phoenixd-healthcheck.service +``` + +That is a complete answer with no monitoring system involved. To report +elsewhere, set `healthcheck_push_url` to anything accepting an HTTP ping. Gone +from this role: the embedded Python that created monitors over the Uptime Kuma +API, the `/tmp` credentials file, the push-URL file written and parsed back, and +the systemd `Environment=` rewrite. + +### Two things the conversion fixed + +**The check used to log an error once a minute.** Its `Environment=` push URL had +been empty since the decommissioning, and the script printed +`ERROR: UPTIME_KUMA_PUSH_URL not set` on every fire — roughly 1,400 times a day. +The exit code was still correct, so nothing was broken; it was pure noise, and +noise that trains you to ignore the log. An unset push URL is now normal and +silent. + +**`Enable and start phoenixd health check timer` was guarded by +`uptime_kuma_enabled`** and so had not run since the decommissioning — while the +timer itself was still live on the host from before. Ansible had quietly stopped +managing something that was still running. Ungated. diff --git a/ansible/services/phoenixd/phoenixd_vars.yml b/ansible/roles/phoenixd/defaults/main.yml similarity index 69% rename from ansible/services/phoenixd/phoenixd_vars.yml rename to ansible/roles/phoenixd/defaults/main.yml index 93921fd..ea9c889 100644 --- a/ansible/services/phoenixd/phoenixd_vars.yml +++ b/ansible/roles/phoenixd/defaults/main.yml @@ -35,15 +35,20 @@ phoenixd_http_bind_port: 9740 # Optional webhook for payment events. Leave empty to disable. phoenixd_webhook_url: "" -# Monitoring -phoenixd_healthcheck_script_path: /usr/local/bin/phoenixd-healthcheck-push.sh -phoenixd_healthcheck_service_name: phoenixd-healthcheck phoenixd_monitor_name: "Phoenixd" -# Remote access -remote_host_name: "{{ groups['edge'] | first }}" -remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_host_name) }}" -remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" -remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" -remote_port: "{{ hostvars.get(remote_host_name, {}).get('ansible_port', 22) }}" + +# --- Health check ----------------------------------------------------------- +# The check asks phoenixd itself whether it is healthy (service active AND the +# node answers getinfo with a nodeId) and records the answer in its exit code, +# which systemd keeps: +# systemctl is-failed phoenixd-healthcheck.service +# That is a complete answer with no monitoring system involved. +phoenixd_healthcheck_script_path: /usr/local/bin/phoenixd-healthcheck-push.sh +phoenixd_healthcheck_service_name: phoenixd-healthcheck + +# WHERE TO REPORT HEALTH — the one place to plug in monitoring. +# Empty means check, log, exit honestly, report nowhere. Any endpoint that +# accepts an HTTP ping works; nothing here is specific to a monitoring product. +healthcheck_push_url: "" diff --git a/ansible/roles/phoenixd/handlers/main.yml b/ansible/roles/phoenixd/handlers/main.yml new file mode 100644 index 0000000..7779684 --- /dev/null +++ b/ansible/roles/phoenixd/handlers/main.yml @@ -0,0 +1,12 @@ +--- +- name: Restart phoenixd + systemd: + name: phoenixd + state: restarted + daemon_reload: yes + +- name: Restart phoenixd health check timer + systemd: + name: "{{ phoenixd_healthcheck_service_name }}.timer" + state: restarted + daemon_reload: yes diff --git a/ansible/roles/phoenixd/tasks/healthcheck.yml b/ansible/roles/phoenixd/tasks/healthcheck.yml new file mode 100644 index 0000000..ad5cc22 --- /dev/null +++ b/ansible/roles/phoenixd/tasks/healthcheck.yml @@ -0,0 +1,68 @@ +--- +# Everything here answers "is phoenixd healthy" and records the answer. The +# Uptime Kuma specifics that used to follow it — an embedded Python script +# creating monitors over the API, a /tmp credentials file, a push-URL file read +# back and parsed, and a systemd Environment= rewrite — are gone. What reports +# where is now one variable, healthcheck_push_url. See the role README. +- name: Create phoenixd health check script + ansible.builtin.template: + src: healthcheck.sh.j2 + dest: "{{ phoenixd_healthcheck_script_path }}" + owner: root + group: root + mode: "0755" + validate: "bash -n %s" + +- name: Create phoenixd health check systemd service + ansible.builtin.template: + src: healthcheck.service.j2 + dest: "/etc/systemd/system/{{ phoenixd_healthcheck_service_name }}.service" + owner: root + group: root + mode: "0644" + notify: Restart phoenixd health check timer + +- name: Create phoenixd health check systemd timer + ansible.builtin.template: + src: healthcheck.timer.j2 + dest: "/etc/systemd/system/{{ phoenixd_healthcheck_service_name }}.timer" + owner: root + group: root + mode: "0644" + notify: Restart phoenixd health check timer + +- name: Reload systemd daemon after health check units + systemd: + daemon_reload: yes + +# Ungated on purpose. This was guarded by `uptime_kuma_enabled`, but enabling a +# timer is deployment, not monitoring — the deprecation banner swept it up with +# the push plumbing. The timer is in fact running on the host, from before the +# decommissioning, so the guard meant Ansible had stopped managing something +# that was still live. +- name: Enable and start phoenixd health check timer + systemd: + name: "{{ phoenixd_healthcheck_service_name }}.timer" + enabled: yes + state: started + +- name: Display post-install information + debug: + msg: | + ✓ phoenixd {{ phoenixd_version }} deployed + + Status: systemctl status phoenixd + Logs: journalctl -u phoenixd -f + CLI: sudo PHOENIX_DATADIR={{ phoenixd_data_dir }} phoenix-cli --http-bind-port {{ phoenixd_http_bind_port }} getinfo + HTTP API: http://{{ phoenixd_http_bind_ip }}:{{ phoenixd_http_bind_port }} (loopback only) + Data dir: {{ phoenixd_data_dir }} + Health: systemctl is-failed {{ phoenixd_healthcheck_service_name }}.service + + API password (needed to wire LNBits up to this node): + sudo grep '^http-password=' {{ phoenixd_data_dir }}/phoenix.conf + + ⚠️ BACK UP THE SEED: {{ phoenixd_data_dir }}/seed.dat + Losing it means losing the funds. phoenixd is deliberately excluded + from the automated backups (Plan 5, Model C) because the seed is 12 + fixed words — write them down offline, once: + sudo cat {{ phoenixd_data_dir }}/seed.dat diff --git a/ansible/roles/phoenixd/tasks/install.yml b/ansible/roles/phoenixd/tasks/install.yml new file mode 100644 index 0000000..89a4fa0 --- /dev/null +++ b/ansible/roles/phoenixd/tasks/install.yml @@ -0,0 +1,125 @@ +--- +- name: Install phoenixd runtime dependencies + apt: + name: + - unzip + - curl + state: present + update_cache: yes + +# System User and Directories +- name: Create phoenixd system group + group: + name: "{{ phoenixd_group }}" + system: yes + +- name: Create phoenixd system user + user: + name: "{{ phoenixd_user }}" + group: "{{ phoenixd_group }}" + system: yes + shell: /usr/sbin/nologin + home: "{{ phoenixd_home }}" + create_home: yes + comment: "phoenixd Lightning node" + +- name: Create phoenixd home directory + file: + path: "{{ phoenixd_home }}" + state: directory + owner: "{{ phoenixd_user }}" + group: "{{ phoenixd_group }}" + mode: "0750" + +- name: Create phoenixd data directory + file: + path: "{{ phoenixd_data_dir }}" + state: directory + owner: "{{ phoenixd_user }}" + group: "{{ phoenixd_group }}" + mode: "0700" + +# Download and Install +- name: Check if phoenixd is already installed + stat: + path: "{{ phoenixd_bin_dir }}/phoenixd" + register: phoenixd_binary + +- name: Check installed phoenixd version + command: "{{ phoenixd_bin_dir }}/phoenixd --version" + register: phoenixd_installed_version + changed_when: false + failed_when: false + when: phoenixd_binary.stat.exists + +- name: Decide whether phoenixd needs installing + set_fact: + phoenixd_needs_install: >- + {{ not phoenixd_binary.stat.exists + or phoenixd_version not in (phoenixd_installed_version.stdout | default('')) }} + +- name: Download phoenixd {{ phoenixd_version }} + get_url: + url: "{{ phoenixd_url }}" + dest: "/tmp/phoenixd-{{ phoenixd_version }}.zip" + mode: "0644" + when: phoenixd_needs_install | bool + +- name: Create temporary extraction directory + file: + path: /tmp/phoenixd-extract + state: directory + mode: "0755" + when: phoenixd_needs_install | bool + +- name: Extract phoenixd archive + unarchive: + src: "/tmp/phoenixd-{{ phoenixd_version }}.zip" + dest: /tmp/phoenixd-extract + remote_src: yes + when: phoenixd_needs_install | bool + +- name: Locate extracted binaries + find: + paths: /tmp/phoenixd-extract + patterns: "{{ item }}" + recurse: yes + file_type: file + register: phoenixd_extracted + loop: + - phoenixd + - phoenix-cli + when: phoenixd_needs_install | bool + +- name: Fail if the archive did not contain the expected binaries + assert: + that: + - item.files | length > 0 + fail_msg: "Could not find '{{ item.item }}' in the phoenixd {{ phoenixd_version }} archive" + loop: "{{ phoenixd_extracted.results }}" + loop_control: + label: "{{ item.item }}" + when: phoenixd_needs_install | bool + +- name: Install phoenixd and phoenix-cli binaries + copy: + src: "{{ item.files[0].path }}" + dest: "{{ phoenixd_bin_dir }}/{{ item.item }}" + remote_src: yes + owner: root + group: root + mode: "0755" + loop: "{{ phoenixd_extracted.results }}" + loop_control: + label: "{{ item.item }}" + when: phoenixd_needs_install | bool + notify: Restart phoenixd + +- name: Clean up phoenixd download artifacts + file: + path: "{{ item }}" + state: absent + loop: + - "/tmp/phoenixd-{{ phoenixd_version }}.zip" + - /tmp/phoenixd-extract + diff --git a/ansible/roles/phoenixd/tasks/main.yml b/ansible/roles/phoenixd/tasks/main.yml new file mode 100644 index 0000000..bc7ff05 --- /dev/null +++ b/ansible/roles/phoenixd/tasks/main.yml @@ -0,0 +1,6 @@ +--- +# import_tasks, not include_tasks: static imports stay visible to --list-tasks, +# which is how this conversion was verified against the playbook it replaced. +- ansible.builtin.import_tasks: install.yml +- ansible.builtin.import_tasks: service.yml +- ansible.builtin.import_tasks: healthcheck.yml diff --git a/ansible/roles/phoenixd/tasks/service.yml b/ansible/roles/phoenixd/tasks/service.yml new file mode 100644 index 0000000..79bf6f5 --- /dev/null +++ b/ansible/roles/phoenixd/tasks/service.yml @@ -0,0 +1,52 @@ +--- +- name: Build phoenixd command line arguments + set_fact: + phoenixd_args: >- + {{ (['--agree-to-terms-of-service'] if phoenixd_agree_tos else []) + + ['--chain', phoenixd_chain] + + ['--auto-liquidity', phoenixd_auto_liquidity] + + ['--http-bind-ip', phoenixd_http_bind_ip] + + ['--http-bind-port', phoenixd_http_bind_port | string] + + (['--max-mining-fee', phoenixd_max_mining_fee | string] if phoenixd_max_mining_fee else []) + + (['--webhook', phoenixd_webhook_url] if phoenixd_webhook_url else []) + + ['--silent'] }} + +- name: Create phoenixd systemd service + ansible.builtin.template: + src: phoenixd.service.j2 + dest: /etc/systemd/system/phoenixd.service + owner: root + group: root + mode: "0644" + notify: Restart phoenixd + +- name: Reload systemd daemon + systemd: + daemon_reload: yes + +- name: Enable and start phoenixd + systemd: + name: phoenixd + enabled: yes + state: started + +- name: Flush handlers so phoenixd is running before we inspect its data dir + meta: flush_handlers + +# --- First boot checks --- +- name: Wait for phoenixd to write its config file + wait_for: + path: "{{ phoenixd_data_dir }}/phoenix.conf" + state: present + timeout: 120 + +- name: Check that the seed file exists + stat: + path: "{{ phoenixd_data_dir }}/seed.dat" + register: phoenixd_seed_file + +- name: Fail if phoenixd did not create a seed + assert: + that: + - phoenixd_seed_file.stat.exists + fail_msg: "phoenixd started but {{ phoenixd_data_dir }}/seed.dat is missing - check 'journalctl -u phoenixd'" diff --git a/ansible/roles/phoenixd/templates/healthcheck.service.j2 b/ansible/roles/phoenixd/templates/healthcheck.service.j2 new file mode 100644 index 0000000..4060dbc --- /dev/null +++ b/ansible/roles/phoenixd/templates/healthcheck.service.j2 @@ -0,0 +1,14 @@ +[Unit] +Description=phoenixd Health Check +After=network.target phoenixd.service + +[Service] +Type=oneshot +User=root +ExecStart={{ phoenixd_healthcheck_script_path }} +Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/phoenixd/templates/healthcheck.sh.j2 b/ansible/roles/phoenixd/templates/healthcheck.sh.j2 new file mode 100644 index 0000000..d2cec21 --- /dev/null +++ b/ansible/roles/phoenixd/templates/healthcheck.sh.j2 @@ -0,0 +1,39 @@ +#!/bin/bash +# phoenixd health check — managed by Ansible (roles/phoenixd) +# +# Asks the node whether it is healthy and records the answer in the exit code, +# which systemd keeps: +# systemctl is-failed {{ phoenixd_healthcheck_service_name }}.service +# That is a complete answer on its own. Reporting anywhere else is optional. +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +export PHOENIX_DATADIR="{{ phoenixd_data_dir }}" + +check_phoenixd() { + # Service must be active and the node must answer getinfo. + # phoenix-cli reads the api password from $PHOENIX_DATADIR/phoenix.conf, + # but not the bind address, so pass it explicitly. + systemctl is-active --quiet phoenixd && \ + {{ phoenixd_bin_dir }}/phoenix-cli \ + --http-bind-ip {{ phoenixd_http_bind_ip }} \ + --http-bind-port {{ phoenixd_http_bind_port }} \ + getinfo 2>/dev/null | grep -q '"nodeId"' +} + +report() { + local status=$1 msg=$2 + # No push URL configured is NORMAL, not an error: the exit code below still + # answers the question. The previous version logged ERROR here on every + # single fire, once a minute, which is noise that trains you to ignore it. + [ -n "$PUSH_URL" ] || return 0 + curl -s --max-time 10 --retry 2 -o /dev/null \ + "${PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true +} + +if check_phoenixd; then + report "up" "OK" + exit 0 +else + echo "phoenixd is not responding" + report "down" "phoenixd not responding" + exit 1 +fi diff --git a/ansible/roles/phoenixd/templates/healthcheck.timer.j2 b/ansible/roles/phoenixd/templates/healthcheck.timer.j2 new file mode 100644 index 0000000..ee07257 --- /dev/null +++ b/ansible/roles/phoenixd/templates/healthcheck.timer.j2 @@ -0,0 +1,10 @@ +[Unit] +Description=phoenixd Health Check Timer + +[Timer] +OnBootSec=2min +OnUnitActiveSec=1min +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/roles/phoenixd/templates/phoenixd.service.j2 b/ansible/roles/phoenixd/templates/phoenixd.service.j2 new file mode 100644 index 0000000..44fdb94 --- /dev/null +++ b/ansible/roles/phoenixd/templates/phoenixd.service.j2 @@ -0,0 +1,31 @@ +[Unit] +Description=phoenixd - Lightning Network Node +Documentation=https://phoenix.acinq.co/server +After=network-online.target +Wants=network-online.target + +[Service] +Type=simple +User={{ phoenixd_user }} +Group={{ phoenixd_group }} +WorkingDirectory={{ phoenixd_home }} +Environment=PHOENIX_DATADIR={{ phoenixd_data_dir }} +ExecStart={{ phoenixd_bin_dir }}/phoenixd {{ phoenixd_args | join(' ') }} +Restart=always +RestartSec=30 +TimeoutStartSec=120 +TimeoutStopSec=120 +StandardOutput=journal +StandardError=journal + +# Hardening: the node only ever writes to its own data directory +NoNewPrivileges=true +PrivateTmp=true +ProtectSystem=strict +ProtectHome=read-only +ReadWritePaths={{ phoenixd_data_dir }} + +LimitNOFILE=65535 + +[Install] +WantedBy=multi-user.target diff --git a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml index 8b30a1f..1e53ff0 100644 --- a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml +++ b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml @@ -1,24 +1,6 @@ --- -# phoenixd Deployment Playbook -# -# Deploys phoenixd (https://phoenix.acinq.co/server), the server version of the -# Phoenix Lightning wallet, on vipy so LNBits can use it as a wallet backend -# over loopback. -# -# What this does: -# 1. Downloads the pinned phoenixd release and installs phoenixd + phoenix-cli -# 2. Creates a dedicated system user and a 0700 data directory -# 3. Creates and enables a systemd service -# 4. Creates a push-monitor health check script + systemd timer -# 5. Registers a push monitor in Uptime Kuma -# -# The HTTP API stays bound to 127.0.0.1 and is NOT proxied by Caddy: phoenixd -# holds funds and its API is protected by a single password. Anything that needs -# it either runs on this host or reaches it over the Tailscale mesh. -# -# ⚠️ After the first run, back up {{ phoenixd_data_dir }}/seed.dat. Losing it -# means losing the funds. See setup_backup_phoenixd_to_lapy.yml. - +# phoenixd: Lightning node on the edge host, used by LNBits as a wallet backend. +# Never exposed through Caddy — the HTTP API stays on loopback. - name: Deploy phoenixd on the edge host hosts: edge become: yes @@ -26,527 +8,11 @@ - ../../infra_vars.yml - ../../services_config.yml - ../../infra_secrets.yml - - ./phoenixd_vars.yml vars: - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - # =========================================== - # Prerequisites - # =========================================== - - name: Install phoenixd runtime dependencies - apt: - name: - - unzip - - curl - state: present - update_cache: yes - - # =========================================== - # System User and Directories - # =========================================== - - name: Create phoenixd system group - group: - name: "{{ phoenixd_group }}" - system: yes - - - name: Create phoenixd system user - user: - name: "{{ phoenixd_user }}" - group: "{{ phoenixd_group }}" - system: yes - shell: /usr/sbin/nologin - home: "{{ phoenixd_home }}" - create_home: yes - comment: "phoenixd Lightning node" - - - name: Create phoenixd home directory - file: - path: "{{ phoenixd_home }}" - state: directory - owner: "{{ phoenixd_user }}" - group: "{{ phoenixd_group }}" - mode: "0750" - - - name: Create phoenixd data directory - file: - path: "{{ phoenixd_data_dir }}" - state: directory - owner: "{{ phoenixd_user }}" - group: "{{ phoenixd_group }}" - mode: "0700" - - # =========================================== - # Download and Install - # =========================================== - - name: Check if phoenixd is already installed - stat: - path: "{{ phoenixd_bin_dir }}/phoenixd" - register: phoenixd_binary - - - name: Check installed phoenixd version - command: "{{ phoenixd_bin_dir }}/phoenixd --version" - register: phoenixd_installed_version - changed_when: false - failed_when: false - when: phoenixd_binary.stat.exists - - - name: Decide whether phoenixd needs installing - set_fact: - phoenixd_needs_install: >- - {{ not phoenixd_binary.stat.exists - or phoenixd_version not in (phoenixd_installed_version.stdout | default('')) }} - - - name: Download phoenixd {{ phoenixd_version }} - get_url: - url: "{{ phoenixd_url }}" - dest: "/tmp/phoenixd-{{ phoenixd_version }}.zip" - mode: "0644" - when: phoenixd_needs_install | bool - - - name: Create temporary extraction directory - file: - path: /tmp/phoenixd-extract - state: directory - mode: "0755" - when: phoenixd_needs_install | bool - - - name: Extract phoenixd archive - unarchive: - src: "/tmp/phoenixd-{{ phoenixd_version }}.zip" - dest: /tmp/phoenixd-extract - remote_src: yes - when: phoenixd_needs_install | bool - - - name: Locate extracted binaries - find: - paths: /tmp/phoenixd-extract - patterns: "{{ item }}" - recurse: yes - file_type: file - register: phoenixd_extracted - loop: - - phoenixd - - phoenix-cli - when: phoenixd_needs_install | bool - - - name: Fail if the archive did not contain the expected binaries - assert: - that: - - item.files | length > 0 - fail_msg: "Could not find '{{ item.item }}' in the phoenixd {{ phoenixd_version }} archive" - loop: "{{ phoenixd_extracted.results }}" - loop_control: - label: "{{ item.item }}" - when: phoenixd_needs_install | bool - - - name: Install phoenixd and phoenix-cli binaries - copy: - src: "{{ item.files[0].path }}" - dest: "{{ phoenixd_bin_dir }}/{{ item.item }}" - remote_src: yes - owner: root - group: root - mode: "0755" - loop: "{{ phoenixd_extracted.results }}" - loop_control: - label: "{{ item.item }}" - when: phoenixd_needs_install | bool - notify: Restart phoenixd - - - name: Clean up phoenixd download artifacts - file: - path: "{{ item }}" - state: absent - loop: - - "/tmp/phoenixd-{{ phoenixd_version }}.zip" - - /tmp/phoenixd-extract - - # =========================================== - # Systemd Service - # =========================================== - - name: Build phoenixd command line arguments - set_fact: - phoenixd_args: >- - {{ (['--agree-to-terms-of-service'] if phoenixd_agree_tos else []) - + ['--chain', phoenixd_chain] - + ['--auto-liquidity', phoenixd_auto_liquidity] - + ['--http-bind-ip', phoenixd_http_bind_ip] - + ['--http-bind-port', phoenixd_http_bind_port | string] - + (['--max-mining-fee', phoenixd_max_mining_fee | string] if phoenixd_max_mining_fee else []) - + (['--webhook', phoenixd_webhook_url] if phoenixd_webhook_url else []) - + ['--silent'] }} - - - name: Create phoenixd systemd service - copy: - dest: /etc/systemd/system/phoenixd.service - content: | - [Unit] - Description=phoenixd - Lightning Network Node - Documentation=https://phoenix.acinq.co/server - After=network-online.target - Wants=network-online.target - - [Service] - Type=simple - User={{ phoenixd_user }} - Group={{ phoenixd_group }} - WorkingDirectory={{ phoenixd_home }} - Environment=PHOENIX_DATADIR={{ phoenixd_data_dir }} - ExecStart={{ phoenixd_bin_dir }}/phoenixd {{ phoenixd_args | join(' ') }} - Restart=always - RestartSec=30 - TimeoutStartSec=120 - TimeoutStopSec=120 - StandardOutput=journal - StandardError=journal - - # Hardening: the node only ever writes to its own data directory - NoNewPrivileges=true - PrivateTmp=true - ProtectSystem=strict - ProtectHome=read-only - ReadWritePaths={{ phoenixd_data_dir }} - - LimitNOFILE=65535 - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: "0644" - notify: Restart phoenixd - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start phoenixd - systemd: - name: phoenixd - enabled: yes - state: started - - - name: Flush handlers so phoenixd is running before we inspect its data dir - meta: flush_handlers - - # =========================================== - # First Boot Checks - # =========================================== - - name: Wait for phoenixd to write its config file - wait_for: - path: "{{ phoenixd_data_dir }}/phoenix.conf" - state: present - timeout: 120 - - - name: Check that the seed file exists - stat: - path: "{{ phoenixd_data_dir }}/seed.dat" - register: phoenixd_seed_file - - - name: Fail if phoenixd did not create a seed - assert: - that: - - phoenixd_seed_file.stat.exists - fail_msg: "phoenixd started but {{ phoenixd_data_dir }}/seed.dat is missing - check 'journalctl -u phoenixd'" - - # =========================================== - # Health Check Script + Systemd Timer - # =========================================== - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create phoenixd health check script - when: uptime_kuma_enabled | default(false) - copy: - dest: "{{ phoenixd_healthcheck_script_path }}" - content: | - #!/bin/bash - # Checks phoenixd and pushes the result to Uptime Kuma. - UPTIME_KUMA_PUSH_URL="${UPTIME_KUMA_PUSH_URL}" - export PHOENIX_DATADIR="{{ phoenixd_data_dir }}" - - check_phoenixd() { - # Service must be active and the node must answer getinfo. - # phoenix-cli reads the api password from $PHOENIX_DATADIR/phoenix.conf, - # but not the bind address, so pass it explicitly. - systemctl is-active --quiet phoenixd && \ - {{ phoenixd_bin_dir }}/phoenix-cli \ - --http-bind-ip {{ phoenixd_http_bind_ip }} \ - --http-bind-port {{ phoenixd_http_bind_port }} \ - getinfo 2>/dev/null | grep -q '"nodeId"' - } - - push_to_uptime_kuma() { - local status=$1 - local msg=$2 - if [ -z "$UPTIME_KUMA_PUSH_URL" ]; then - echo "ERROR: UPTIME_KUMA_PUSH_URL not set" - return 1 - fi - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${UPTIME_KUMA_PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true - } - - if check_phoenixd; then - push_to_uptime_kuma "up" "OK" - exit 0 - else - push_to_uptime_kuma "down" "phoenixd not responding" - exit 1 - fi - owner: root - group: root - mode: "0755" - - - name: Create phoenixd health check systemd service - copy: - dest: "/etc/systemd/system/{{ phoenixd_healthcheck_service_name }}.service" - content: | - [Unit] - Description=phoenixd Health Check - After=network.target phoenixd.service - - [Service] - Type=oneshot - User=root - ExecStart={{ phoenixd_healthcheck_script_path }} - Environment=UPTIME_KUMA_PUSH_URL= - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: "0644" - - - name: Create phoenixd health check systemd timer - copy: - dest: "/etc/systemd/system/{{ phoenixd_healthcheck_service_name }}.timer" - content: | - [Unit] - Description=phoenixd Health Check Timer - - [Timer] - OnBootSec=2min - OnUnitActiveSec=1min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: "0644" - - - name: Reload systemd daemon after health check units - systemd: - daemon_reload: yes - - - name: Enable and start phoenixd health check timer - when: uptime_kuma_enabled | default(false) - systemd: - name: "{{ phoenixd_healthcheck_service_name }}.timer" - enabled: yes - state: started - - # =========================================== - # Uptime Kuma Push Monitor Setup - # =========================================== - - name: Create Uptime Kuma push monitor setup script for phoenixd - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_phoenixd_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import time - import traceback - import yaml - - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_phoenixd_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - try: - api.add_monitor(type='group', name='services') - except Exception: - time.sleep(2) - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - existing = next((m for m in monitors if m.get('name') == monitor_name), None) - - push_url = None - - if existing: - print(f"Monitor '{monitor_name}' already exists (ID: {existing['id']})") - push_token = existing.get('pushToken') or existing.get('push_token') - if push_token: - push_url = f"{url}/api/push/{push_token}" - else: - print(f"Creating push monitor '{monitor_name}'...") - try: - api.add_monitor( - type=MonitorType.PUSH, - name=monitor_name, - parent=group['id'], - interval=90, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - except Exception as e: - # socketio timeout: add_monitor may have succeeded server-side - print(f"add_monitor raised (possibly timeout): {e}", file=sys.stderr) - time.sleep(2) - - monitors = api.get_monitors() - new_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - if new_monitor: - push_token = new_monitor.get('pushToken') or new_monitor.get('push_token') - if push_token: - push_url = f"{url}/api/push/{push_token}" - - api.disconnect() - - if push_url: - print(f"PUSH_URL={push_url}") - with open('/tmp/phoenixd_push_url.txt', 'w') as f: - f.write(push_url) - - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: "0755" - - - name: Create temporary config for push monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_phoenixd_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_name: "{{ phoenixd_monitor_name }}" - mode: "0644" - - - name: Run Uptime Kuma push monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_phoenixd_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Display monitor setup output - debug: - msg: "{{ monitor_setup.stdout_lines }}" - when: monitor_setup.stdout is defined - - - name: Read push URL from file - when: uptime_kuma_enabled | default(false) - slurp: - src: /tmp/phoenixd_push_url.txt - delegate_to: localhost - become: no - register: push_url_file - ignore_errors: yes - - - name: Parse push URL - set_fact: - phoenixd_push_url: "{{ push_url_file.content | b64decode | trim }}" - when: push_url_file.content is defined - - - name: Update health check service with push URL - lineinfile: - path: "/etc/systemd/system/{{ phoenixd_healthcheck_service_name }}.service" - regexp: "^Environment=UPTIME_KUMA_PUSH_URL=" - line: "Environment=UPTIME_KUMA_PUSH_URL={{ phoenixd_push_url }}" - when: phoenixd_push_url is defined - notify: Restart phoenixd health check timer - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_phoenixd_monitor.py - - /tmp/ansible_phoenixd_config.yml - - /tmp/phoenixd_push_url.txt - - # =========================================== - # Post-install Notes - # =========================================== - - name: Display post-install information - debug: - msg: | - ✓ phoenixd {{ phoenixd_version }} deployed - - Status: systemctl status phoenixd - Logs: journalctl -u phoenixd -f - CLI: sudo PHOENIX_DATADIR={{ phoenixd_data_dir }} phoenix-cli --http-bind-port {{ phoenixd_http_bind_port }} getinfo - HTTP API: http://{{ phoenixd_http_bind_ip }}:{{ phoenixd_http_bind_port }} (loopback only) - Data dir: {{ phoenixd_data_dir }} - - API password (needed to wire LNBits up to this node): - sudo grep '^http-password=' {{ phoenixd_data_dir }}/phoenix.conf - - ⚠️ BACK UP THE SEED NOW: {{ phoenixd_data_dir }}/seed.dat - Losing it means losing the funds. Run - services/phoenixd/setup_backup_phoenixd_to_lapy.yml and also keep - the 12 words somewhere offline. - - handlers: - - name: Restart phoenixd - systemd: - name: phoenixd - state: restarted - daemon_reload: yes - - - name: Restart phoenixd health check timer - systemd: - name: "{{ phoenixd_healthcheck_service_name }}.timer" - state: restarted - daemon_reload: yes + # phoenixd's health check has never reported anywhere since the Uptime Kuma + # decommissioning — its systemd Environment= was left empty. Leaving it empty + # preserves that; the check still runs and its exit code is still the answer. + # Set this to plug in whatever monitoring replaces it. + healthcheck_push_url: "{{ healthcheck_push_urls.phoenixd | default('') }}" + roles: + - phoenixd From 35b3817e158e8cbd9134b4fd6951d94435e7e43f Mon Sep 17 00:00:00 2001 From: counterweight Date: Sat, 12 Sep 2026 18:43:14 +0200 Subject: [PATCH 49/67] tofu: stop gitignoring the lock file and the VM inventory MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The root .gitignore excluded .terraform.lock.hcl and every *.tfvars, which hid two things that belong in version control: - .terraform.lock.hcl pins the provider hashes. versions.tf tracks Telmate/proxmox 3.0.2-rc05, a release candidate, so the version constraint alone is not enough if that tag is ever re-published. - terraform.tfvars held one real secret (proxmox_api_token_secret) plus the entire vms map — 7 VMs with their vmids, sizes and static IPs. That is infra definition, and it existed only on one laptop. Meanwhile the committed terraform.tfvars.example still advertised web1/db1. Split at the credential boundary: the provider auth triple stays in the gitignored terraform.tfvars, everything else moves to vms.auto.tfvars, which is committed and auto-loaded (no -var-file needed). terraform.tfvars.example is now credentials-only. `tofu plan` reports no changes. State stays ignored — it carries cloud-init attributes and should not be in git. Noted in the README that it has no remote backend, and that state manages two VMs (bastion-box, nonkeiwaisi-box) the map does not declare. Co-Authored-By: Claude Opus 5 (1M context) --- .gitignore | 11 ++-- tofu/nodito/.terraform.lock.hcl | 24 +++++++++ tofu/nodito/README.md | 22 +++++--- tofu/nodito/terraform.tfvars.example | 40 +++------------ tofu/nodito/vms.auto.tfvars | 77 ++++++++++++++++++++++++++++ 5 files changed, 130 insertions(+), 44 deletions(-) create mode 100644 tofu/nodito/.terraform.lock.hcl create mode 100644 tofu/nodito/vms.auto.tfvars diff --git a/.gitignore b/.gitignore index 9ca7b8e..471cd6f 100644 --- a/.gitignore +++ b/.gitignore @@ -1,13 +1,16 @@ # OpenTofu / Terraform .terraform/ .tofu/ -.terraform.lock.hcl -.tofu.lock.hcl terraform.tfstate terraform.tfstate.* crash.log -*.tfvars -*.tfvars.json + +# Provider credentials only. Non-secret infra config (the vms map) is committed +# as *.auto.tfvars, and *.lock.hcl is committed on purpose so provider hashes +# are pinned. +terraform.tfvars +terraform.tfvars.json +*secrets.auto.tfvars venv/* .env diff --git a/tofu/nodito/.terraform.lock.hcl b/tofu/nodito/.terraform.lock.hcl new file mode 100644 index 0000000..fadff38 --- /dev/null +++ b/tofu/nodito/.terraform.lock.hcl @@ -0,0 +1,24 @@ +# This file is maintained automatically by "tofu init". +# Manual edits may be lost in future updates. + +provider "registry.opentofu.org/telmate/proxmox" { + version = "3.0.2-rc05" + constraints = "3.0.2-rc05" + hashes = [ + "h1:QfHovHn8h9uJXdJ+urOuiD7R46OXmdZQiRcCBaV6AD4=", + "zh:042d748367f33aaf440698644be4f2a2875f9db31915c1ef84616f176fc6174f", + "zh:1488781da1920d60d933c8ce926c34b5e989ffae58e3fbe437973d2b1d2faafc", + "zh:283dd6f74627f1d1d75d616b31f8ced3f97fd5277a07c9535e85cfa765d7a321", + "zh:378f1c2da21aeea083ac2e632db274a02c7a01e2486a40d3c813d05a21142db3", + "zh:38d63d0961f8c32273392caaace30f50cff8ab06e5dda17f67a8827ebffeba98", + "zh:52159782df101ec98f20faff81e8f2d9d92cb4ec903314fcddcc57ec16cdaacb", + "zh:6ca47b90c66b1d2706cb3cbb05da8b3f90a202c4865010202b2962e2b64d217e", + "zh:6e7b85cb2380e4dc0be694dd0e4a24927f7f66df41960eca3cfe907443d4f0b9", + "zh:758775f733673ab5c196db6a33648458037746f94d4bef7ce148cb01474efe2d", + "zh:7c31a3ca6d52db39da2bdd60be37af71d59d808fc206de50fe661535ea436da3", + "zh:af16984350a2f4d77c21f66a479007801e2527543310567c99cd82eb421e249e", + "zh:c1f965d3f96cf3f87af2c12ab9d4bde42f8ef660f8dc34ba3cfc9b20435a7269", + "zh:c2b9022a31103919a5ffbac6ee8d7feb6c4f5f580c1766f769569c2e8e4ce7f1", + "zh:e90162c42f1237323291e3d0de0c62701b3f89350fae18246da06702f41a6123", + ] +} diff --git a/tofu/nodito/README.md b/tofu/nodito/README.md index 3a0b18f..bff6585 100644 --- a/tofu/nodito/README.md +++ b/tofu/nodito/README.md @@ -27,16 +27,17 @@ This directory lets you declare VMs on the `nodito` Proxmox node and apply with - The Ansible template exists: `debian-13-cloud-init` (VMID 9001 by default). ### Provider Auth -Create a `terraform.tfvars` (copy from `terraform.tfvars.example`) and set: +Credentials are the only thing not in git. Copy `terraform.tfvars.example` to +`terraform.tfvars` (gitignored) and set: - `proxmox_api_url` (e.g. `https://nodito:8006/api2/json`) - `proxmox_api_token_id` (e.g. `root@pam!tofu`) - `proxmox_api_token_secret` -- `ssh_authorized_keys` (your public key content) -Alternatively, you can export env vars and reference them in a tfvars file. +Alternatively, export them as `TF_VAR_proxmox_api_token_secret` etc. ### Declare VMs -Edit `terraform.tfvars` and fill the `vms` map. Example entry: +VMs are declared in `vms.auto.tfvars`, which is committed. `*.auto.tfvars` is +loaded automatically, so it needs no `-var-file`. Example entry: ``` vms = { web1 = { @@ -54,15 +55,24 @@ All VM disks are created on `zfs_storage_name` (defaults to `proxmox-tank-1`). N ### Usage ``` tofu init -tofu plan -var-file=terraform.tfvars -tofu apply -var-file=terraform.tfvars +tofu plan +tofu apply ``` +`terraform.tfvars` and `vms.auto.tfvars` are both auto-loaded. > VMs are created once and then protected: the module sets `lifecycle.prevent_destroy = true` and ignores subsequent config changes. After the initial apply, manage day‑2 changes directly in Proxmox (or remove the lifecycle block if you need OpenTofu to own ongoing updates). ### Notes - Clones are full clones by default (`full_clone = true`). - Cloud-init injects `cloud_init_user` and `ssh_authorized_keys`. +- `.terraform.lock.hcl` is committed: it pins the provider hashes, which matters + because `versions.tf` tracks a release candidate (`3.0.2-rc05`). +- State is local (`terraform.tfstate`, gitignored) and has no remote backend, so + it exists only on the machine that last ran `tofu apply`. +- The map is not a complete inventory of nodito: state also manages + `bastion-box` (1100) and `nonkeiwaisi-box` (3300), which are not declared in + `vms.auto.tfvars`. `tofu plan` is clean today, but relaxing the `lifecycle` + block without first declaring them would put them up for destruction. - Disks use `scsi0` on ZFS with `discard` enabled. diff --git a/tofu/nodito/terraform.tfvars.example b/tofu/nodito/terraform.tfvars.example index cc88b3f..37b6d7a 100644 --- a/tofu/nodito/terraform.tfvars.example +++ b/tofu/nodito/terraform.tfvars.example @@ -1,35 +1,7 @@ -proxmox_api_url = "https://nodito:8006/api2/json" -proxmox_api_token_id = "root@pam!tofu" -proxmox_api_token_secret = "REPLACE_ME" - -proxmox_node = "nodito" -zfs_storage_name = "proxmox-tank-1" -template_name = "debian-13-cloud-init" -cloud_init_user = "counterweight" - -# paste your ~/.ssh/id_ed25519.pub or similar -ssh_authorized_keys = < Date: Sat, 12 Sep 2026 18:49:11 +0200 Subject: [PATCH 50/67] mempool: convert to a role, de-Uptime-Kuma the health checks 745-line playbook becomes 37 lines (the role, plus the Caddy play for the edge host) and a 408-line role with docker/deploy/healthcheck phases and six templates. mempool_vars.yml is deleted; its content is the role's defaults. Three health checks are kept, not collapsed: Mempool is three moving parts and knowing which one is down is the point. Each has its own script, unit, timer and push_url, driven by a mempool_healthchecks list. The Uptime Kuma specifics are gone - the embedded Python creating monitors over the API, the /tmp credentials file, the push-URL file read back and parsed, three Environment= rewrites - and the three live push URLs are preserved from the vault, so reporting is unchanged. `Enable and start health check timers` and `Display deployment status` were both guarded by uptime_kuma_enabled despite being deployment tasks. Third service in a row with that pattern: the deprecation banner was applied to contiguous blocks, so anything sitting near the push plumbing was disabled with it. Ungated. TWO OWNERSHIP PROBLEMS, different in kind: - MINE: I wrote `owner: root` on docker-compose.yml where the original says `owner: "{{ ansible_user }}"`. A straight violation of extract-mechanically- change-nothing, caught only by reading the check-mode diff line by line. Reverted to match the original. - PRE-EXISTING, and dangerous: the playbook declared `owner: "{{ ansible_user }}"` (1000) on the MariaDB data directory, which the container owns as uid 999. Confirmed against `git show HEAD:` before concluding it was not mine. It had drifted since the containers were created and went unnoticed because the playbook had not been run since. This was not academic. The first real run pulled a newer mariadb:10.11 and recreated mempool-db; with the chown still in place MariaDB would have come back to a data directory it could not write. The role now ensures the directory exists and leaves ownership to the container. Verified after the run: /opt/mempool/mysql is still 999:999 and all three containers are healthy. This is a deliberate behaviour change, not part of the extraction. It is in this commit rather than a follow-up because the faithful version was never safe to run, so there was no intermediate state worth recording as verified. mempool_frontend_port moved to services_config.yml: two hosts need it (this role deploys the frontend, the Caddy play proxies to it from the edge host) and a role default is invisible to the second play. caddy_site's parameter assert caught this loudly - "'mempool_frontend_port' is undefined" - rather than silently. Verified: check-mode diff clean apart from unavoidable check-mode artifacts; first run ok=24 changed=5, zero failures; second run changed=2 - the two bare `command:` tasks (pull, compose up) that have no changed_when and always report changed. That is the idempotent floor. All three health checks report ExecMainStatus 0 with their push URLs intact. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 334 ++++---- ansible/infra_secrets.yml | 334 ++++---- ansible/roles/mempool/README.md | 69 ++ ansible/roles/mempool/defaults/main.yml | 55 ++ ansible/roles/mempool/tasks/deploy.yml | 75 ++ ansible/roles/mempool/tasks/docker.yml | 72 ++ ansible/roles/mempool/tasks/healthcheck.yml | 58 ++ ansible/roles/mempool/tasks/main.yml | 6 + .../mempool/templates/docker-compose.yml.j2 | 75 ++ .../templates/healthcheck-backend.sh.j2 | 18 + .../templates/healthcheck-frontend.sh.j2 | 18 + .../templates/healthcheck-mariadb.sh.j2 | 21 + .../mempool/templates/healthcheck.service.j2 | 14 + .../mempool/templates/healthcheck.timer.j2 | 10 + .../mempool/deploy_mempool_playbook.yml | 734 +----------------- ansible/services/mempool/mempool_vars.yml | 33 - ansible/services_config.yml | 6 + 17 files changed, 856 insertions(+), 1076 deletions(-) create mode 100644 ansible/roles/mempool/README.md create mode 100644 ansible/roles/mempool/defaults/main.yml create mode 100644 ansible/roles/mempool/tasks/deploy.yml create mode 100644 ansible/roles/mempool/tasks/docker.yml create mode 100644 ansible/roles/mempool/tasks/healthcheck.yml create mode 100644 ansible/roles/mempool/tasks/main.yml create mode 100644 ansible/roles/mempool/templates/docker-compose.yml.j2 create mode 100644 ansible/roles/mempool/templates/healthcheck-backend.sh.j2 create mode 100644 ansible/roles/mempool/templates/healthcheck-frontend.sh.j2 create mode 100644 ansible/roles/mempool/templates/healthcheck-mariadb.sh.j2 create mode 100644 ansible/roles/mempool/templates/healthcheck.service.j2 create mode 100644 ansible/roles/mempool/templates/healthcheck.timer.j2 delete mode 100644 ansible/services/mempool/mempool_vars.yml diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index 6c01d31..6ac207c 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,162 +1,174 @@ $ANSIBLE_VAULT;1.1;AES256 -32333164303734353564643266316239636365313631613866306666643538353537323939373066 -3439396330326266396531303737363131313831396365650a343832363564356337396530383438 -36356465623634623361643436383262393339663134373630666363613464653437666164393731 -3462326463346535310a396531393761643062643563613964613234666531643139666535323734 -63653630616138373539343434636434326466353134643864396466373435613738653536356532 -36313432373138316532663464353638386530666231376636643931663338303663346665363039 -39323734343934393037633766626132333835363265653538373266323236336137356630353834 -63306164313434363130306339346435643463376137366534643033366537383861363531353861 -34363966303234333133336466646635623136396138663637613133613861313866646634643132 -34383461366133333430633631353561653037346339626165346564353630373363643065323535 -61386237353135636665363538623538313039343965363535323734653763366238373334323065 -64626135383736316135303731656231313263306266636563343465653666326663333536383435 -32643537616161623830656161633763303566656462356235633866303165383663386435363133 -61353430656337656364646231383332316534316233393333646436353062316461366439613030 -62346366383566356661376463616137373062303363346333636134633465396238343761363139 -38616336633532366165376237626337333933353935613066303865303536303464633834643533 -31376236333133336166373635626130316632396561393930353065616465663336393938323539 -35343531306261343733626539346265386436643135326461353734346164633233383237393731 -32303030623564666539383835333064323630393539393062313435383663303334383134666436 -63336564353464313534323866383533323365393761643565633430346263346666623638303030 -38653461616661346661316537306666313165613163303835363636616561343335623131636633 -36636263356530366663616461316635316365393666313066326265306335363038366561356339 -37366231656330643764613664653735633963666638623961653134316136666336396438333864 -65653738623638373636366666663463373035363862396565396166643332343934626261353938 -34666638383531366532663963323533323965373439623735353661373862393830623934313535 -62363237663764356338383133336463303234393262623061363062373938613262383836323431 -35336133306139386336373965393132393431343535306162383337643961323039373530646330 -35383962303263353234376164633530666264386338376335653465616161383532613830636562 -36613837326564313565633632353964346537303337316233623033383961363737313861393234 -64313130306264636134626638396661353362346439373463663965653165613436363633323238 -39356430613033643334363731346333346639643563633162333636653066386233373063653138 -38623630353265663332366630316362633135633362313735306533333962373433376237366266 -64643266323862383363633033656465303032336134623036646530323264323532653234306363 -64326663343064643865373164306661613463366561383737363535303861646634353139636666 -66626466303037363064303865313830356531313834353165303839326238613962313261353536 -63626133663765373763623930623963653038313661656131666261356236366565626663323831 -32313130366135616662616639386436656265346635613762353832626466323261346333373661 -33656338326631393064313762343161363832643030303737663639356261633131353937396135 -34666361353266373661626534353134663235343662636435383032636261636637636361613631 -30373965373638386664313432386264303761376266363161363633343831633032356639343836 -36306163303363353534313466353863633834393337303161313431343165346162653537346130 -36393439386232613865613837346563643031393530646433353936383463303562326331313738 -30646537616264333332323562363237323530313333386531323066343335323133366566383935 -34613232323965626339653132323162626234356135323436353263306137343130346337326531 -62396164386661303366363566393833353130643636613865616433633166666532653937646262 -63323530616533373962626264633236313064306336633063356536343862316237383166346264 -65666466366435653134303164613632666336626630373764333534393164303132656530343031 -32386137633135376531373662336131313030303436333833663234343938323232343832633234 -35623361393533316232393431646561616535643638383533353266646235313736356366343231 -30383962353833326161663534396565316139393439366631313731663737656332646361326331 -35623038343931666231653731346435653162326237376339663936343933346264343564666132 -62393230363262323366313963373737646138336163313434376462383964396337313030383831 -35643865303738393736323032353239633631306262663536663732623933383431383563646664 -65326230323239353831623232633935636632636438636165376633313364643637373233323231 -30323534333131623138636266396565373964633963626536383930316663656634636638316363 -36326438363463386138633133346666383163323936653131666339333430316363393966316130 -66333732333565646564303633373464666461373437656634306564336432323465386631323034 -32376635666139313465393539353438626534646332323537363163653233353030366231306234 -31663065393062633438343438303762633931653564366666336662323430366466333334303763 -64356138373035323862326137643963323865666539316439333064343961303838636265623735 -36636233373065303239363461373239646662346162326164613332373761326561656234636438 -39636231386531353236303933343661393333383730636366626263393534656363366337316432 -66393162313731616165343362643162376334346262353730653738323138646463666433383963 -31626131373736633662366366376566656438343330383330616235376239353661663663313233 -63373836363638366432376130633930343862613436633263613538336536616163656364656561 -35353163323663663833343834373036366531623536303138613035353064303065623761376462 -32656363613765316635666639383865636538326130386535316132623238393730353631346136 -65353864393863376632356437303565626262343039636436636335383337623465336263623439 -61393030636466303836663166613766326164626639313965373734643466373663343565343631 -35356431303139396131346335346664373830663361616136306535643431353037636431343132 -32613164333232346335663236343238653033643133316564666534323130373861623962393461 -62386336326461326239383534303631346639343765393839656364353762356266653733303238 -37393533366164393863373134633439383765616435346239333065623438353431663035306132 -30363663383765356365333234323061326163653463646566373764363037373932623032366133 -64626633643737323631306539333531343239326639643166393435323731353932336262633963 -62363561616635356230626431626232383439633738306532636361336334616238643665323266 -35666165396566323639303733623364353436613966336337656363633762393664303939656537 -33656431663431396237643536613233376561313261323635613634613439633533393361333435 -33616136633936386135623563616530663833643339623439366430646639646162613863643565 -32343732613134656332653163623437366637656537666334653239366638653537393639396234 -63613362623037663465626438646136396362663033376261376166306138363436666566393838 -63363262386261316432313235323166333334346539343663303265303535636439653130366566 -38383632326433323036356266646138366233656131663633613236303137336437383333653430 -35653738623635656661393232333334643937303631663464383239353635393735353833333265 -38303831623130303634666464316465653639383230623662326534333136616561666166613930 -37643632393265326364376230643634356538626337633638663634326538396536383633303633 -62613532336435326161626263376462363162303762613835663831623539623562336361333031 -37346438343230363433633738643064336235323438346534643463333930626430653131623538 -61363735336332633663383136613633376430303133366634643839636562656431663737336431 -65653164393331626532356566623830623931373431393330633638306663376566346136643863 -61653461633637643663303761326464636339336635383938633039613132336139663762363734 -34313831366232383562363834653232613665616266323033336635653933626630346132643164 -66336663363161616431316666383430326266636134396239656561643763386438306434373166 -65616637356366353336346539396665396364383162383533323837303134313664326338366131 -64336333316531353661616561656430386462666565373535613134336634383136613638323565 -34303130316162376137303137396330343466383263363638306334386161613535303735623537 -30613065383963666166336135663632303038326163346630376466393761663738653532333466 -35346161373237383766393063623939363664356639653862353132353930383261643761363135 -66323364376432363437656239626362313835333537663030386435623764373937636130663331 -31663636616261636663306230653264353566313966333665383938393731336538363665636665 -33383334613565393139613138633636656234386263663932323364653361623433326438326132 -34636633663838636634323236636564333630666362356333333230346333373833653365363135 -37326630386439356139653865643039666237313833306161643263663363393965633163366436 -65373132623265363134343233653366383665633163363766353066653863373937613631373332 -34653835636565353731396331633133393435373739306663343632323234633462356630666238 -36316634303031346138393835363565366131663534323933623166653662323931386135363432 -36626434323536326636653165626661386663393134653061326634633162653666393661353762 -31343766383933633261366437396332356362663539623032663435356231656430366532623535 -35636361343330333263636230383030326139323065373032376238333162343462383566333264 -66386439303163616638353635303736376461326533623234633264353865393065363765366636 -63383034373631656331386439633863613731633930646330613931303430366236643735633434 -33346638356330643534396236633630616134626237613363306334383639336161306235626133 -32616339633033343431633131666131383036373335303739663634653831376563363034363365 -63366532393761323862373434656264323938313334333234323162353663306263636435653636 -65343531663234636362623139653439386132316430346536623864383539313037666439643638 -35323235303164396361383832303362613235386530303538376265373964623363653365383961 -61306365346532336164303763396364363038646563323562653865303461353339366334376534 -61346230343662336633316133666631613136316537393464666635643336393566613730333837 -62643232623831333730313737666536353137613138333566323934353236616464656238303339 -63346665326661653832646466636165366338376638353431346130323230646534303137386634 -66646362663831623234333335376561336664343232323361326430313138633266343065366535 -39396463636330663738333033616334623366643636623735396435643865396630656234333433 -63346437663331653233313430666633366264343565373838656536323231633966663262346434 -35386631333063666233303364346139336461333736366438396239353136373339636638323534 -61663564353733363131363139663761646561313337656232336164653234646563666266656362 -36613666303165383530386530646331346661323633393232376533386661376533663166346337 -63316431393964333731373332313732643765386662656437373136663532663531383436393761 -36323939643464343035313633313436373038646234343366356264386131623462343161313261 -62383839343863386365366432396666303065616137396631373330643339656433363463313131 -30653431393630653331633634363233323632306434626566393662323566613565633639313938 -65626434353530306630636430646261373365386538613232383663336339303838626533306164 -62386166363861613863393331343362303665613239626265316366386165633264636163343366 -30643965633239333162633433366132323162396564383530373839333331643632333930643035 -33313663623165613033353130616666643136303137356462613237653430366663653061653035 -64653961343163646530623863313663303231633764626264363366613162313535653631656365 -39633561643833313334363465626263656132623463633335333465306266373634386238666165 -37373134646438303134653734323665666531626533666663363534316238666232316438393939 -32633531313565636531646361356566663133316164623133653935313335323835616238336431 -61663664393239343837353039303334353035323861323533663161376361353933313032306136 -36623463363235303166616162373665616333346366336333333531623932376664383861616232 -38393263336364343566653735396561363836373064346131336366316231303539386531643835 -34363463636232653137656339643630336333653163383332663763663633393837343963336666 -61316236366436363064396464383138373639373864636261383533333139306237303030373037 -36333563653036393734393439636262303837356133636233663064623136356133613932656365 -62356630323134373063663664316133663761383035633065383838333364663732353866356230 -33333263316666323133323433653330646161386462303163303235343031343761633234306166 -63363138376531353130663431636637646262366436393161376633306332346465316333656432 -34333238306263666537626231376665376435623466643131363736656432353938653236656132 -37313633323863656463633030336566343265666534663737396530316234343433383931336462 -35306236626564373766393438653134353332616532373132336663626130666532373933323863 -39656463386136646131356462653933306532383232313334326334613738356339623566636661 -62373139326138326663666237663436656631343635373733626439643466326439663361313266 -61323966336532653263633530646363366439326663356664313065306166313863396439396564 -30303138326261303661333831306536333363663536323235356263623533346233386161396131 -64356265363964316231333933613463623163316538393830393539306533313035303765656437 -38623262333662613665396537323331393937366533303039386138643639383664663263643734 -65313531343636393232343230613639363466616235633162353430646364363934393530643263 -39633539333237653733363132303163613564623232313464376336306431666534653338323765 -35646564653736383962343664383666333936363864396334333463313637343336396434633766 -35396466306235323338393937643362373764613161333462303665646239636562306635663330 -3630 +38656563383931366464306463373265623631353331376532333134313463303635656162373437 +3735623062343764316262613966353338326535313161360a323831623936373966633535396632 +63303961636566646338373464336637323830663932653039306362653832306163313938396135 +3633663532373362390a333233366636666437626365373732633361653833363264623138386661 +35636665666230326164343339633834616134666531343839623134343163333864333834303862 +37313864303335326133343666333733363333663336356234633332323262616530356661316363 +62306338303966306632303531613161386135346439333137393062343938303134656436656539 +64326134363664353165616436353635633265643161383335393633383563656231336139346138 +38366238653236623664313662626566373631633136613864373032316539646332363035343865 +32666664363737373962396162313438303030613264366232373030316335366534666531646239 +62306630316432346131383738373764313263363039376435653062313136356531383534343831 +37336536313564323336613639386138303562666437376238376630623665373230653232396261 +34333433343361643065643032366433643137396231343331326362626434643365356131613766 +34613039306130653535653330623333333631653538643536616530373538386332346132363739 +65653135306663336163336263376332326230616238666365653663663462306238366466663366 +31633833333465653534613863643465323631336661393366356363386462623434636631623738 +30376265663739333336393664626539353531653637303464316562613436373739353039653430 +36353034643839323131353033643866303333306435653532656239653061656235393536306463 +31626162393462356238323737313933633465623566646161363964393164393762626531326138 +61343336373065313232343064646463346638613661393263326265366539623861623537643164 +32663365313161316464333137383337656662323137356635636637326163343562323965666166 +30396531326364656461613232653361313061623835663663643861306334386539626530386666 +63306331313466623163366332326564353639383961656362633435363065313830613764306237 +37303530643266646464653263306432313766633835343739636261623464366536346665636135 +31333638306237623535623439613636363937333430633831306162333466323137323932663432 +35616135363134306231323933303635336332376163303131656131376131393465353434353330 +38623339313964313730663263373963306337373938393863323062366534613131623764386638 +65616463376633663864653363643135623663316439346336376366346166303962633366636565 +65373335623962613131336364396133623738613364383639356533316138383333353464616639 +62316161656633666431356462313634383239363464613031313232393131643936333237616364 +36646530646239626662386231313239396238323139383037653337376134316535373464613235 +35346439646234623462623262336461326434346632373966626464313633343266616463643764 +37633830653230666631383134313738326265343738386631346261343439356163353262626161 +37343134663764396439633535616537386638633731643164383064333266643830356563383066 +34356161623361623366363435653739383438396363643338353736323139313661373031396132 +61373834303439663939376339633832313639643332383239313337666433623435613161306139 +62633162346338663335663661643938316539316139356165346461326433646366306134356338 +35613934623666386463356536303362626466663062643236346163323337616232323535396265 +36636165656366396136666533626433363562353537343430333361643635313335336435626265 +36663233383265636530313332336562623833373930626136316265393634653466333732666563 +36383037336566373566313661363062346431343534396533656135326534646161346639383336 +35306633303836346365383533316661613630326533393836353036636132636663653530656336 +35656364613130633761666632333633333137656637353362353337626266616238336266306636 +61393932343736306465666338626365613531386362376565383738366230333465663137343363 +64336661343563643136626362303937653632316230356531323231653063306538616633616265 +64306332396639323461653136356662376236363861643466633466666265646538653833373138 +38623130633037333666623862636165666333366335643765613834383533343436626638356536 +32333939326532396339666637386237303730363332643861316236613634353230353565316239 +30356135323237633963343461376233656333633636303662333365333735383930656538356238 +31373862613534386464333865653361363663373664663234346636316262356137613037353135 +37656534383164363131613834633062656437373066646336666533313033643130636361383038 +61336366343230303036336563663537623230643061383732623865383134366535353365346363 +33373331323831396462633031353665363033346536306334366237646363636633353732626166 +66363938663861616461613562646530656366333166303264363030306139343161313938613361 +66376263326131626132646530643439633634303539613863623965373432363863326130616536 +64626336633238613034353166353664363230393732366465346363303537336161623834333265 +61346365393964393365656565323331303265616630613766656239656438316539636632373038 +62313139386233633732643765383039646534326236333134613531313437346161343638376432 +66336432363739633231303138383338313164313930373866366431316638353761323531333930 +31353136326630363038626135383939393639306466383832373630623565383766663736363937 +38656132303738333863623262656131396366313663666461666231626364646433613137333062 +32336661373236666235306534616535323061663634383763386339353732646636306563323632 +34313433653137613930363631373666613330363761333639373630306537343633643561366231 +38363833363139666631333431646262656239316138373339326533623437383336353139646431 +31313833363733653534333133633636616636343635393039666233383661666363373263313631 +34363064353135383261633666316462356231616163373634383730386536613462346535383331 +61336632643363306435323866346665626464366163343735616130613335663235376132396663 +35363635663664626638663331366131646333623533613065323731643366656562353138363335 +34653863323337653331306232306133323666356464323932323562376632333439303537306534 +66346462363535353961383061323762343237393535626638633365396566616636616236396462 +38353433393631393338646338383830663331663538336465366161653333373736623063666437 +34633365383736653230316631636666386664643731323434613761613139666432636236653137 +30393666663261616332323466353161636632643534386131306632313236653063633531356564 +66353964393634336530396162393862653837376436623738663135303632383166613333336165 +62333434613432626264653865353035643563663435316233376465383961303534646339373632 +31343530353636343238623565653566663936633763323861383231366533303164316339653637 +36393364616365666335386539653439653735323162396131643931333866613862633437333438 +34663233356461383537343738343134626532323361653666633733353939333065363131316133 +63616632336130653962636236336264363565303839653531326264386331316264623038326266 +62336233303332333638613461366231343935323036366536383639373837616638616362643331 +35663331653133343661346635613036643065303637613964666237353166386666313439383936 +31663730356636333061326535316638366531656430353633383663336535663830323063356666 +62333762393865333132653161346361313832643163356265616133373663363836303662613935 +63623662313830623434303530303833663063336164363836643531323239653966366233373266 +35373037323132653833646432356435333834346138636133633339613165303362336265303065 +31343931386136323362376431336164343931343934643031336435303738393764613332393561 +31363333623165643363646638336435633539626635656536323030646130323734643134336562 +62353233366533363131313462656239366337343532336332323563343931376237306231656532 +63326439303930326563323238613538386435393735356438316536373064643465306362346135 +63646532343334323538343838653764303332386139313239376332656664316361343536663766 +61363065343862373165366563366363376362303862633037373763663132663631346562313161 +62636635303738393763373232643862386135386537333637636631626638363232323036316366 +38383235366333363264653163313937333566633066356663383466393535333535313939326266 +30626531386333626334366533613366353433396438313838346339396433666436663961643438 +62633336633066616163626232326334633265356532663535323264373166316133386466653131 +66623261333964623235386433346236353338363961323230356633613864303735386530303165 +30646265346566626562643330333233636164386432613436383034623337323531613735656133 +64653232343839653661303662336435653439663732303638323732333064646333313732356339 +31303934613938326365656161383262363633373265616466616562313237363763613265376664 +38636534646237616332653531323036313430643862393362383630653339343931376164373735 +39323136653730316462643964643939396662333462653934623136643839373864613335303463 +61643765656431363734616435343564393663346535636134383637653538646336666164336339 +31353230656364633936643338613563306639316464653864396230626263306338323639323238 +35396334613039343734303232363962653164636561633164303132366234376632653339376538 +37366638376666353733633132636131653462613732656364316530306333313431316164663839 +34646366393835353338373763333666646535363634633864306433316135376663343833613435 +37356636346234623735663361336366363264646535336566363933653832613039663161663762 +65663234666437363834646163643733356136396464633830336533366337303438313665336166 +33376331373032316464353037383433623334643965656434636134333166323238666438653337 +32306234636661656639306464653238326536376661363036626561386630326239616137376565 +32666438356161336437646562373534326632306165666439666130313036373262633164376362 +66646639663663323134626636313130343362343966653564306265623630326664336534333037 +65626562666337653436373130613230656535386264303132373634353861376238386435366636 +38316637613036333836356536306132626631326562363535613835313432353133373138613264 +62396261303239396364613039346135623437336565663034313264643535366336653363336330 +31623539396435643132383363353861323064336430653936623438303933356562613335643861 +39663437333836383331663830626433386431383338373261353266386531373931383632306135 +36316339366539333730656638386635333733363764613364303563323665396336613930346233 +34333433323030643262373636663838343131303637613237376463386662303235623338393063 +66313166343730383961363765663463303266383938653638393830613662643362386361626132 +66306533313461663131613565343534303735366232383164333837363330626534643363626164 +33326134376263333530663631383930623066313035373135616665323564323033323639623235 +35383538356434663035306137366437613537373531316436336366646165346330383538386331 +64666232343961343261393339356531363636303165393532313632613363316665633233316330 +35623939613463333164643736313531666531626232386361663435653938343536363236616431 +61343430363861626339336166316135663234366132313762616230393239663931656536306663 +31343366663063396139336237373361333030353333623064336262313465316538623438313265 +62656232613435643437373233396566623537353038316262643239653432373863626530323334 +35633132353463356634633237386134346532353930343339303337643637373932666665313561 +39346531383663663131623863383832346237356231386263656566653631363836323132333135 +37316135613730386362666564643037653366373464323431303332306364373432646338333031 +31613533633664343038376533306235393030646236346232386236653138636234666561646263 +31386633343265663335323236346435653261363763366234346263626663643964393962343766 +63656237326563653633383461333364396165626138343461353963656131663265313835326465 +65643164316537333531363139623732376364346363613363663066663733616665346233633662 +35343065316437373136656235323466343732613335393761313061353432623664363462383562 +38333139636336366462326565383631633864633165613332343833396237336561346164656664 +63316463323233383434663061633564666264313261353632316164643035306234393732333435 +66623663316334373431633734323232313436366364623037303965366663386237343439613433 +65383763393061633431303736656238396364343765363463333433313836386666613964623634 +30643065626437653966393237326433373936653861306536363834623363613830343036363766 +64386234653331363564373065343139623965646562373933333162386264353832303164336362 +33613036666139396533333862316366396464646465346263623730313336663935356139663762 +65623466323239383732636165333564356534333939316539663366363065623631303933663863 +31393637643761303731646638323133636464623366363032656335643435373064666463366365 +65653831666262616165306537623534653763636633633137653733323234346266363630306434 +64616633326266636134343536363365393033663934663965323537633136356535353436316634 +61643636303237363033313637396533303761303536373836343332636466396339353065386539 +38366530623332316439393633356235386665636364313739643432623032303363613666656264 +31353063633862366638363838663932386131366434646462313130366633643430373230313338 +33383362313230373137393934323561343064623065353063653535366266373237656530303335 +38616263613232353632333361383030626133323261323033396638666231396638353537383338 +33393466663938356363623438366265336330313039313666343462356331626265383565623364 +66323835303365366333623733343664626431343663623465326235363430356661616166346630 +30336266383830336461663130616363663637653964366634663930393066633134626530653766 +34393439363332383831383865393464656165643331333661643138663133643133626236333363 +34643438353037346334316233653034313737616134653762353239636234383233336338633031 +32386332656530313361393532623661393538633463393861636539393666626362653031636566 +34616265353736633539313032346334313339313934323962313463376664373336666236346633 +31306563633966653134343937323839316137356164373338343965643934636631646666386361 +62363165663838626538363966396161363366303162356664313437613034636531356638333237 +35613833366462653834363261653365356336643635363831383930323566623262353865653233 +63333138633466653165666236356430666532353639326261613462636161636634613235353165 +38316566306533373138336334643465616339383739653639303632303661656135313039303632 +66613966633036653461396330663630396663623732346261653265343233393730306537363032 +66653737306431343435316433646164303338353865653336303731636663303863363662333863 +33323334656631326137646563616236393035336539646563643064663564383736343961656435 +36306136353037616634383866303636383531616136633230346538336563656366393466613262 +32663266323333393764616561316636356664346265353433653262313239326264306434383030 +34356263323331313364653966623630326433343863643839313165313063626261646339323837 +64613838353334613661346662346636313432393837386139386136613366353038343962366639 +32626265313764613733313334353432346434323439326637383462313863383165303963353665 +34313635623766313361633030616439343433353735326433383563656435393563 diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml index c597ce2..6bec8f7 100644 --- a/ansible/infra_secrets.yml +++ b/ansible/infra_secrets.yml @@ -1,162 +1,174 @@ $ANSIBLE_VAULT;1.1;AES256 -34313539653264336537326165616161313931303664633032373566626264643439393136386230 -6330353962336634623963336535333130626662336561650a303864333966333336376661356139 -35303531343330356331356166666666343739636564353039313637663066356135623630396236 -6434663137386232300a323334313332373630366133393530313438623733666333623437666539 -65376166643436303661343763383166616662653137636539623435653632623433396465616363 -33336431363266653138366564343862626634373838316439313339313961323437656435623635 -37653534376262656130373734353335663764386633646436396335616437633636376462343861 -38343430363733316166353032343164313333663865356335393634656335373636346565653233 -63386535366533303736623463623830363166646530666631643730303265396163353538656662 -34363964356665666464336636373535613639343535633361383064623065643230303066303661 -35333666353331363031353732643233613439666539373539343437336362656132373634663636 -30383039303266663961666662613730646664386266356338366335633663666162343338363263 -63323531613735363333613637666530323830346135653839353733306561396239373734313530 -61616233313264323264396230393163303766666665656330636636306661303838333265383535 -63383330363331663038643735653266333233613666393161636634396365663832653736306437 -39636230336336613863623562376134616234323434653061656534383132353038333966663631 -36666562623234303130373661636136336538386163353230383830303438336435346432623761 -36636263306564323039356165623835363730643463343065366436323235616639376363363037 -38623165623635363239366361616337663932653734383162376363396533373563643439633632 -36306564336165363035386663323366343636383366393366373732393634356236626535356333 -38383633316362396133346631303234363366666664343137633765353231306534653837323035 -65333762383164663433356438333264323631373561353762646265386630616536633833306565 -32336635623036666131613634333332393037643063373561303938313762666262353564303039 -31333731396661373865313862633732313864306131303936333638323965613831323034303130 -39393864663136383430626137323736313132303030326131616463653635383262353731653034 -65643437323434626637313333666538613039633635333231386433363963653432376164383364 -34376437386166363335373964666631393230383038333137636439643936393866356537393535 -35306135313532363462653734633162363236356436386161656164666261393236393162353865 -38613964613434636334663361396362396163623533623166306261366130633962653131386365 -65626464663736326365376166633538303439326561643362633163626330643265393630303163 -62383835623639663838663934663030383363646339663632666364613830373263313437303964 -37343664356135663361323539663562323265373163303035633463326162363462636131363436 -31343764313537666661656631643263353239616563636361633561313966663932616163313932 -66363837656538373337653737653061356237383436643964653230333263313063333033663866 -61613634363463363232663765646438653266386166623431363932383931396339633665633430 -62306437633438393136643930653939623734643162353961313831613337646435656436353436 -30396265643732636337303233623931643361663032383335333431623063393731633764636166 -38663465366537636539376661643237623936333565383666333163363066366532633965346463 -38653265373262663039353433663134373361663535336633383566303462393663623432653836 -64323838343432343061323338353633613930613463383465633465313436343932366163313361 -32356431353434353434306462306134643332326363353462666561663035356530376263653831 -30616533323666363161653231663835663532336236663034316364333438336130383537346633 -65656565643363303036396439373963623338626461366234623233626231323534616130623537 -66323666346439333763646635643635663538633361356637646565353234316264306339373765 -37356236323763383761323637306163613933623661353262643362636330393362646236386563 -38323236666633306336393839396439303536643933393331333363626533373636356339623235 -66643436343736383136656630356139653537356632343838636364306461326539373932383430 -66343736333261316363316230623336313366303963366134666533626363353937373434383965 -65363737626534363030386461333261363736373230323437303636353030373335366266643566 -34313863363864306331306130626135623265643935376130303566636432646532356438613635 -36353766373434323638626330313030333737346262653833643735306432623836323834663638 -34646130623363613763346163363435643561336332663865623563343666616338616134336361 -31313731646434666263303136613338613637613530303932646561616138393032343838623766 -35633163366463346663656331386539376263646264343466366466663664643661383238326138 -38646231616133373038343636383733643661366131646335623564623663326537373039353535 -36643062623063663131363765666265303231333332333535316634336262306437633236323961 -66653263613836373235363166326337393762346534636330633931356265613765363336386162 -61393630636136313630613363383138333939313231396332353132313139323362386162613364 -63633261313965323966396635623334333633353963663263666433616632613463353231386238 -61333439613839353431613637633465616338363334333962666464396563356266323538373761 -64343436636138663934306130336663633665356465303064363637316332656432666639633330 -39383664616536383265303265663437303438393833663066343735616261653736346230663663 -34663563653862316335326661663264333138613430643934303334343838623934356431313264 -61393166616239613131636131643762313563623137633966326362326664346438366564306130 -30663161306462383836646436316138373431636339633438343439633633663735336433356232 -36373437323031303637636231646262343133646632633561393339303835326336656533613562 -36616530623530386238623262393537626632376637663232336138353232396333383837303362 -36623538333432396439306261356663383163653665393530313434616137626230356463333434 -30383861363164343866326264366135346166633936313834636162303764613735383565663264 -30616264643535353730396435633863646634616332303337363230366666373563656534383661 -34613835346565393730353165663264656661623930623361383465366336383933626162393530 -39353535613237313064613537383564323164323337313039393637396534383563373637373730 -66653539303963623034333862613539613663323238333765646138656464386535336334626437 -35626465623864376539333964316533326235626265346438333737633738393532323162306430 -66353633386337626462643466343333353130356663363766353161333662323830643261376139 -32616465326135373037636265306436653630646164323365636166356362353634343036623035 -35303136373235313431613932666230353034396334336561333732353762656434633037666664 -31306438386635613038623662336137336430653434383730653161383239376466326232323163 -66653738363030626231383930303335333663623837366464633764333461636461303131393966 -32303035376661623738323032346638613261313963626130386464383032303334383735396564 -63326633376139316436393236643066656631396134393962366363346233666232393434626134 -31653933383631613064303563656665383834623662306137303861623063663437636335326430 -62306630393938323832613830353230313136383834333832343366646632366634393632383539 -61373531653862623261633766613531346263343030373733383465353233316162656566386264 -35313834333739393064303730306262663032376335663032613233343131336662623065346231 -37653164633632396635363434623936346135616531353536323961336662646334666335306137 -38343037303337363966663934663733303364323134613633313566346161376661383032393832 -35363531316431303937323761323234666436313661623833323235373539393134346261666533 -32396665383961316166633836656562623665656134353762303238373633376333356664666539 -33356430613663366162316663623735646434656261613835363835343935323765626335326139 -34636439313466363438343862313636303163376635383361633338363664623830613837616464 -34306636346438313437613430306466383166373431353435656166643638626236373533393939 -62366538353262383239396236653664613664626236326531386536643433313361393065316562 -37303435333231313434653939313264613031343266616537383464356334333137646132396536 -30623238346662323163656365323933373366323530306530613032363861643037393332626531 -30353838393263306137633564376430323038313162626633623238353136653032653936303863 -33373135656434316466663363616138623734313063666632393265343532333433303731373239 -30393635333465313735363030386634323635393335383532326138346538633264386163623463 -34396336366638306633633732363162633336353962303261346439396231366337653633653161 -34383732333530366136366334396630303533336165653333326465306439313233373335326237 -30663331643839663534666338616466323633636364363663353462613632353764353931663164 -39306462623434653035396534663232326533643032343139663237366235653265393164373735 -30626435626235343730366265633935663463663435633137646138393765323738633162623239 -65666666353339656636633765313164306339383135623633643763396530643061383764373364 -38616233613562353831343838343830323130646662363061316534346564336631343639623161 -38666338346330643537366631313165366632643932323131613836303362383733633165653762 -64383539303139386337323931623531373461343737303236396534346566363338306138393334 -62363961386234326539663433646530663233306365613662633737376237393838306361343365 -31623538353362663830646436376236333161656238303332303339353531613461313839643239 -65613230373035336434363632633735643063646363656133326630336461313035633662613763 -65333665323764306535656137343034656165366530386166353964636338653532613332633437 -64323665333335386661376437303539336534393865353262623366323762326435313262626234 -37323433333138363837613034616662383362636462616134393431303737306166646233346362 -33633536636537633230353039623562636635396138366131366337613131323334356139346433 -62393564656635663836383164343562303637303533376665383638343238653939663962303535 -61383232316264323434366530623635623637623062333235646436646531643261633264393461 -36333766356434633532363536353734383866656461613338633234363966313661333537613439 -37343236396535666661656634326439633539303339616563643465613732343933393131623038 -63353231633633356563373438396335636439326136363463366663646565313532303230313836 -33613734323538316534386531616233646431623435326338366339386431376363656439326535 -39373065386463353435396466323433343436306639303266323431653037336463316532376533 -39656365303166306430646664346539376366343332633837303030376561623966393732393466 -65353563346234633965623036636264386239336537656237313433633932323063636239616561 -34323261373938333462623862393063333963306333616530656232646461363132323932663231 -65656361326139323030326332613166303461326236383432323761663061353134633736353665 -39643932363538376163613063316533346530646139626536356663633464636235353035653362 -61313838363065623365666538643562356336373539343237366161316130373461306362353133 -37343965666436643633626461613738303738313838326265373164373464643037363861643730 -30326431393865373362363836333631393039313938343834366237653961353163326665306262 -34643836653434313838303035623961613036656630666266366330313031386338326563663833 -64323535663438313865306539343936393864376263393532336538316530663839643161346233 -65383831316566373734636462396161323630376434666430653235373238303932616662666437 -31373532353339366534316463623237313835366164663630343561623861383263626365663566 -66383034613939363736356262336336356363313332363466653534653336353964313831363139 -34626436623231306434393464376639303462356132636630373762366135613935363033623137 -39643036366263353139623539666130333163616562353239353731623534346635653239396636 -39363636346234356465383365356462333430366265306464666666393033656638393930626661 -36616365393637316231656239336135333664346139373163303231363266393562626364346561 -36333738373738643136653465353331316561646162383230666133636135646330636363653061 -35623234303233326531356537616131356533386334623938303462613337323930623566343664 -34323831616636393264643765653262316434623635373733623139396131636366353033653263 -66393361653766623161343935626234336163363765313339653266343837313731653839643066 -36633861663130346635343863646166393238333165316365333230376433333439613731333461 -37353035323931306462646465323066333463333236353665303664393461353030343963346132 -32396237626335313831616662383165333439393739353865623631333630363432633366646263 -32633435366461656535323230613739323634386536666132633935313532663763353963343065 -61616134333630633333616639333062663636326237366265326264613863353537323232373165 -32663964313736313031376662663631663635393731333264613566336661623030616235333163 -31343865333735633933373933303030393632656435333032373730663866343736626663623136 -62643566356665356234306134653037383638303064316138313638653336616364393934623862 -38613864313962633335653165363437383631366531613731633339313562366364306164386430 -34326133643436616537336132306634613461353963306664313531633564653134383330336537 -64393331346566313966626238663733646565343761353563623461663662336331656634323964 -34323965326638363038316233623431656233346165373136343532316635326561356265336232 -63396563383765333736356638363261396634303730383136663566383135303331373534313766 -66363335356265396164326438393862613333663936666230316133396563376535663365633061 -35323665373233656631633038343631356233633934666161633766306331373537646231306437 -32623965656139373835333238343565643635306437656262623334316633646361623262386435 -32663031323338313339386663656363316164353666346237316137313562623838326436383862 -35616663633433613064613264363637396632333734326231343830633537336364346336316461 -3833 +30383238343938316437303961393961633538303835356661366465306161353738616335343563 +6135613063376562303539613535633463616433383239380a323163346131633838393836383539 +38623766303739623131613935363037323462353664386563356661393665306136323863656363 +3433313638363731630a336165343563653930663435333037653330386364336530373830346336 +34396361653637626137343033343432303438383761393765313164393631666230616537366636 +33313234636535306535336132373830383438636235636466623861363734636238393234316137 +36643439643130363462643765353061623436376463386139616435323136376366653932323766 +34393161353062386666626633393639623434653864383439343339363665636134636139326538 +36323966373230363938623434336161396235316532663739353837623238343862633032613264 +32626565313637306131393466653739383935646134346330333835393932396634323565313861 +38613864656139646261343835316562323863363836376266653139336638366161323035373265 +39366566333136646134353065653065333337303832363237653361353234386335643165376331 +35316339396130366163313130633561643062316164376262353131323237303861666263643536 +32333062316330313532343832336434653664656565616634643436656466373533376535396533 +35626636393135626539633365326364353462633462656364316361303732643632356230643333 +30396233616633326438393836383237623636383464333830343464343837383231653139386336 +65313337366665343936366430623162643032626430336466643230313530343832383730303263 +65386235633832643931656562303561666131646362303533323430336235643436653866366161 +64636365633731613365663833663264343233313037326537616164656231656332663461653635 +31323237656361666630323930336630626437393836313431353839636435363466376535633962 +63323462626662616139653665346132343736636361643930333366666237303438356364656464 +36663363386131646361316535653463366234663062323539646562633962653537383465366439 +36343635313934353063373734386137653239343538663037646362376434383234363361623165 +66663734313133343062363137323931643338326365346234313835303431613632623461366339 +64656138303166356136313962646133306536333465316239383532353362333932326639393363 +63393430373133316532633562386432663332623934376636306438623061353238333562313737 +63393338333534646265643462666163356365653937643634316266303730346333346465313662 +33326638643639326361386461663931613766373937656266313964336461633939333535646231 +35316331333537303830653336373530353364616161346232323130336137373737616534373435 +35383466666262333834323332393737613763303562643232616432316636386635383934373633 +34316462393064656633373562333039646135373332373063653539313230646466326233313536 +38346238623162353338656437336536663534643062376138643030363935336461333638363134 +30653738383233636531333165386131383061363466313736336632303136616261623137313065 +30323766393862663965623337303638366239663130643638383765383930353533323635316232 +34646339366162653663323964363863623962643166646130633236353563313061353434376334 +66313862313265653238633034616530343164303432666539363733666662653733666462303662 +65663638653534643132353361313130323763636338303836346430623664303732356636636232 +62396135343236646537386334373166303663353661323236336365336533613139373430333364 +65363863323335353663396366616530633836363837653963656636333264323434323365396135 +62613862386361616633333636663263376465613465633362333733326665326639663961313631 +32653735353563366236363936343438623665363834633436393232653737623436303237376530 +34353032633433363865303866643761343130303431326236353332356464396433323430613830 +38373035306137363562376539633863343737616339346463636236393537346163323731333733 +63656161306437343862643836633539326661653062383361613739366237623162653939366335 +30613466643665313830353332393365623933376638326534336664616561363461643237386533 +64376532306663373934356361663832323337393266356138343236663964666637343431663237 +62303531353266633164336363356336616332663730663933353730356365393866313662363730 +63346434663162313333663738393461386638343864616634323938323437363232356466303637 +32663630633466623839306537366465653164616631373564366132636165346231303065653036 +66333134336132376232393639616163393366343365336661633330316436656163396434343963 +30356331393432613561636562343138643062613738613764666232656639393532623433343138 +37623963303237663337626661656638636364323931643730343535303737656461663831336336 +65653031656633386562663837386130613635306438313764653830663966663232646663356132 +61613338663331666265393566613661653462323530623930373034363363363238663465303162 +39303465613435343934636365336262376431616136383538353862323235643835373638663832 +38303336306665616666353963653532313162336536336137613337653363323138626338333861 +36663364656463623337663362643936393664363937306135353464363061393837346161333435 +63343030333830626438653565623065383264616264303035613432313266643739333063653066 +39623863373632363337663065653865343433393333386633626237393162383338343038646165 +30396331643033386331333039663066356432613733336235353266356336326138323864396132 +38313164393833343339663363366438313931393232613935343063306233323061363066616366 +62303564643930643262666364346130363530363630306331393032623736643630376563636533 +62656338343734343739383565396339323138666265626235376235613963666566333862656533 +63626566343737636532363535656238316332363831303862313265383936613339616430376231 +32663363613135626463373166623063336135303437313031353337323939633732396139623731 +36653061313233363837396636643234666439663261316361346263343934393237626134656230 +66326135396661383432323939666662376163376233323339303939336363336531653163633464 +33346266346533346631393666316239343463366636613336313833663730323734613135343930 +62613130613933656237366262363433353434303763326564333965383662336661313037616430 +32643061313638373931646464383465626234323866386336313861336531383638363736633438 +38643662376364333564383963613339623964396666313263653336393666623932393738656665 +63653234316265316535303531646230326530616639653036653731373934316238313563313338 +62373162346333666634346362333232373234613430386237656337353661643839663465663265 +36303738636137633636646537613336386230316334613739383362656661643163373936663130 +37393361323837323032376464323961306335623835616134633735626164306539383936613064 +36653133666230336262616438343539346561363863653364343662356233633239623765383165 +30336635393763646432396437613838383733626565353737336165626333643339376432393366 +31386334326130346164633534366532363037316334373133326362303965346633633939333539 +35613435626232653733303266666431666133366430633538623931623231353235383364303765 +64376133653562336631326630363036623033303039363561643236653531663861336262393865 +37346633626334313736303831633236313562393363313133613836326137303435356435613337 +31613335336562376435393237326439613863383264316537353165373437333866666364316366 +36326332336335396430663332326562613235623338633930633437663162643433663065323234 +36613663356534366433383431386666643736323936383030396139326464623961666666323462 +35613335633666313330636161613836343861313861666530323935343961316631666566653565 +61363033306338323966376631343232343038393564616262323532626131643265303865396539 +62363433353666626666393331636537363535306534383634333262326438366463363138666230 +32386366303462323930343863333463383638653532386362323338306266393337326535363663 +36346366393631396163343930626537643161353038383934313331363834323431653364303939 +37326532643864343338626430363065353833353862353262336235336637663865653263663339 +62363366663838366236653364363064356137346463383933626533343461356162363062653932 +37663966373031363133326534623763643030356335386139363565613135353835343233636330 +66393532396530643532626336303737306134323537373135663435376632343331323663333737 +66316638396563626663313636386566363036383663656230336530376234376138303462323030 +61613931336364383265366234393933653432313562666238666639326430376238623365373838 +38363339616365623065376438323738306633666561326432303935393938373638613736303332 +35353833663431313462333362313732353534383630343030393564626139353139623530633332 +34653639386263313738613430323664383264646632633062643031303437636262623966313630 +30663533623334633164353964643536623566623132666166373039303361346534346337396430 +66633632633162316462636662363938656263663135326237303361386234623661333938616262 +62343731623838646133333135306438623538626639663863313337313862626535303937346138 +64623338336333323239643565336533323234353436616230373063663539613839646238643735 +32613163343637656337653139346463383734363636633635316261646133343731653133613238 +35613265633538323039663665353664333434373834393264653830343766343136653632363532 +65356331633262643635343861393535353532383137346534633661613963333531626463346562 +63323638376536626534353062646561373566396235386634646663616230633635386661333334 +34643833653065633033333839633736656161396236363364393838613038326532353130353063 +62373635356665643133353561633861643436613035636239373630373164386531343736643564 +32656536633631353931666137663234343861633865303436323434303636376634366665313234 +33336163396330383263346333323065616437383564303839636235313964343633626166623833 +62316436326532386338356563643038663737616338363636383032623665373862643631643331 +62336364366537366531663762393964633531333166303662356334383562383466303334663261 +63663638633736353934663530343934613837623963366232323637626466343138613165626634 +38663338386465353463633438356434353839303334343165386430623733393934343466333134 +35373137653635653966383865333637623930666562323236323434653063613464353362396164 +31353632353637326330316239393665653833393438663436376562326638336239396337633134 +39343031323931303061663837353931396263333637633264643966343633363966626135356333 +30643564666336636535663264396632636532616234316130313761333731626639333665393430 +37653765336533346564306139333436623466626664393735363865663733306161393634323161 +38653262303363313562363233643634623364396237316636623431656366393538633131366139 +64383833313537653137363961306538353230656164626663613462323962363631643336356561 +61343539393239623861373633386338663534616438333538613632356265376132356539653964 +61643234373639643933653963323631336433633863326263303938643166356330396339626637 +30353337626231333730386465326533346233626561373564376234343763616431313031656531 +65396363396436383337373133333231393139303734643533623233613136363031616430653530 +31383231333763646437646430346564623739376365656661363539386164346365633730623738 +65373334613138336263623262653833363566633638646139656433326165353637363131316230 +31653436353336636333656430656564356233626535333263616136626136363430353237366230 +37346533316562653764396262646661623232346335633966303833356332346133653535356638 +61386338363139343561636564396232643361333936656637363130323165666536613066663137 +64356233376662666432366364626639373534663865343865323363333239636461326665306537 +35666533643834373839326139623564316533303562306235386238313764666438656139393362 +61353337303761666538636533613839336237646331663231353831303663323739623362326363 +66646136383765356431306639383336623237346534623535613435346238353866316164333763 +62646665396233646632633535633332313934653964396562656464376635306434613932623664 +66303131616137616439353939663731306438383662393039333036663862336537356335306266 +66646538316432346132633764626361353561663937323039383138636331326232643932633932 +34353061396461646664633364303536353833646130303032656238313637363736643663356237 +34356161313463626361393031643332656464326166636333616162383431396662383261383033 +31303537326532323238363539613439626365613538643364646366326236353934376539383865 +32373731323133383335343639626133353232613033306331373237646364643830613532306237 +37663034613363363164363935633332633732653761623439366636646137326163306632316264 +35356164343462633938333331323836313539303961316532613032396632666138646265636337 +38313539633430646166313435376532323630343232393735346335363731363666663363663138 +38373465396461306665303331383266663836666434626137636632653266393239656631353833 +32626531363763303439356565666434333334383737316431353032333638366338383430313436 +62326331353834303261663730303066623062336638323433353565656638336235663734323530 +66396666306534393139656138666430373537653862343232386233313730313831643734353539 +35636361643230316665306563633836656261353338663538336666383462343132666361356233 +32623732646564316461373537373331363138653332393638656437643635376663303430373933 +31383837643033323334326562663135613537373661396166656130613963383265303961323230 +63376662656232393833373731653234316462333834356534303738323430336633646437323332 +34376263396662373931333938623330626438663336323066326239636535323562623936636335 +39373335363663643238316230356534643033393363383436613865336361653566626336666630 +37313739646337653436663665386666393063316263653833633239396632313734616262616166 +38303363646661373139343039353532363736336261356462366238343135313637636166643431 +38633236386535346166386434616234326164346661386534663834646464303038313561613034 +33656265653739383932323632613766333335363633393666396663306337333231316639633731 +35383062613633363065616535666638326138306431366130336238343263356362333535346435 +35643866386430333730663461353136313630323961633764616134333835643234656633303232 +35363333656436653563353336393066306139633937623037643566633166646264313864646364 +36636663613465656339613534613634633733616662346530356361383934633466616138383865 +66313930613030386163346363333836633139376637323337396434613761636434666164393534 +37346437356431313930333932303465326138393262643665643139646134386161613465666363 +32376464653737363130303761303838663061626636643534326235333062643735653837633938 +35336161306435323665653035313930646666366631353464626235313761386135303665636137 +62646466313662613630616234373933376138383037363933626437306466633832343238633964 +30346265653865623537323833393763343632323839353237363466336663396131333266396533 +65626561646233626366353862396564373463623563343961646662633133613333646435303030 +31323332626534383732643839353739653835333934373835653331663762376337393930363833 +36313033633239383830363861623538396463396163333138613437303535396265336664313639 +31653837623938393137643137306566386666653564373838373265306462303062623436653261 +31333138396664613962303965613536373239643166353831623565643662323564 diff --git a/ansible/roles/mempool/README.md b/ansible/roles/mempool/README.md new file mode 100644 index 0000000..d8282cc --- /dev/null +++ b/ansible/roles/mempool/README.md @@ -0,0 +1,69 @@ +# `mempool` + +Deploys the [Mempool](https://mempool.space) block explorer as a three-container +Docker Compose stack — MariaDB, backend, frontend — on `mempool-box`, and keeps a +health check on each. + +Converted from `deploy_mempool_playbook.yml` (745 lines) under Plan 6. The +playbook is now 37 lines: this role, plus a second play that publishes the +frontend through Caddy on the edge host. + +## Phases + +| | | +|---|---| +| `docker.yml` | Docker engine: repo, key, packages, service | +| `deploy.yml` | directories, `docker-compose.yml`, pull, up, wait-for-healthy | +| `healthcheck.yml` | three check scripts, three services, three timers | + +## Three health checks, not one + +Mempool is three moving parts and knowing *which* one is down is the point, so +each gets its own check, unit and timer, driven by the `mempool_healthchecks` +list: + +| | checks | +|---|---| +| `mariadb` | `docker inspect` health status of `mempool-db` | +| `backend` | `GET /api/v1/backend-info` | +| `frontend` | `GET /` | + +Each records its answer in its exit code, which systemd keeps: +`systemctl is-failed mempool-backend-healthcheck.service`. Reporting elsewhere +is one field per check, `push_url`, and is the plug-in point for whatever +monitoring exists. Empty means check, exit honestly, report nowhere. The URLs +are credentials, so callers pass them from the vault. + +Nothing here is specific to a monitoring product. The embedded Python that +created monitors over the Uptime Kuma API, the `/tmp` credentials file, the +push-URL file read back and parsed, and three systemd `Environment=` rewrites +are gone. + +## MariaDB owns its own data directory + +`{{ mempool_mysql_dir }}` is bind-mounted into the container, which runs as uid +**999** and must create files there. The playbook this replaced declared +`owner: "{{ ansible_user }}"` (1000) on it, which had drifted from reality ever +since the containers were created — unnoticed, because the playbook had not been +run since. + +That was not academic. The first real run of this role pulled a newer +`mariadb:10.11` and recreated `mempool-db`; had the chown still been in place, +MariaDB would have come back to a directory it could not write. The role now +ensures the directory exists and leaves ownership to the container. + +## `mempool_frontend_port` lives in `services_config.yml` + +Two hosts need it: this role deploys the frontend on `mempool-box`, and the Caddy +play proxies to it from the edge host. A role default is invisible to the second +play, so the value lives in `service_settings.mempool.frontend_port` and the role +default derives from it. + +## Expect `changed=2` on a converged host + +`Pull Mempool images` and `Deploy Mempool containers with docker compose` are +bare `command:` tasks with no `changed_when`, so they always report changed. +That is the idempotent floor, not drift. Everything else reports `ok`. + +**`mariadb:10.11` is a moving tag**, so a run can pull a newer patch release and +recreate the database container. Pin it if that is not what you want. diff --git a/ansible/roles/mempool/defaults/main.yml b/ansible/roles/mempool/defaults/main.yml new file mode 100644 index 0000000..d0e36b2 --- /dev/null +++ b/ansible/roles/mempool/defaults/main.yml @@ -0,0 +1,55 @@ +# Mempool Configuration Variables + +# Version - Pinned to specific release +mempool_version: "v3.2.1" + +# Directories +mempool_dir: /opt/mempool +mempool_data_dir: "{{ mempool_dir }}/data" +mempool_mysql_dir: "{{ mempool_dir }}/mysql" + +# Network - Bitcoin Core/Knots connection (via Tailnet Magic DNS) +bitcoin_host: "knots-box" +bitcoin_rpc_port: 8332 +# Note: bitcoin_rpc_user and bitcoin_rpc_password are loaded from infra_secrets.yml + +# Network - Fulcrum Electrum server (via Tailnet Magic DNS) +fulcrum_host: "fulcrum-box" +fulcrum_port: 50001 +fulcrum_tls: "false" + +# Mempool network mode +mempool_network: "mainnet" + +# Container ports (internal) +# Sourced from services_config.yml: the Caddy play on the edge host needs this +# too, and a role default is not visible outside this role. +mempool_frontend_port: "{{ service_settings.mempool.frontend_port }}" +mempool_backend_port: 8999 + +# MariaDB settings +mariadb_database: "mempool" +mariadb_user: "mempool" +# Note: mariadb_mempool_password is loaded from infra_secrets.yml + + + +# --- Health checks ---------------------------------------------------------- +# Three independent checks, because Mempool is three moving parts and knowing +# WHICH one is down is the whole point. Each records its answer in its exit +# code, which systemd keeps: +# systemctl is-failed mempool-backend-healthcheck.service +# +# push_url is where to report, and is the single plug-in point for whatever +# monitoring exists. Empty means check, exit honestly, report nowhere. +# The URLs are credentials, so callers pass them from the vault. +mempool_healthchecks: + - name: mariadb + label: MariaDB + push_url: "" + - name: backend + label: Backend + push_url: "" + - name: frontend + label: Frontend + push_url: "" diff --git a/ansible/roles/mempool/tasks/deploy.yml b/ansible/roles/mempool/tasks/deploy.yml new file mode 100644 index 0000000..d98df7e --- /dev/null +++ b/ansible/roles/mempool/tasks/deploy.yml @@ -0,0 +1,75 @@ +--- +- name: Create mempool directories + file: + path: "{{ item }}" + state: directory + owner: "{{ ansible_user }}" + group: "{{ ansible_user }}" + mode: '0755' + loop: + - "{{ mempool_dir }}" + - "{{ mempool_data_dir }}" + +# MariaDB owns its own data directory. The container runs as uid 999 and has to +# create files in there; this playbook declared owner: {{ ansible_user }} (1000), +# which had been drifting from reality ever since the containers were created and +# would have broken MariaDB the first time it needed a new file. It went +# unnoticed only because the playbook had not been run since. +# +# So: ensure the directory exists, and let the container own it. On a fresh +# install the mariadb image's entrypoint sets ownership itself. +- name: Ensure the MariaDB data directory exists + file: + path: "{{ mempool_mysql_dir }}" + state: directory + +- name: Create docker-compose.yml for Mempool + ansible.builtin.template: + src: docker-compose.yml.j2 + dest: "{{ mempool_dir }}/docker-compose.yml" + owner: "{{ ansible_user }}" + group: "{{ ansible_user }}" + mode: '0644' +- name: Pull Mempool images + command: docker compose pull + args: + chdir: "{{ mempool_dir }}" + +- name: Deploy Mempool containers with docker compose + command: docker compose up -d + args: + chdir: "{{ mempool_dir }}" + +- name: Wait for MariaDB to be healthy + command: docker inspect --format='{{ '{{' }}.State.Health.Status{{ '}}' }}' mempool-db + register: mariadb_health + until: mariadb_health.stdout == 'healthy' + retries: 30 + delay: 10 + changed_when: false + +- name: Wait for Mempool backend to start + uri: + url: "http://localhost:{{ mempool_backend_port }}/api/v1/backend-info" + method: GET + status_code: 200 + timeout: 10 + register: backend_check + until: backend_check.status == 200 + retries: 30 + delay: 10 + ignore_errors: yes + +- name: Wait for Mempool frontend to be available + uri: + url: "http://localhost:{{ mempool_frontend_port }}" + method: GET + status_code: 200 + timeout: 10 + register: frontend_check + until: frontend_check.status == 200 + retries: 20 + delay: 5 + ignore_errors: yes + +# ═════════════════════════════════════════════════════════════════════════ diff --git a/ansible/roles/mempool/tasks/docker.yml b/ansible/roles/mempool/tasks/docker.yml new file mode 100644 index 0000000..b97e63e --- /dev/null +++ b/ansible/roles/mempool/tasks/docker.yml @@ -0,0 +1,72 @@ +--- +- name: Remove old Docker-related packages + apt: + name: + - docker.io + - docker-doc + - docker-compose + - podman-docker + - containerd + - runc + state: absent + purge: yes + autoremove: yes + +- name: Update apt cache + apt: + update_cache: yes + +- name: Install prerequisites + apt: + name: + - ca-certificates + - curl + state: present + +- name: Create directory for Docker GPG key + file: + path: /etc/apt/keyrings + state: directory + mode: '0755' + +- name: Download Docker GPG key + get_url: + url: https://download.docker.com/linux/debian/gpg + dest: /etc/apt/keyrings/docker.asc + mode: '0644' + +- name: Get Debian architecture + command: dpkg --print-architecture + register: deb_arch + changed_when: false + +- name: Add Docker repository + apt_repository: + repo: "deb [arch={{ deb_arch.stdout }} signed-by=/etc/apt/keyrings/docker.asc] https://download.docker.com/linux/debian {{ ansible_distribution_release }} stable" + filename: docker + state: present + update_cache: yes + +- name: Install Docker packages + apt: + name: + - docker-ce + - docker-ce-cli + - containerd.io + - docker-buildx-plugin + - docker-compose-plugin + state: present + update_cache: yes + +- name: Ensure Docker is started and enabled + systemd: + name: docker + enabled: yes + state: started + +- name: Add user to docker group + user: + name: "{{ ansible_user }}" + groups: docker + append: yes + diff --git a/ansible/roles/mempool/tasks/healthcheck.yml b/ansible/roles/mempool/tasks/healthcheck.yml new file mode 100644 index 0000000..fa53a9c --- /dev/null +++ b/ansible/roles/mempool/tasks/healthcheck.yml @@ -0,0 +1,58 @@ +--- +# Three checks, one per moving part. The Uptime Kuma specifics that used to +# follow — an embedded Python script creating monitors over the API, a /tmp +# credentials file, a push-URL file read back and parsed, and three systemd +# Environment= rewrites — are gone. Where each reports is now hc.push_url. +- name: Create Mempool health check scripts + ansible.builtin.template: + src: "healthcheck-{{ hc.name }}.sh.j2" + dest: "/usr/local/bin/mempool-{{ hc.name }}-healthcheck-push.sh" + owner: root + group: root + mode: '0755' + validate: "bash -n %s" + loop: "{{ mempool_healthchecks }}" + loop_control: + loop_var: hc + label: "{{ hc.name }}" + +- name: Create systemd services for health checks + ansible.builtin.template: + src: healthcheck.service.j2 + dest: "/etc/systemd/system/mempool-{{ hc.name }}-healthcheck.service" + owner: root + group: root + mode: '0644' + loop: "{{ mempool_healthchecks }}" + loop_control: + loop_var: hc + label: "{{ hc.name }}" + +- name: Create systemd timers for health checks + ansible.builtin.template: + src: healthcheck.timer.j2 + dest: "/etc/systemd/system/mempool-{{ hc.name }}-healthcheck.timer" + owner: root + group: root + mode: '0644' + loop: "{{ mempool_healthchecks }}" + loop_control: + loop_var: hc + label: "{{ hc.name }}" + +- name: Reload systemd daemon + systemd: + daemon_reload: yes + +# Ungated on purpose: enabling a timer is deployment, not monitoring. The +# deprecation banner swept this up with the push plumbing, so Ansible stopped +# managing three timers that are in fact running on the host. +- name: Enable and start health check timers + systemd: + name: "mempool-{{ hc.name }}-healthcheck.timer" + enabled: yes + state: started + loop: "{{ mempool_healthchecks }}" + loop_control: + loop_var: hc + label: "{{ hc.name }}" diff --git a/ansible/roles/mempool/tasks/main.yml b/ansible/roles/mempool/tasks/main.yml new file mode 100644 index 0000000..9feb652 --- /dev/null +++ b/ansible/roles/mempool/tasks/main.yml @@ -0,0 +1,6 @@ +--- +# import_tasks, not include_tasks: static imports stay visible to --list-tasks, +# which is how this conversion was verified against the playbook it replaced. +- ansible.builtin.import_tasks: docker.yml +- ansible.builtin.import_tasks: deploy.yml +- ansible.builtin.import_tasks: healthcheck.yml diff --git a/ansible/roles/mempool/templates/docker-compose.yml.j2 b/ansible/roles/mempool/templates/docker-compose.yml.j2 new file mode 100644 index 0000000..fda2ad1 --- /dev/null +++ b/ansible/roles/mempool/templates/docker-compose.yml.j2 @@ -0,0 +1,75 @@ +# All containers use host network for Tailscale MagicDNS resolution +services: + mariadb: + image: mariadb:10.11 + container_name: mempool-db + restart: unless-stopped + network_mode: host + environment: + MYSQL_DATABASE: "{{ mariadb_database }}" + MYSQL_USER: "{{ mariadb_user }}" + MYSQL_PASSWORD: "{{ mariadb_mempool_password }}" + MYSQL_ROOT_PASSWORD: "{{ mariadb_mempool_password }}" + volumes: + - {{ mempool_mysql_dir }}:/var/lib/mysql + healthcheck: + test: ["CMD", "healthcheck.sh", "--connect", "--innodb_initialized"] + interval: 10s + timeout: 5s + retries: 5 + start_period: 30s + + mempool-backend: + image: mempool/backend:{{ mempool_version }} + container_name: mempool-backend + restart: unless-stopped + network_mode: host + environment: + # Database (localhost since all containers share host network) + DATABASE_ENABLED: "true" + DATABASE_HOST: "127.0.0.1" + DATABASE_DATABASE: "{{ mariadb_database }}" + DATABASE_USERNAME: "{{ mariadb_user }}" + DATABASE_PASSWORD: "{{ mariadb_mempool_password }}" + # Bitcoin Core/Knots (via Tailnet MagicDNS) + CORE_RPC_HOST: "{{ bitcoin_host }}" + CORE_RPC_PORT: "{{ bitcoin_rpc_port }}" + CORE_RPC_USERNAME: "{{ bitcoin_rpc_user }}" + CORE_RPC_PASSWORD: "{{ bitcoin_rpc_password }}" + # Electrum (Fulcrum via Tailnet MagicDNS) + ELECTRUM_HOST: "{{ fulcrum_host }}" + ELECTRUM_PORT: "{{ fulcrum_port }}" + ELECTRUM_TLS_ENABLED: "{{ fulcrum_tls }}" + # Mempool settings + MEMPOOL_NETWORK: "{{ mempool_network }}" + MEMPOOL_BACKEND: "electrum" + MEMPOOL_CLEAR_PROTECTION_MINUTES: "20" + MEMPOOL_INDEXING_BLOCKS_AMOUNT: "52560" + volumes: + - {{ mempool_data_dir }}:/backend/cache + depends_on: + mariadb: + condition: service_healthy + healthcheck: + test: ["CMD", "curl", "-f", "http://localhost:8999/api/v1/backend-info"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 60s + + mempool-frontend: + image: mempool/frontend:{{ mempool_version }} + container_name: mempool-frontend + restart: unless-stopped + network_mode: host + environment: + FRONTEND_HTTP_PORT: "{{ mempool_frontend_port }}" + BACKEND_MAINNET_HTTP_HOST: "127.0.0.1" + depends_on: + - mempool-backend + healthcheck: + test: ["CMD", "curl", "-f", "http://localhost:{{ mempool_frontend_port }}"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 30s diff --git a/ansible/roles/mempool/templates/healthcheck-backend.sh.j2 b/ansible/roles/mempool/templates/healthcheck-backend.sh.j2 new file mode 100644 index 0000000..3a6630a --- /dev/null +++ b/ansible/roles/mempool/templates/healthcheck-backend.sh.j2 @@ -0,0 +1,18 @@ +#!/bin/bash +# Mempool backend health check — managed by Ansible (roles/mempool) +# The exit code is the answer; systemd keeps it. Reporting is optional. +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +BACKEND_PORT="{{ mempool_backend_port }}" + +check() { + curl -sf --max-time 5 "http://localhost:${BACKEND_PORT}/api/v1/backend-info" > /dev/null 2>&1 +} + +report() { + [ -n "$PUSH_URL" ] || return 0 + curl -s --max-time 10 --retry 2 -o /dev/null \ + "${PUSH_URL}?status=$1&msg=${2// /%20}&ping=" || true +} + +if check; then report up "OK"; exit 0 +else echo "Mempool backend not responding"; report down "Mempool backend not responding"; exit 1; fi diff --git a/ansible/roles/mempool/templates/healthcheck-frontend.sh.j2 b/ansible/roles/mempool/templates/healthcheck-frontend.sh.j2 new file mode 100644 index 0000000..b2541d3 --- /dev/null +++ b/ansible/roles/mempool/templates/healthcheck-frontend.sh.j2 @@ -0,0 +1,18 @@ +#!/bin/bash +# Mempool frontend health check — managed by Ansible (roles/mempool) +# The exit code is the answer; systemd keeps it. Reporting is optional. +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +FRONTEND_PORT="{{ mempool_frontend_port }}" + +check() { + curl -sf --max-time 5 "http://localhost:${FRONTEND_PORT}" > /dev/null 2>&1 +} + +report() { + [ -n "$PUSH_URL" ] || return 0 + curl -s --max-time 10 --retry 2 -o /dev/null \ + "${PUSH_URL}?status=$1&msg=${2// /%20}&ping=" || true +} + +if check; then report up "OK"; exit 0 +else echo "Mempool frontend not responding"; report down "Mempool frontend not responding"; exit 1; fi diff --git a/ansible/roles/mempool/templates/healthcheck-mariadb.sh.j2 b/ansible/roles/mempool/templates/healthcheck-mariadb.sh.j2 new file mode 100644 index 0000000..cbc36f5 --- /dev/null +++ b/ansible/roles/mempool/templates/healthcheck-mariadb.sh.j2 @@ -0,0 +1,21 @@ +#!/bin/bash +# Mempool MariaDB health check — managed by Ansible (roles/mempool) +# The exit code is the answer; systemd keeps it. Reporting is optional. +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" + +check() { +{% raw %} + [ "$(docker inspect --format='{{.State.Health.Status}}' mempool-db 2>/dev/null)" = "healthy" ] +{% endraw %} +} + +report() { + # No push URL is normal, not an error. The previous version logged + # "ERROR: UPTIME_KUMA_PUSH_URL not set" on every fire, once a minute. + [ -n "$PUSH_URL" ] || return 0 + curl -s --max-time 10 --retry 2 -o /dev/null \ + "${PUSH_URL}?status=$1&msg=${2// /%20}&ping=" || true +} + +if check; then report up "OK"; exit 0 +else echo "MariaDB container unhealthy"; report down "MariaDB container unhealthy"; exit 1; fi diff --git a/ansible/roles/mempool/templates/healthcheck.service.j2 b/ansible/roles/mempool/templates/healthcheck.service.j2 new file mode 100644 index 0000000..44548bd --- /dev/null +++ b/ansible/roles/mempool/templates/healthcheck.service.j2 @@ -0,0 +1,14 @@ +[Unit] +Description=Mempool {{ hc.label }} Health Check +After=network.target docker.service + +[Service] +Type=oneshot +User=root +ExecStart=/usr/local/bin/mempool-{{ hc.name }}-healthcheck-push.sh +Environment=HEALTHCHECK_PUSH_URL={{ hc.push_url }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/mempool/templates/healthcheck.timer.j2 b/ansible/roles/mempool/templates/healthcheck.timer.j2 new file mode 100644 index 0000000..868cb7c --- /dev/null +++ b/ansible/roles/mempool/templates/healthcheck.timer.j2 @@ -0,0 +1,10 @@ +[Unit] +Description=Mempool {{ hc.name }} Health Check Timer + +[Timer] +OnBootSec=2min +OnUnitActiveSec=1min +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/services/mempool/deploy_mempool_playbook.yml b/ansible/services/mempool/deploy_mempool_playbook.yml index e9cab0e..0766da0 100644 --- a/ansible/services/mempool/deploy_mempool_playbook.yml +++ b/ansible/services/mempool/deploy_mempool_playbook.yml @@ -1,3 +1,4 @@ +--- - name: Deploy Mempool Block Explorer with Docker hosts: mempool become: yes @@ -5,603 +6,17 @@ - ../../infra_vars.yml - ../../services_config.yml - ../../infra_secrets.yml - - ./mempool_vars.yml vars: - mempool_subdomain: "{{ subdomains.mempool }}" - mempool_domain: "{{ mempool_subdomain }}.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - # =========================================== - # Docker Installation (from 910_docker_playbook.yml) - # =========================================== - - name: Remove old Docker-related packages - apt: - name: - - docker.io - - docker-doc - - docker-compose - - podman-docker - - containerd - - runc - state: absent - purge: yes - autoremove: yes - - - name: Update apt cache - apt: - update_cache: yes - - - name: Install prerequisites - apt: - name: - - ca-certificates - - curl - state: present - - - name: Create directory for Docker GPG key - file: - path: /etc/apt/keyrings - state: directory - mode: '0755' - - - name: Download Docker GPG key - get_url: - url: https://download.docker.com/linux/debian/gpg - dest: /etc/apt/keyrings/docker.asc - mode: '0644' - - - name: Get Debian architecture - command: dpkg --print-architecture - register: deb_arch - changed_when: false - - - name: Add Docker repository - apt_repository: - repo: "deb [arch={{ deb_arch.stdout }} signed-by=/etc/apt/keyrings/docker.asc] https://download.docker.com/linux/debian {{ ansible_distribution_release }} stable" - filename: docker - state: present - update_cache: yes - - - name: Install Docker packages - apt: - name: - - docker-ce - - docker-ce-cli - - containerd.io - - docker-buildx-plugin - - docker-compose-plugin - state: present - update_cache: yes - - - name: Ensure Docker is started and enabled - systemd: - name: docker - enabled: yes - state: started - - - name: Add user to docker group - user: - name: "{{ ansible_user }}" - groups: docker - append: yes - - # =========================================== - # Mempool Deployment - # =========================================== - - name: Create mempool directories - file: - path: "{{ item }}" - state: directory - owner: "{{ ansible_user }}" - group: "{{ ansible_user }}" - mode: '0755' - loop: - - "{{ mempool_dir }}" - - "{{ mempool_data_dir }}" - - "{{ mempool_mysql_dir }}" - - - name: Create docker-compose.yml for Mempool - copy: - dest: "{{ mempool_dir }}/docker-compose.yml" - content: | - # All containers use host network for Tailscale MagicDNS resolution - services: - mariadb: - image: mariadb:10.11 - container_name: mempool-db - restart: unless-stopped - network_mode: host - environment: - MYSQL_DATABASE: "{{ mariadb_database }}" - MYSQL_USER: "{{ mariadb_user }}" - MYSQL_PASSWORD: "{{ mariadb_mempool_password }}" - MYSQL_ROOT_PASSWORD: "{{ mariadb_mempool_password }}" - volumes: - - {{ mempool_mysql_dir }}:/var/lib/mysql - healthcheck: - test: ["CMD", "healthcheck.sh", "--connect", "--innodb_initialized"] - interval: 10s - timeout: 5s - retries: 5 - start_period: 30s - - mempool-backend: - image: mempool/backend:{{ mempool_version }} - container_name: mempool-backend - restart: unless-stopped - network_mode: host - environment: - # Database (localhost since all containers share host network) - DATABASE_ENABLED: "true" - DATABASE_HOST: "127.0.0.1" - DATABASE_DATABASE: "{{ mariadb_database }}" - DATABASE_USERNAME: "{{ mariadb_user }}" - DATABASE_PASSWORD: "{{ mariadb_mempool_password }}" - # Bitcoin Core/Knots (via Tailnet MagicDNS) - CORE_RPC_HOST: "{{ bitcoin_host }}" - CORE_RPC_PORT: "{{ bitcoin_rpc_port }}" - CORE_RPC_USERNAME: "{{ bitcoin_rpc_user }}" - CORE_RPC_PASSWORD: "{{ bitcoin_rpc_password }}" - # Electrum (Fulcrum via Tailnet MagicDNS) - ELECTRUM_HOST: "{{ fulcrum_host }}" - ELECTRUM_PORT: "{{ fulcrum_port }}" - ELECTRUM_TLS_ENABLED: "{{ fulcrum_tls }}" - # Mempool settings - MEMPOOL_NETWORK: "{{ mempool_network }}" - MEMPOOL_BACKEND: "electrum" - MEMPOOL_CLEAR_PROTECTION_MINUTES: "20" - MEMPOOL_INDEXING_BLOCKS_AMOUNT: "52560" - volumes: - - {{ mempool_data_dir }}:/backend/cache - depends_on: - mariadb: - condition: service_healthy - healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:8999/api/v1/backend-info"] - interval: 30s - timeout: 10s - retries: 3 - start_period: 60s - - mempool-frontend: - image: mempool/frontend:{{ mempool_version }} - container_name: mempool-frontend - restart: unless-stopped - network_mode: host - environment: - FRONTEND_HTTP_PORT: "{{ mempool_frontend_port }}" - BACKEND_MAINNET_HTTP_HOST: "127.0.0.1" - depends_on: - - mempool-backend - healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:{{ mempool_frontend_port }}"] - interval: 30s - timeout: 10s - retries: 3 - start_period: 30s - owner: "{{ ansible_user }}" - group: "{{ ansible_user }}" - mode: '0644' - - - name: Pull Mempool images - command: docker compose pull - args: - chdir: "{{ mempool_dir }}" - - - name: Deploy Mempool containers with docker compose - command: docker compose up -d - args: - chdir: "{{ mempool_dir }}" - - - name: Wait for MariaDB to be healthy - command: docker inspect --format='{{ '{{' }}.State.Health.Status{{ '}}' }}' mempool-db - register: mariadb_health - until: mariadb_health.stdout == 'healthy' - retries: 30 - delay: 10 - changed_when: false - - - name: Wait for Mempool backend to start - uri: - url: "http://localhost:{{ mempool_backend_port }}/api/v1/backend-info" - method: GET - status_code: 200 - timeout: 10 - register: backend_check - until: backend_check.status == 200 - retries: 30 - delay: 10 - ignore_errors: yes - - - name: Wait for Mempool frontend to be available - uri: - url: "http://localhost:{{ mempool_frontend_port }}" - method: GET - status_code: 200 - timeout: 10 - register: frontend_check - until: frontend_check.status == 200 - retries: 20 - delay: 5 - ignore_errors: yes - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Display deployment status - when: uptime_kuma_enabled | default(false) - debug: - msg: - - "Mempool deployment complete!" - - "Frontend: http://localhost:{{ mempool_frontend_port }}" - - "Backend API: http://localhost:{{ mempool_backend_port }}/api/v1/backend-info" - - "Backend check: {{ 'OK' if backend_check.status == 200 else 'Still initializing...' }}" - - "Frontend check: {{ 'OK' if frontend_check.status == 200 else 'Still initializing...' }}" - - # =========================================== - # Health Check Scripts for Uptime Kuma Push Monitors - # =========================================== - - name: Create Mempool MariaDB health check script - when: uptime_kuma_enabled | default(false) - copy: - dest: /usr/local/bin/mempool-mariadb-healthcheck-push.sh - content: | - #!/bin/bash - UPTIME_KUMA_PUSH_URL="${UPTIME_KUMA_PUSH_URL}" - - check_container() { - local status=$(docker inspect --format='{{ '{{' }}.State.Health.Status{{ '}}' }}' mempool-db 2>/dev/null) - [ "$status" = "healthy" ] - } - - push_to_uptime_kuma() { - local status=$1 - local msg=$2 - if [ -z "$UPTIME_KUMA_PUSH_URL" ]; then - echo "ERROR: UPTIME_KUMA_PUSH_URL not set" - return 1 - fi - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${UPTIME_KUMA_PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true - } - - if check_container; then - push_to_uptime_kuma "up" "OK" - exit 0 - else - push_to_uptime_kuma "down" "MariaDB container unhealthy" - exit 1 - fi - owner: root - group: root - mode: '0755' - - - name: Create Mempool backend health check script - when: uptime_kuma_enabled | default(false) - copy: - dest: /usr/local/bin/mempool-backend-healthcheck-push.sh - content: | - #!/bin/bash - UPTIME_KUMA_PUSH_URL="${UPTIME_KUMA_PUSH_URL}" - BACKEND_PORT={{ mempool_backend_port }} - - check_backend() { - curl -sf --max-time 5 "http://localhost:${BACKEND_PORT}/api/v1/backend-info" > /dev/null 2>&1 - } - - push_to_uptime_kuma() { - local status=$1 - local msg=$2 - if [ -z "$UPTIME_KUMA_PUSH_URL" ]; then - echo "ERROR: UPTIME_KUMA_PUSH_URL not set" - return 1 - fi - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${UPTIME_KUMA_PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true - } - - if check_backend; then - push_to_uptime_kuma "up" "OK" - exit 0 - else - push_to_uptime_kuma "down" "Backend API not responding" - exit 1 - fi - owner: root - group: root - mode: '0755' - - - name: Create Mempool frontend health check script - when: uptime_kuma_enabled | default(false) - copy: - dest: /usr/local/bin/mempool-frontend-healthcheck-push.sh - content: | - #!/bin/bash - UPTIME_KUMA_PUSH_URL="${UPTIME_KUMA_PUSH_URL}" - FRONTEND_PORT={{ mempool_frontend_port }} - - check_frontend() { - curl -sf --max-time 5 "http://localhost:${FRONTEND_PORT}" > /dev/null 2>&1 - } - - push_to_uptime_kuma() { - local status=$1 - local msg=$2 - if [ -z "$UPTIME_KUMA_PUSH_URL" ]; then - echo "ERROR: UPTIME_KUMA_PUSH_URL not set" - return 1 - fi - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${UPTIME_KUMA_PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true - } - - if check_frontend; then - push_to_uptime_kuma "up" "OK" - exit 0 - else - push_to_uptime_kuma "down" "Frontend not responding" - exit 1 - fi - owner: root - group: root - mode: '0755' - - # =========================================== - # Systemd Timers for Health Checks - # =========================================== - - name: Create systemd services for health checks - copy: - dest: "/etc/systemd/system/mempool-{{ item.name }}-healthcheck.service" - content: | - [Unit] - Description=Mempool {{ item.label }} Health Check - After=network.target docker.service - - [Service] - Type=oneshot - User=root - ExecStart=/usr/local/bin/mempool-{{ item.name }}-healthcheck-push.sh - Environment=UPTIME_KUMA_PUSH_URL= - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - loop: - - { name: "mariadb", label: "MariaDB" } - - { name: "backend", label: "Backend" } - - { name: "frontend", label: "Frontend" } - - - name: Create systemd timers for health checks - copy: - dest: "/etc/systemd/system/mempool-{{ item }}-healthcheck.timer" - content: | - [Unit] - Description=Mempool {{ item }} Health Check Timer - - [Timer] - OnBootSec=2min - OnUnitActiveSec=1min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - loop: - - mariadb - - backend - - frontend - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start health check timers - when: uptime_kuma_enabled | default(false) - systemd: - name: "mempool-{{ item }}-healthcheck.timer" - enabled: yes - state: started - loop: - - mariadb - - backend - - frontend - - # =========================================== - # Uptime Kuma Push Monitor Setup - # =========================================== - - name: Create Uptime Kuma push monitor setup script for Mempool - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_mempool_monitors.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_mempool_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitors_to_create = config['monitors'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - api.add_monitor(type='group', name='services') - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - results = {} - for monitor_name in monitors_to_create: - existing = next((m for m in monitors if m.get('name') == monitor_name), None) - - if existing: - print(f"Monitor '{monitor_name}' already exists (ID: {existing['id']})") - push_token = existing.get('pushToken') or existing.get('push_token') - if push_token: - results[monitor_name] = f"{url}/api/push/{push_token}" - print(f"Push URL ({monitor_name}): {results[monitor_name]}") - else: - print(f"Creating push monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.PUSH, - name=monitor_name, - parent=group['id'], - interval=90, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - monitors = api.get_monitors() - new_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - if new_monitor: - push_token = new_monitor.get('pushToken') or new_monitor.get('push_token') - if push_token: - results[monitor_name] = f"{url}/api/push/{push_token}" - print(f"Push URL ({monitor_name}): {results[monitor_name]}") - - api.disconnect() - print("SUCCESS") - - # Write results to file for Ansible to read - with open('/tmp/mempool_push_urls.yml', 'w') as f: - yaml.dump(results, f) - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_mempool_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitors: - - "Mempool MariaDB" - - "Mempool Backend" - - "Mempool Frontend" - mode: '0644' - - - name: Run Uptime Kuma push monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_mempool_monitors.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Display monitor setup output - debug: - msg: "{{ monitor_setup.stdout_lines }}" - when: monitor_setup.stdout is defined - - - name: Read push URLs from file - when: uptime_kuma_enabled | default(false) - slurp: - src: /tmp/mempool_push_urls.yml - delegate_to: localhost - become: no - register: push_urls_file - ignore_errors: yes - - - name: Parse push URLs - set_fact: - push_urls: "{{ push_urls_file.content | b64decode | from_yaml }}" - when: push_urls_file.content is defined - ignore_errors: yes - - - name: Update MariaDB health check service with push URL - lineinfile: - path: /etc/systemd/system/mempool-mariadb-healthcheck.service - regexp: '^Environment=UPTIME_KUMA_PUSH_URL=' - line: "Environment=UPTIME_KUMA_PUSH_URL={{ push_urls['Mempool MariaDB'] }}" - insertafter: '^\[Service\]' - when: push_urls is defined and push_urls['Mempool MariaDB'] is defined - - - name: Update Backend health check service with push URL - lineinfile: - path: /etc/systemd/system/mempool-backend-healthcheck.service - regexp: '^Environment=UPTIME_KUMA_PUSH_URL=' - line: "Environment=UPTIME_KUMA_PUSH_URL={{ push_urls['Mempool Backend'] }}" - insertafter: '^\[Service\]' - when: push_urls is defined and push_urls['Mempool Backend'] is defined - - - name: Update Frontend health check service with push URL - lineinfile: - path: /etc/systemd/system/mempool-frontend-healthcheck.service - regexp: '^Environment=UPTIME_KUMA_PUSH_URL=' - line: "Environment=UPTIME_KUMA_PUSH_URL={{ push_urls['Mempool Frontend'] }}" - insertafter: '^\[Service\]' - when: push_urls is defined and push_urls['Mempool Frontend'] is defined - - - name: Reload systemd after push URL updates - systemd: - daemon_reload: yes - when: push_urls is defined - - - name: Restart health check timers - systemd: - name: "mempool-{{ item }}-healthcheck.timer" - state: restarted - loop: - - mariadb - - backend - - frontend - when: push_urls is defined - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_mempool_monitors.py - - /tmp/ansible_mempool_config.yml - - /tmp/mempool_push_urls.yml + # Preserves the three push URLs these checks have been reporting to all + # along, so the move to a role changes no behaviour. The role knows nothing + # about Uptime Kuma — these are just "URLs that accept a ping", and whatever + # replaces it sets the same values. + mempool_healthchecks: + - {name: mariadb, label: MariaDB, push_url: "{{ healthcheck_push_urls.mempool.mariadb | default('') }}"} + - {name: backend, label: Backend, push_url: "{{ healthcheck_push_urls.mempool.backend | default('') }}"} + - {name: frontend, label: Frontend, push_url: "{{ healthcheck_push_urls.mempool.frontend | default('') }}"} + roles: + - mempool - name: Configure Caddy reverse proxy for Mempool on the edge host hosts: edge @@ -609,12 +24,8 @@ vars_files: - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - - ./mempool_vars.yml vars: - mempool_subdomain: "{{ subdomains.mempool }}" - mempool_domain: "{{ mempool_subdomain }}.{{ root_domain }}" - + mempool_domain: "{{ subdomains.mempool }}.{{ root_domain }}" tasks: - name: Publish Mempool through Caddy (via Tailscale) ansible.builtin.include_role: @@ -622,124 +33,5 @@ vars: caddy_site_name: mempool caddy_site_domain: "{{ mempool_domain }}" - caddy_site_upstream: "mempool-box:{{ mempool_frontend_port }}" + caddy_site_upstream: "mempool-box:{{ service_settings.mempool.frontend_port }}" caddy_site_resolvers: "100.100.100.100" - - - name: Display Mempool URL - when: uptime_kuma_enabled | default(false) - debug: - msg: "Mempool is now available at https://{{ mempool_domain }}" - - # =========================================== - # Uptime Kuma HTTP Monitor for Public Endpoint - # =========================================== - - name: Create Uptime Kuma HTTP monitor setup script for Mempool - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_mempool_http_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_mempool_http_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - api.add_monitor(type='group', name='services') - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing = next((m for m in monitors if m.get('name') == monitor_name), None) - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing: - print(f"Monitor '{monitor_name}' already exists (ID: {existing['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating HTTP monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for HTTP monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_mempool_http_config.yml - content: | - uptime_kuma_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ mempool_domain }}" - monitor_name: "Mempool Public Access" - mode: '0644' - - - name: Run Uptime Kuma HTTP monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_mempool_http_monitor.py - delegate_to: localhost - become: no - register: http_monitor_setup - changed_when: "'SUCCESS' in http_monitor_setup.stdout" - ignore_errors: yes - - - name: Display HTTP monitor setup output - debug: - msg: "{{ http_monitor_setup.stdout_lines }}" - when: http_monitor_setup.stdout is defined - - - name: Clean up HTTP monitor temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_mempool_http_monitor.py - - /tmp/ansible_mempool_http_config.yml - diff --git a/ansible/services/mempool/mempool_vars.yml b/ansible/services/mempool/mempool_vars.yml deleted file mode 100644 index d3051c3..0000000 --- a/ansible/services/mempool/mempool_vars.yml +++ /dev/null @@ -1,33 +0,0 @@ -# Mempool Configuration Variables - -# Version - Pinned to specific release -mempool_version: "v3.2.1" - -# Directories -mempool_dir: /opt/mempool -mempool_data_dir: "{{ mempool_dir }}/data" -mempool_mysql_dir: "{{ mempool_dir }}/mysql" - -# Network - Bitcoin Core/Knots connection (via Tailnet Magic DNS) -bitcoin_host: "knots-box" -bitcoin_rpc_port: 8332 -# Note: bitcoin_rpc_user and bitcoin_rpc_password are loaded from infra_secrets.yml - -# Network - Fulcrum Electrum server (via Tailnet Magic DNS) -fulcrum_host: "fulcrum-box" -fulcrum_port: 50001 -fulcrum_tls: "false" - -# Mempool network mode -mempool_network: "mainnet" - -# Container ports (internal) -mempool_frontend_port: 8080 -mempool_backend_port: 8999 - -# MariaDB settings -mariadb_database: "mempool" -mariadb_user: "mempool" -# Note: mariadb_mempool_password is loaded from infra_secrets.yml - - diff --git a/ansible/services_config.yml b/ansible/services_config.yml index a20c853..5014297 100644 --- a/ansible/services_config.yml +++ b/ansible/services_config.yml @@ -40,3 +40,9 @@ service_settings: topic: alerts headscale: namespace: counter-net + mempool: + # The frontend port is needed in two places on two different hosts: the + # mempool role deploys it on mempool-box, and the Caddy play proxies to it + # from the edge host. A role default cannot serve the second play, so it + # lives here rather than in roles/mempool/defaults. + frontend_port: 8080 From e83191c02904fe7cfdd3f809bea8360b965e9c7b Mon Sep 17 00:00:00 2001 From: counterweight Date: Sun, 13 Sep 2026 18:02:00 +0200 Subject: [PATCH 51/67] fulcrum: convert to a role, de-Uptime-Kuma the health check 685-line playbook becomes 33 lines plus a 426-line role (install/service/healthcheck phases, six templates, one handler). fulcrum_vars.yml is deleted; its content is the role's defaults. Verified: fulcrum untouched - active since 2026-07-29 (no restart), 192G datadir, height 966842, bitcoind and db_mem unchanged on disk. Second run changed=1 (the arming run, changed_when: false aside). vipy changed=0. THREE PRE-EXISTING LANDMINES the check-mode diff caught, any of which a faithful extraction would have detonated: - bitcoin_rpc_host was "192.168.1.140", commented "IP of knots_box_local". But .140 is fulcrum-box ITSELF; knots-box is .135. The DHCP leases had reshuffled - the fifth instance of this same disease in this estate. The live config had been hand-corrected to knots-box; running the playbook would have reverted it and pointed Fulcrum at itself. Now addressed by Tailscale name. - The `Restart fulcrum` handler was guarded by uptime_kuma_enabled, so the three tasks that notify it (SSL cert, fulcrum.conf, systemd unit) could not restart anything. A config change applied to disk, reported success, and silently never took effect. That is worse than the other banner casualties: it makes the deployment itself lie. Ungated. - db_mem was about to go 2048 -> 4448 (75% of 5931MB RAM), leaving ~1.4GB for the OS and Fulcrum's non-cache memory. The live value had been hand-tuned down. fulcrum_db_mem_mb_override pins it. Note set_fact outranks role defaults, so the calculation itself has to honour the override. MY OWN ERROR, third instance: retyping `copy:` as `template:` lost `owner:` on the banner and on fulcrum.conf. Rather than keep catching these by eye, every managed path's owner/group/mode is now compared against `git show HEAD:` mechanically - 12/12 match. The health check timer had not fired since 2026-02-17 while reporting `active` and `enabled`. It is OnBootSec + OnUnitActiveSec with no OnCalendar: OnBootSec elapses once, and OnUnitActiveSec needs the SERVICE to have run this boot to have anything to schedule from. Restarting the timer does not supply that; running the service does, so the role now runs the check once after enabling. Also dropped `Requires=fulcrum.service` from the timer - on a timer that means "stop watching when the watched thing stops". Diagnostic note: NextElapseUSecRealtime is always empty for a monotonic timer, so it reads as broken even when healthy. I misread it once and wrongly called the timer dead. Use NextElapseUSecMonotonic or systemctl list-timers. fulcrum_ssl_port and fulcrum_tailscale_hostname moved to services_config.yml - the socket-proxy play on the edge host needs them and a role default cannot reach a second play. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/roles/fulcrum/README.md | 65 ++ .../fulcrum/defaults/main.yml} | 33 +- ansible/roles/fulcrum/handlers/main.yml | 15 + ansible/roles/fulcrum/tasks/healthcheck.yml | 62 ++ ansible/roles/fulcrum/tasks/install.yml | 146 ++++ ansible/roles/fulcrum/tasks/main.yml | 6 + ansible/roles/fulcrum/tasks/service.yml | 54 ++ ansible/roles/fulcrum/templates/banner.txt.j2 | 3 + .../roles/fulcrum/templates/fulcrum.conf.j2 | 29 + .../fulcrum/templates/fulcrum.service.j2 | 24 + .../fulcrum/templates/healthcheck.service.j2 | 14 + .../roles/fulcrum/templates/healthcheck.sh.j2 | 34 + .../fulcrum/templates/healthcheck.timer.j2 | 17 + .../fulcrum/deploy_fulcrum_playbook.yml | 674 +----------------- ansible/services_config.yml | 6 + 15 files changed, 516 insertions(+), 666 deletions(-) create mode 100644 ansible/roles/fulcrum/README.md rename ansible/{services/fulcrum/fulcrum_vars.yml => roles/fulcrum/defaults/main.yml} (54%) create mode 100644 ansible/roles/fulcrum/handlers/main.yml create mode 100644 ansible/roles/fulcrum/tasks/healthcheck.yml create mode 100644 ansible/roles/fulcrum/tasks/install.yml create mode 100644 ansible/roles/fulcrum/tasks/main.yml create mode 100644 ansible/roles/fulcrum/tasks/service.yml create mode 100644 ansible/roles/fulcrum/templates/banner.txt.j2 create mode 100644 ansible/roles/fulcrum/templates/fulcrum.conf.j2 create mode 100644 ansible/roles/fulcrum/templates/fulcrum.service.j2 create mode 100644 ansible/roles/fulcrum/templates/healthcheck.service.j2 create mode 100644 ansible/roles/fulcrum/templates/healthcheck.sh.j2 create mode 100644 ansible/roles/fulcrum/templates/healthcheck.timer.j2 diff --git a/ansible/roles/fulcrum/README.md b/ansible/roles/fulcrum/README.md new file mode 100644 index 0000000..ddefabb --- /dev/null +++ b/ansible/roles/fulcrum/README.md @@ -0,0 +1,65 @@ +# `fulcrum` + +Deploys [Fulcrum](https://github.com/cculianu/Fulcrum), an Electrum server +indexing the Bitcoin Knots node, on `fulcrum-box`. The second play in the +calling playbook publishes its SSL port from the edge host via `socket_proxy`. + +Converted from `deploy_fulcrum_playbook.yml` (685 lines) under Plan 6. The +playbook is now 33 lines. + +## The index is the expensive thing + +`{{ fulcrum_db_dir }}` is ~192 GB and takes days to rebuild. Nothing in this role +touches it beyond `state: directory` with the ownership it already has +(`fulcrum:fulcrum 0755`). Restarting Fulcrum re-opens the database; it does not +reindex. + +## Three things this conversion fixed, all pre-existing + +**`bitcoind` pointed at the wrong machine.** The vars file carried +`bitcoin_rpc_host: "192.168.1.140"` commented "IP of knots_box_local", but `.140` +is **fulcrum-box itself** — knots-box is `.135`. The DHCP leases had reshuffled. +The live config had already been hand-corrected to `knots-box`; running the +playbook would have reverted it and broken indexing. Now addressed by Tailscale +name, like everything else in this repo. + +**The restart handler was inert.** It carried +`when: uptime_kuma_enabled | default(false)`, so the three tasks that notify it +(SSL certificate, `fulcrum.conf`, systemd unit) could not restart anything. A +configuration change would write to disk, report success, and silently never take +effect. Ungated. + +**`db_mem` was about to quadruple.** The role computes a share of RAM; on this +5931 MB host 75% is 4448 MB, leaving ~1.4 GB for the OS and Fulcrum's non-cache +memory. The live value had been hand-tuned to 2048. `fulcrum_db_mem_mb_override` +pins it. Note `set_fact` outranks role defaults, so the *calculation* has to +honour the override — pinning it in `defaults/` alone is silently ignored. + +## The health check timer, and how to read it + +The timer is `OnBootSec` + `OnUnitActiveSec` with no `OnCalendar`. That +combination has a failure mode worth knowing: `OnBootSec` is monotonic and +elapses once; `OnUnitActiveSec` schedules relative to the **service** last being +active. If the service does not run in a given boot, there is no reference to +schedule from and the timer sits `active` and `enabled` doing nothing. That is +exactly what had happened here — last trigger **2026-02-17**, seven months of no +health check, with every surface-level indicator green. + +Restarting the timer does not supply that reference; running the service does. +So the role runs the check once after enabling the timer, which is both the fix +and a smoke test. + +**Diagnosing this is easy to get wrong**: `NextElapseUSecRealtime` is always +empty for a monotonic timer, so it looks broken even when it is fine. Read +`NextElapseUSecMonotonic`, or just use `systemctl list-timers`. + +The timer also no longer carries `Requires=fulcrum.service`. On a timer that +means "stop watching when the watched thing stops", which is backwards for a +health check. + +## Monitoring: one variable, no product knowledge + +The check tests the Electrum TCP port and records the answer in its exit code, +which systemd keeps: `systemctl is-failed fulcrum-healthcheck.service`. To report +elsewhere set `healthcheck_push_url` to any endpoint accepting an HTTP ping. The +Uptime Kuma API calls, monitor creation and token handling are gone. diff --git a/ansible/services/fulcrum/fulcrum_vars.yml b/ansible/roles/fulcrum/defaults/main.yml similarity index 54% rename from ansible/services/fulcrum/fulcrum_vars.yml rename to ansible/roles/fulcrum/defaults/main.yml index 8486b14..df213be 100644 --- a/ansible/services/fulcrum/fulcrum_vars.yml +++ b/ansible/roles/fulcrum/defaults/main.yml @@ -12,13 +12,22 @@ fulcrum_binary_path: /usr/local/bin/Fulcrum # Network - Bitcoin RPC connection # Bitcoin Knots is on a different host (knots_box_local) # Using RPC user/password authentication (credentials from infra_secrets.yml) -bitcoin_rpc_host: "192.168.1.140" # Bitcoin Knots RPC host (IP of knots_box_local) +# Addressed by Tailscale name, never a LAN IP. This was +# bitcoin_rpc_host: "192.168.1.140" # IP of knots_box_local +# but .140 is fulcrum-box ITSELF - knots-box is .135. The DHCP leases had +# reshuffled (the same drift that transposed the inventory), so running this +# playbook would have pointed Fulcrum at itself and broken indexing. The live +# config had already been hand-corrected to knots-box; this makes the repo +# agree with it. +bitcoin_rpc_host: "knots-box" bitcoin_rpc_port: 8332 # Bitcoin Knots RPC port # Note: bitcoin_rpc_user and bitcoin_rpc_password are loaded from infra_secrets.yml # Network - Fulcrum server fulcrum_tcp_port: 50001 -fulcrum_ssl_port: 50002 +# Shared with the socket-proxy play on the edge host, so it lives in +# services_config.yml rather than only here. +fulcrum_ssl_port: "{{ service_settings.fulcrum.ssl_port }}" # Binding address for Fulcrum TCP/SSL server: # - "127.0.0.1" = localhost only (use when Caddy is on the same box) # - "0.0.0.0" = all interfaces (use when Caddy is on a different box) @@ -34,10 +43,14 @@ fulcrum_ssl_key_path: "{{ fulcrum_config_dir }}/fulcrum.key" fulcrum_ssl_cert_days: 3650 # 10 years validity for self-signed cert # Port forwarding configuration (for public access via VPS) -fulcrum_tailscale_hostname: "fulcrum-box" +fulcrum_tailscale_hostname: "{{ service_settings.fulcrum.tailscale_hostname }}" # Performance # db_mem will be calculated as 75% of available RAM automatically in playbook +# db_mem is computed as this share of RAM unless fulcrum_db_mem_mb is set +# explicitly. On a 5931 MB host 75% is 4448 MB, which leaves ~1.4 GB for the +# OS and for Fulcrum's non-cache memory; the live config had been hand-tuned +# down to 2048 and that setting is respected below. fulcrum_db_mem_percent: 0.75 # 75% of RAM for database cache # Configuration options @@ -49,3 +62,17 @@ fulcrum_zmq_allow_hashtx: true # Allow ZMQ hashtx notifications fulcrum_user: fulcrum fulcrum_group: fulcrum + +# --- Health check ----------------------------------------------------------- +# Checks the Electrum TCP port and records the answer in its exit code, which +# systemd keeps: `systemctl is-failed fulcrum-healthcheck.service`. +# +# WHERE TO REPORT HEALTH — the one place to plug in monitoring. Empty means +# check, exit honestly, report nowhere. Any endpoint accepting an HTTP ping +# works; nothing here is specific to a monitoring product. +healthcheck_push_url: "" + +# Explicit db_mem in MB. When set it wins over fulcrum_db_mem_percent; empty +# means compute from RAM. Set here because the live host had been hand-tuned to +# 2048 and a silent jump to 4448 is not something a refactor should do. +fulcrum_db_mem_mb_override: 2048 diff --git a/ansible/roles/fulcrum/handlers/main.yml b/ansible/roles/fulcrum/handlers/main.yml new file mode 100644 index 0000000..c0d9cc4 --- /dev/null +++ b/ansible/roles/fulcrum/handlers/main.yml @@ -0,0 +1,15 @@ +--- +# Ungated on purpose. The hand-written handler carried +# when: uptime_kuma_enabled | default(false) +# so it has been inert since the decommissioning: three tasks notify it (the SSL +# certificate, fulcrum.conf and the systemd unit), and none of them could +# actually restart Fulcrum. A configuration change therefore applied to disk and +# silently never took effect — the worst kind of quiet failure, because the +# playbook reports success and the running service keeps its old settings. +# +# Restarting Fulcrum re-opens its database; it does not reindex. +- name: Restart fulcrum + systemd: + name: fulcrum + state: restarted + daemon_reload: yes diff --git a/ansible/roles/fulcrum/tasks/healthcheck.yml b/ansible/roles/fulcrum/tasks/healthcheck.yml new file mode 100644 index 0000000..e86726e --- /dev/null +++ b/ansible/roles/fulcrum/tasks/healthcheck.yml @@ -0,0 +1,62 @@ +--- +# Everything here answers "is Fulcrum healthy" and records the answer. The +# Uptime Kuma specifics that used to follow — an embedded Python script creating +# monitors over the API, a /tmp credentials file, push-URL extraction and a +# systemd Environment= rewrite — are gone. Where it reports is now one variable, +# healthcheck_push_url. See the role README. +- name: Create Fulcrum health check script + ansible.builtin.template: + src: healthcheck.sh.j2 + dest: /usr/local/bin/fulcrum-healthcheck-push.sh + owner: root + group: root + mode: '0755' + validate: "bash -n %s" + +- name: Create systemd service for Fulcrum health check + ansible.builtin.template: + src: healthcheck.service.j2 + dest: /etc/systemd/system/fulcrum-healthcheck.service + owner: root + group: root + mode: '0644' + +- name: Create systemd timer for Fulcrum health check + ansible.builtin.template: + src: healthcheck.timer.j2 + dest: /etc/systemd/system/fulcrum-healthcheck.timer + owner: root + group: root + mode: '0644' + +- name: Reload systemd daemon for health check + systemd: + daemon_reload: yes + +# state: restarted, not started. The hand-written timer had got itself stuck +# `active` with no next elapse and had not fired since 2026-02-17; `started` on +# an already-active timer is a no-op and would have left it stuck. Restarting +# re-arms it. See the note in healthcheck.timer.j2. +- name: Enable and restart the Fulcrum health check timer + systemd: + name: fulcrum-healthcheck.timer + enabled: yes + state: restarted + daemon_reload: yes + +# Run the check once, which is both a smoke test and the thing that actually +# arms the timer. +# +# This timer is OnBootSec + OnUnitActiveSec with no OnCalendar. OnBootSec is +# monotonic and had long since elapsed; OnUnitActiveSec schedules relative to the +# SERVICE last being active, and the service had not run since 2026-02-17 — so +# there was no reference to schedule from and the timer sat `active` and +# `enabled` with NextElapseUSecMonotonic=infinity. Restarting the timer alone +# does not supply that reference; running the service does. +# +# (Diagnosing this is easy to get wrong: NextElapseUSecRealtime is always empty +# for a monotonic timer, so it looks broken even when it is fine. Read +# NextElapseUSecMonotonic, or just use `systemctl list-timers`.) +- name: Run the Fulcrum health check once to arm the timer + command: systemctl start fulcrum-healthcheck.service + changed_when: false diff --git a/ansible/roles/fulcrum/tasks/install.yml b/ansible/roles/fulcrum/tasks/install.yml new file mode 100644 index 0000000..1216a96 --- /dev/null +++ b/ansible/roles/fulcrum/tasks/install.yml @@ -0,0 +1,146 @@ +--- +- name: Calculate db_mem as a share of system RAM + set_fact: + fulcrum_db_mem_mb: "{{ (ansible_memtotal_mb | float * fulcrum_db_mem_percent) | int }}" + when: fulcrum_db_mem_mb_override | string | length == 0 + +- name: Use the explicit db_mem override + set_fact: + fulcrum_db_mem_mb: "{{ fulcrum_db_mem_mb_override }}" + when: fulcrum_db_mem_mb_override | string | length > 0 + changed_when: false + +- name: Display calculated db_mem value + debug: + msg: "Setting db_mem to {{ fulcrum_db_mem_mb }} MB ({{ (fulcrum_db_mem_percent * 100) | int }}% of {{ ansible_memtotal_mb }} MB total RAM)" + +- name: Display Fulcrum version to install + debug: + msg: "Installing Fulcrum version {{ fulcrum_version }}" + +- name: Install required packages + apt: + name: + - curl + - wget + - openssl + state: present + update_cache: yes + +- name: Create fulcrum group + group: + name: "{{ fulcrum_group }}" + system: yes + state: present + +- name: Create fulcrum user + user: + name: "{{ fulcrum_user }}" + group: "{{ fulcrum_group }}" + system: yes + shell: /usr/sbin/nologin + home: /home/{{ fulcrum_user }} + create_home: yes + state: present + +- name: Create Fulcrum database directory (heavy data on special mount) + file: + path: "{{ fulcrum_db_dir }}" + state: directory + owner: "{{ fulcrum_user }}" + group: "{{ fulcrum_group }}" + mode: '0755' + +- name: Create Fulcrum config directory + file: + path: "{{ fulcrum_config_dir }}" + state: directory + owner: root + group: "{{ fulcrum_group }}" + mode: '0755' + +- name: Create Fulcrum lib directory (for banner and other data files) + file: + path: "{{ fulcrum_lib_dir }}" + state: directory + owner: "{{ fulcrum_user }}" + group: "{{ fulcrum_group }}" + mode: '0755' + +# =========================================== +# SSL Certificate Generation +# =========================================== +- name: Check if SSL certificate already exists + stat: + path: "{{ fulcrum_ssl_cert_path }}" + register: fulcrum_ssl_cert_exists + when: fulcrum_ssl_enabled | default(false) + +- name: Generate self-signed SSL certificate for Fulcrum + command: > + openssl req -x509 -newkey rsa:4096 + -keyout {{ fulcrum_ssl_key_path }} + -out {{ fulcrum_ssl_cert_path }} + -sha256 -days {{ fulcrum_ssl_cert_days }} + -nodes + -subj "/C=XX/ST=Decentralized/L=Bitcoin/O=Fulcrum/OU=Electrum/CN=fulcrum.local" + args: + creates: "{{ fulcrum_ssl_cert_path }}" + when: fulcrum_ssl_enabled | default(false) + notify: Restart fulcrum + +- name: Set SSL certificate permissions + file: + path: "{{ fulcrum_ssl_cert_path }}" + owner: "{{ fulcrum_user }}" + group: "{{ fulcrum_group }}" + mode: '0644' + when: fulcrum_ssl_enabled | default(false) and fulcrum_ssl_cert_exists.stat.exists | default(false) or fulcrum_ssl_enabled | default(false) + +- name: Set SSL key permissions + file: + path: "{{ fulcrum_ssl_key_path }}" + owner: "{{ fulcrum_user }}" + group: "{{ fulcrum_group }}" + mode: '0600' + when: fulcrum_ssl_enabled | default(false) + +- name: Check if Fulcrum binary already exists + stat: + path: "{{ fulcrum_binary_path }}" + register: fulcrum_binary_exists + changed_when: false + +- name: Download Fulcrum binary tarball + get_url: + url: "https://github.com/cculianu/Fulcrum/releases/download/v{{ fulcrum_version }}/Fulcrum-{{ fulcrum_version }}-x86_64-linux.tar.gz" + dest: "/tmp/Fulcrum-{{ fulcrum_version }}-x86_64-linux.tar.gz" + mode: '0644' + when: not fulcrum_binary_exists.stat.exists + +- name: Extract Fulcrum binary + unarchive: + src: "/tmp/Fulcrum-{{ fulcrum_version }}-x86_64-linux.tar.gz" + dest: "/tmp" + remote_src: yes + when: not fulcrum_binary_exists.stat.exists + +- name: Install Fulcrum binary + copy: + src: "/tmp/Fulcrum-{{ fulcrum_version }}-x86_64-linux/Fulcrum" + dest: "{{ fulcrum_binary_path }}" + owner: root + group: root + mode: '0755' + remote_src: yes + when: not fulcrum_binary_exists.stat.exists + +- name: Verify Fulcrum binary installation + command: "{{ fulcrum_binary_path }} --version" + register: fulcrum_version_check + changed_when: false + +- name: Display Fulcrum version + debug: + msg: "{{ fulcrum_version_check.stdout_lines }}" + diff --git a/ansible/roles/fulcrum/tasks/main.yml b/ansible/roles/fulcrum/tasks/main.yml new file mode 100644 index 0000000..bc7ff05 --- /dev/null +++ b/ansible/roles/fulcrum/tasks/main.yml @@ -0,0 +1,6 @@ +--- +# import_tasks, not include_tasks: static imports stay visible to --list-tasks, +# which is how this conversion was verified against the playbook it replaced. +- ansible.builtin.import_tasks: install.yml +- ansible.builtin.import_tasks: service.yml +- ansible.builtin.import_tasks: healthcheck.yml diff --git a/ansible/roles/fulcrum/tasks/service.yml b/ansible/roles/fulcrum/tasks/service.yml new file mode 100644 index 0000000..716b44d --- /dev/null +++ b/ansible/roles/fulcrum/tasks/service.yml @@ -0,0 +1,54 @@ +--- +- name: Create Fulcrum banner file + ansible.builtin.template: + src: banner.txt.j2 + dest: "{{ fulcrum_lib_dir }}/fulcrum-banner.txt" + owner: "{{ fulcrum_user }}" + group: "{{ fulcrum_group }}" + mode: '0644' + +- name: Create Fulcrum configuration file + ansible.builtin.template: + src: fulcrum.conf.j2 + dest: "{{ fulcrum_config_dir }}/fulcrum.conf" + owner: "{{ fulcrum_user }}" + group: "{{ fulcrum_group }}" + mode: '0640' + notify: Restart fulcrum + +- name: Create systemd service file for Fulcrum + ansible.builtin.template: + src: fulcrum.service.j2 + dest: /etc/systemd/system/fulcrum.service + owner: root + group: root + mode: '0644' + notify: Restart fulcrum + +- name: Reload systemd daemon + systemd: + daemon_reload: yes + +- name: Enable and start Fulcrum service + systemd: + name: fulcrum + enabled: yes + state: started + +- name: Wait for Fulcrum to start + wait_for: + port: "{{ fulcrum_tcp_port }}" + host: "{{ fulcrum_tcp_bind }}" + delay: 5 + timeout: 30 + ignore_errors: yes + +- name: Check Fulcrum service status + systemd: + name: fulcrum + register: fulcrum_service_status + changed_when: false + +- name: Display Fulcrum service status + debug: + msg: "Fulcrum service is {{ 'running' if fulcrum_service_status.status.ActiveState == 'active' else 'not running' }}" diff --git a/ansible/roles/fulcrum/templates/banner.txt.j2 b/ansible/roles/fulcrum/templates/banner.txt.j2 new file mode 100644 index 0000000..1ce4ced --- /dev/null +++ b/ansible/roles/fulcrum/templates/banner.txt.j2 @@ -0,0 +1,3 @@ +counterinfra + +PER ASPERA AD ASTRA diff --git a/ansible/roles/fulcrum/templates/fulcrum.conf.j2 b/ansible/roles/fulcrum/templates/fulcrum.conf.j2 new file mode 100644 index 0000000..69f072c --- /dev/null +++ b/ansible/roles/fulcrum/templates/fulcrum.conf.j2 @@ -0,0 +1,29 @@ +# Fulcrum Configuration +# Generated by Ansible + +# Bitcoin Core/Knots RPC settings +bitcoind = {{ bitcoin_rpc_host }}:{{ bitcoin_rpc_port }} +rpcuser = {{ bitcoin_rpc_user }} +rpcpassword = {{ bitcoin_rpc_password }} + +# Fulcrum server general settings +datadir = {{ fulcrum_db_dir }} +tcp = {{ fulcrum_tcp_bind }}:{{ fulcrum_tcp_port }} +peering = {{ 'true' if fulcrum_peering else 'false' }} +zmq_allow_hashtx = {{ 'true' if fulcrum_zmq_allow_hashtx else 'false' }} + +# SSL/TLS Configuration +{% if fulcrum_ssl_enabled | default(false) %} +ssl = {{ fulcrum_ssl_bind }}:{{ fulcrum_ssl_port }} +cert = {{ fulcrum_ssl_cert_path }} +key = {{ fulcrum_ssl_key_path }} +{% endif %} + +# Anonymize client IP addresses and TxIDs in logs +anon_logs = {{ 'true' if fulcrum_anon_logs else 'false' }} + +# Max RocksDB Memory in MiB +db_mem = {{ fulcrum_db_mem_mb }}.0 + +# Banner +banner = {{ fulcrum_lib_dir }}/fulcrum-banner.txt diff --git a/ansible/roles/fulcrum/templates/fulcrum.service.j2 b/ansible/roles/fulcrum/templates/fulcrum.service.j2 new file mode 100644 index 0000000..02f4aa5 --- /dev/null +++ b/ansible/roles/fulcrum/templates/fulcrum.service.j2 @@ -0,0 +1,24 @@ +# MiniBolt: systemd unit for Fulcrum +# /etc/systemd/system/fulcrum.service + +[Unit] +Description=Fulcrum +After=network.target + +StartLimitBurst=2 +StartLimitIntervalSec=20 + +[Service] +ExecStart={{ fulcrum_binary_path }} {{ fulcrum_config_dir }}/fulcrum.conf + +User={{ fulcrum_user }} +Group={{ fulcrum_group }} + +# Process management +#################### +Type=simple +KillSignal=SIGINT +TimeoutStopSec=300 + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/fulcrum/templates/healthcheck.service.j2 b/ansible/roles/fulcrum/templates/healthcheck.service.j2 new file mode 100644 index 0000000..27995f4 --- /dev/null +++ b/ansible/roles/fulcrum/templates/healthcheck.service.j2 @@ -0,0 +1,14 @@ +[Unit] +Description=Fulcrum Health Check +After=network.target fulcrum.service + +[Service] +Type=oneshot +User=root +ExecStart=/usr/local/bin/fulcrum-healthcheck-push.sh +Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/fulcrum/templates/healthcheck.sh.j2 b/ansible/roles/fulcrum/templates/healthcheck.sh.j2 new file mode 100644 index 0000000..ee1c7da --- /dev/null +++ b/ansible/roles/fulcrum/templates/healthcheck.sh.j2 @@ -0,0 +1,34 @@ +#!/bin/bash +# Fulcrum health check — managed by Ansible (roles/fulcrum) +# +# Checks that Fulcrum's Electrum TCP port is accepting connections, and records +# the answer in the exit code, which systemd keeps: +# systemctl is-failed fulcrum-healthcheck.service +# That is a complete answer with no monitoring system involved. Reporting +# elsewhere is optional and generic — set healthcheck_push_url. + +FULCRUM_HOST="{{ fulcrum_tcp_bind }}" +FULCRUM_PORT={{ fulcrum_tcp_port }} +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" + +check_fulcrum() { + timeout 5 bash -c "echo > /dev/tcp/${FULCRUM_HOST}/${FULCRUM_PORT}" 2>/dev/null +} + +report() { + local status=$1 msg=$2 + # No push URL is normal, not an error: the exit code below is still a + # complete answer for anything reading unit state. + [ -n "$PUSH_URL" ] || return 0 + curl -s --max-time 10 --retry 2 -o /dev/null \ + "${PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true +} + +if check_fulcrum; then + report "up" "OK" + exit 0 +else + echo "Fulcrum TCP port ${FULCRUM_PORT} not responding" + report "down" "Fulcrum TCP port not responding" + exit 1 +fi diff --git a/ansible/roles/fulcrum/templates/healthcheck.timer.j2 b/ansible/roles/fulcrum/templates/healthcheck.timer.j2 new file mode 100644 index 0000000..df16ff3 --- /dev/null +++ b/ansible/roles/fulcrum/templates/healthcheck.timer.j2 @@ -0,0 +1,17 @@ +[Unit] +Description=Fulcrum Health Check Timer +# NOTE: this deliberately does NOT carry `Requires=fulcrum.service`, which the +# hand-written unit had. Requires on a timer means the timer is stopped when the +# required unit stops — i.e. "if the thing I am watching goes down, stop +# watching it", which is backwards for a health check and leaves nothing to +# re-arm the timer when the service returns. The live timer had been `active` +# and `enabled` with NextElapseUSecMonotonic=infinity and a last trigger of +# 2026-02-17: seven months with no health check and no outward sign of it. + +[Timer] +OnBootSec=1min +OnUnitActiveSec=1min +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml index 01e9f17..6f5c55f 100644 --- a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml +++ b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml @@ -1,3 +1,7 @@ +--- +# Fulcrum: Electrum server indexing the Bitcoin Knots node. +# The index takes days to rebuild, so nothing here touches its data directory +# beyond asserting that it exists. - name: Deploy Fulcrum Electrum Server hosts: electrum become: yes @@ -5,540 +9,12 @@ - ../../infra_vars.yml - ../../services_config.yml - ../../infra_secrets.yml - - ./fulcrum_vars.yml vars: - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - - name: Calculate 75% of system RAM for db_mem - set_fact: - fulcrum_db_mem_mb: "{{ (ansible_memtotal_mb | float * fulcrum_db_mem_percent) | int }}" - changed_when: false - - - name: Display calculated db_mem value - debug: - msg: "Setting db_mem to {{ fulcrum_db_mem_mb }} MB ({{ (fulcrum_db_mem_percent * 100) | int }}% of {{ ansible_memtotal_mb }} MB total RAM)" - - - name: Display Fulcrum version to install - debug: - msg: "Installing Fulcrum version {{ fulcrum_version }}" - - - name: Install required packages - apt: - name: - - curl - - wget - - openssl - state: present - update_cache: yes - - - name: Create fulcrum group - group: - name: "{{ fulcrum_group }}" - system: yes - state: present - - - name: Create fulcrum user - user: - name: "{{ fulcrum_user }}" - group: "{{ fulcrum_group }}" - system: yes - shell: /usr/sbin/nologin - home: /home/{{ fulcrum_user }} - create_home: yes - state: present - - - name: Create Fulcrum database directory (heavy data on special mount) - file: - path: "{{ fulcrum_db_dir }}" - state: directory - owner: "{{ fulcrum_user }}" - group: "{{ fulcrum_group }}" - mode: '0755' - - - name: Create Fulcrum config directory - file: - path: "{{ fulcrum_config_dir }}" - state: directory - owner: root - group: "{{ fulcrum_group }}" - mode: '0755' - - - name: Create Fulcrum lib directory (for banner and other data files) - file: - path: "{{ fulcrum_lib_dir }}" - state: directory - owner: "{{ fulcrum_user }}" - group: "{{ fulcrum_group }}" - mode: '0755' - - # =========================================== - # SSL Certificate Generation - # =========================================== - - name: Check if SSL certificate already exists - stat: - path: "{{ fulcrum_ssl_cert_path }}" - register: fulcrum_ssl_cert_exists - when: fulcrum_ssl_enabled | default(false) - - - name: Generate self-signed SSL certificate for Fulcrum - command: > - openssl req -x509 -newkey rsa:4096 - -keyout {{ fulcrum_ssl_key_path }} - -out {{ fulcrum_ssl_cert_path }} - -sha256 -days {{ fulcrum_ssl_cert_days }} - -nodes - -subj "/C=XX/ST=Decentralized/L=Bitcoin/O=Fulcrum/OU=Electrum/CN=fulcrum.local" - args: - creates: "{{ fulcrum_ssl_cert_path }}" - when: fulcrum_ssl_enabled | default(false) - notify: Restart fulcrum - - - name: Set SSL certificate permissions - file: - path: "{{ fulcrum_ssl_cert_path }}" - owner: "{{ fulcrum_user }}" - group: "{{ fulcrum_group }}" - mode: '0644' - when: fulcrum_ssl_enabled | default(false) and fulcrum_ssl_cert_exists.stat.exists | default(false) or fulcrum_ssl_enabled | default(false) - - - name: Set SSL key permissions - file: - path: "{{ fulcrum_ssl_key_path }}" - owner: "{{ fulcrum_user }}" - group: "{{ fulcrum_group }}" - mode: '0600' - when: fulcrum_ssl_enabled | default(false) - - - name: Check if Fulcrum binary already exists - stat: - path: "{{ fulcrum_binary_path }}" - register: fulcrum_binary_exists - changed_when: false - - - name: Download Fulcrum binary tarball - get_url: - url: "https://github.com/cculianu/Fulcrum/releases/download/v{{ fulcrum_version }}/Fulcrum-{{ fulcrum_version }}-x86_64-linux.tar.gz" - dest: "/tmp/Fulcrum-{{ fulcrum_version }}-x86_64-linux.tar.gz" - mode: '0644' - when: not fulcrum_binary_exists.stat.exists - - - name: Extract Fulcrum binary - unarchive: - src: "/tmp/Fulcrum-{{ fulcrum_version }}-x86_64-linux.tar.gz" - dest: "/tmp" - remote_src: yes - when: not fulcrum_binary_exists.stat.exists - - - name: Install Fulcrum binary - copy: - src: "/tmp/Fulcrum-{{ fulcrum_version }}-x86_64-linux/Fulcrum" - dest: "{{ fulcrum_binary_path }}" - owner: root - group: root - mode: '0755' - remote_src: yes - when: not fulcrum_binary_exists.stat.exists - - - name: Verify Fulcrum binary installation - command: "{{ fulcrum_binary_path }} --version" - register: fulcrum_version_check - changed_when: false - - - name: Display Fulcrum version - debug: - msg: "{{ fulcrum_version_check.stdout_lines }}" - - - name: Create Fulcrum banner file - copy: - dest: "{{ fulcrum_lib_dir }}/fulcrum-banner.txt" - content: | - counterinfra - - PER ASPERA AD ASTRA - owner: "{{ fulcrum_user }}" - group: "{{ fulcrum_group }}" - mode: '0644' - - - name: Create Fulcrum configuration file - copy: - dest: "{{ fulcrum_config_dir }}/fulcrum.conf" - content: | - # Fulcrum Configuration - # Generated by Ansible - - # Bitcoin Core/Knots RPC settings - bitcoind = {{ bitcoin_rpc_host }}:{{ bitcoin_rpc_port }} - rpcuser = {{ bitcoin_rpc_user }} - rpcpassword = {{ bitcoin_rpc_password }} - - # Fulcrum server general settings - datadir = {{ fulcrum_db_dir }} - tcp = {{ fulcrum_tcp_bind }}:{{ fulcrum_tcp_port }} - peering = {{ 'true' if fulcrum_peering else 'false' }} - zmq_allow_hashtx = {{ 'true' if fulcrum_zmq_allow_hashtx else 'false' }} - - # SSL/TLS Configuration - {% if fulcrum_ssl_enabled | default(false) %} - ssl = {{ fulcrum_ssl_bind }}:{{ fulcrum_ssl_port }} - cert = {{ fulcrum_ssl_cert_path }} - key = {{ fulcrum_ssl_key_path }} - {% endif %} - - # Anonymize client IP addresses and TxIDs in logs - anon_logs = {{ 'true' if fulcrum_anon_logs else 'false' }} - - # Max RocksDB Memory in MiB - db_mem = {{ fulcrum_db_mem_mb }}.0 - - # Banner - banner = {{ fulcrum_lib_dir }}/fulcrum-banner.txt - owner: "{{ fulcrum_user }}" - group: "{{ fulcrum_group }}" - mode: '0640' - notify: Restart fulcrum - - - name: Create systemd service file for Fulcrum - copy: - dest: /etc/systemd/system/fulcrum.service - content: | - # MiniBolt: systemd unit for Fulcrum - # /etc/systemd/system/fulcrum.service - - [Unit] - Description=Fulcrum - After=network.target - - StartLimitBurst=2 - StartLimitIntervalSec=20 - - [Service] - ExecStart={{ fulcrum_binary_path }} {{ fulcrum_config_dir }}/fulcrum.conf - - User={{ fulcrum_user }} - Group={{ fulcrum_group }} - - # Process management - #################### - Type=simple - KillSignal=SIGINT - TimeoutStopSec=300 - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - notify: Restart fulcrum - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start Fulcrum service - systemd: - name: fulcrum - enabled: yes - state: started - - - name: Wait for Fulcrum to start - wait_for: - port: "{{ fulcrum_tcp_port }}" - host: "{{ fulcrum_tcp_bind }}" - delay: 5 - timeout: 30 - ignore_errors: yes - - - name: Check Fulcrum service status - systemd: - name: fulcrum - register: fulcrum_service_status - changed_when: false - - - name: Display Fulcrum service status - debug: - msg: "Fulcrum service is {{ 'running' if fulcrum_service_status.status.ActiveState == 'active' else 'not running' }}" - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create Fulcrum health check and push script - when: uptime_kuma_enabled | default(false) - copy: - dest: /usr/local/bin/fulcrum-healthcheck-push.sh - content: | - #!/bin/bash - # - # Fulcrum Health Check and Push to Uptime Kuma - # Checks if Fulcrum TCP port is responding and pushes status to Uptime Kuma - # - - FULCRUM_HOST="{{ fulcrum_tcp_bind }}" - FULCRUM_PORT={{ fulcrum_tcp_port }} - UPTIME_KUMA_PUSH_URL="${UPTIME_KUMA_PUSH_URL}" - - # Check if Fulcrum TCP port is responding - check_fulcrum() { - # Try to connect to TCP port - timeout 5 bash -c "echo > /dev/tcp/${FULCRUM_HOST}/${FULCRUM_PORT}" 2>/dev/null - return $? - } - - # Push status to Uptime Kuma - push_to_uptime_kuma() { - local status=$1 - local msg=$2 - - if [ -z "$UPTIME_KUMA_PUSH_URL" ]; then - echo "ERROR: UPTIME_KUMA_PUSH_URL not set" - return 1 - fi - - # URL encode spaces in message - local encoded_msg="${msg// /%20}" - - if ! curl -s --max-time 10 --retry 2 -o /dev/null \ - "${UPTIME_KUMA_PUSH_URL}?status=${status}&msg=${encoded_msg}&ping="; then - echo "ERROR: Failed to push to Uptime Kuma" - return 1 - fi - } - - # Main health check - if check_fulcrum; then - push_to_uptime_kuma "up" "OK" - exit 0 - else - push_to_uptime_kuma "down" "Fulcrum TCP port not responding" - exit 1 - fi - owner: root - group: root - mode: '0755' - - - name: Create systemd timer for Fulcrum health check - copy: - dest: /etc/systemd/system/fulcrum-healthcheck.timer - content: | - [Unit] - Description=Fulcrum Health Check Timer - Requires=fulcrum.service - - [Timer] - OnBootSec=1min - OnUnitActiveSec=1min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Create systemd service for Fulcrum health check - when: uptime_kuma_enabled | default(false) - copy: - dest: /etc/systemd/system/fulcrum-healthcheck.service - content: | - [Unit] - Description=Fulcrum Health Check and Push to Uptime Kuma - After=network.target fulcrum.service - - [Service] - Type=oneshot - User=root - ExecStart=/usr/local/bin/fulcrum-healthcheck-push.sh - Environment=UPTIME_KUMA_PUSH_URL= - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon for health check - systemd: - daemon_reload: yes - - - name: Enable and start Fulcrum health check timer - systemd: - name: fulcrum-healthcheck.timer - enabled: yes - state: started - - - name: Create Uptime Kuma push monitor setup script for Fulcrum - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_fulcrum_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - # Load configs - with open('/tmp/ansible_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_name = config['monitor_name'] - - # Connect to Uptime Kuma - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - push_token = existing_monitor.get('pushToken') or existing_monitor.get('push_token') - if not push_token: - raise ValueError("Could not find push token for monitor") - push_url = f"{url}/api/push/{push_token}" - print(f"Push URL: {push_url}") - else: - print(f"Creating push monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.PUSH, - name=monitor_name, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - monitors = api.get_monitors() - new_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - if new_monitor: - push_token = new_monitor.get('pushToken') or new_monitor.get('push_token') - if not push_token: - raise ValueError("Could not find push token for new monitor") - push_url = f"{url}/api/push/{push_token}" - print(f"Push URL: {push_url}") - - api.disconnect() - print("SUCCESS") - - except Exception as e: - error_msg = str(e) if str(e) else repr(e) - print(f"ERROR: {error_msg}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_name: "Fulcrum" - mode: '0644' - - - name: Run Uptime Kuma push monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_fulcrum_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Extract push URL from monitor setup output - set_fact: - uptime_kuma_push_url: "{{ monitor_setup.stdout | regex_search('Push URL: (https?://[^\\s]+)', '\\1') | first | default('') }}" - delegate_to: localhost - become: no - when: monitor_setup.stdout is defined - - - name: Display extracted push URL - debug: - msg: "Uptime Kuma Push URL: {{ uptime_kuma_push_url }}" - when: uptime_kuma_push_url | default('') != '' - - - name: Set push URL in systemd service environment - lineinfile: - path: /etc/systemd/system/fulcrum-healthcheck.service - regexp: '^Environment=UPTIME_KUMA_PUSH_URL=' - line: "Environment=UPTIME_KUMA_PUSH_URL={{ uptime_kuma_push_url }}" - state: present - insertafter: '^\[Service\]' - when: uptime_kuma_push_url | default('') != '' - - - name: Reload systemd daemon after push URL update - systemd: - daemon_reload: yes - when: uptime_kuma_push_url | default('') != '' - - - name: Restart health check timer to pick up new environment - systemd: - name: fulcrum-healthcheck.timer - state: restarted - when: uptime_kuma_push_url | default('') != '' - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_fulcrum_monitor.py - - /tmp/ansible_config.yml - - /tmp/Fulcrum-{{ fulcrum_version }}-x86_64-linux.tar.gz - - /tmp/Fulcrum-{{ fulcrum_version }}-x86_64-linux - - handlers: - - name: Restart fulcrum - when: uptime_kuma_enabled | default(false) - systemd: - name: fulcrum - state: restarted - + # Preserves the push URL this check has been configured with. The role knows + # nothing about Uptime Kuma — this is just "a URL that accepts a ping". + healthcheck_push_url: "{{ healthcheck_push_urls.fulcrum | default('') }}" + roles: + - fulcrum - name: Setup public Fulcrum SSL forwarding on the edge host hosts: edge @@ -546,11 +22,6 @@ vars_files: - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - - ./fulcrum_vars.yml - vars: - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - tasks: - name: Expose Fulcrum SSL through a socket proxy ansible.builtin.include_role: @@ -558,128 +29,5 @@ vars: socket_proxy_name: fulcrum-ssl socket_proxy_description: "Fulcrum SSL" - socket_proxy_listen_port: "{{ fulcrum_ssl_port }}" - socket_proxy_upstream_host: "{{ fulcrum_tailscale_hostname }}" - - - name: Display public endpoint - when: uptime_kuma_enabled | default(false) - debug: - msg: "Fulcrum SSL public endpoint: {{ ansible_host }}:{{ fulcrum_ssl_port }}" - - # =========================================== - # Uptime Kuma TCP Monitor for Public SSL Port - # =========================================== - - name: Create Uptime Kuma TCP monitor setup script for Fulcrum SSL - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_fulcrum_ssl_tcp_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_fulcrum_ssl_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_host = config['monitor_host'] - monitor_port = config['monitor_port'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - api.add_monitor(type='group', name='services') - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing = next((m for m in monitors if m.get('name') == monitor_name), None) - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing: - print(f"Monitor '{monitor_name}' already exists (ID: {existing['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating TCP monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.PORT, - name=monitor_name, - hostname=monitor_host, - port=monitor_port, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for TCP monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_fulcrum_ssl_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_host: "{{ ansible_host }}" - monitor_port: {{ fulcrum_ssl_port }} - monitor_name: "Fulcrum SSL Public" - mode: '0644' - - - name: Run Uptime Kuma TCP monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_fulcrum_ssl_tcp_monitor.py - delegate_to: localhost - become: no - register: tcp_monitor_setup - changed_when: "'SUCCESS' in tcp_monitor_setup.stdout" - ignore_errors: yes - - - name: Display TCP monitor setup output - debug: - msg: "{{ tcp_monitor_setup.stdout_lines }}" - when: tcp_monitor_setup.stdout is defined - - - name: Clean up TCP monitor temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_fulcrum_ssl_tcp_monitor.py - - /tmp/ansible_fulcrum_ssl_config.yml - - + socket_proxy_listen_port: "{{ service_settings.fulcrum.ssl_port }}" + socket_proxy_upstream_host: "{{ service_settings.fulcrum.tailscale_hostname }}" diff --git a/ansible/services_config.yml b/ansible/services_config.yml index 5014297..d9a9c13 100644 --- a/ansible/services_config.yml +++ b/ansible/services_config.yml @@ -46,3 +46,9 @@ service_settings: # from the edge host. A role default cannot serve the second play, so it # lives here rather than in roles/mempool/defaults. frontend_port: 8080 + fulcrum: + # Same shape as mempool: the fulcrum role deploys on fulcrum-box, and the + # socket-proxy play publishes the SSL port from the edge host. A role default + # is invisible to that second play. + ssl_port: 50002 + tailscale_hostname: fulcrum-box From d26dc78b3c8c35aa8fd7047d826228bc4fd4fb65 Mon Sep 17 00:00:00 2001 From: counterweight Date: Sun, 13 Sep 2026 18:13:40 +0200 Subject: [PATCH 52/67] bitcoin-knots: convert to a role, de-Uptime-Kuma the health check MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 892-line playbook becomes 40 lines plus a role with install/build/configure/ service/healthcheck phases, five templates and one handler. bitcoin_knots_vars.yml is deleted; its content is the role's defaults. Verified after a real run: bitcoind still active since 2026-08-19 (NO restart), chain at 966844 blocks / 875 GB, DATUM config intact, dbcache still 200, health check timer firing again. changed=3, all health-check. vipy changed=0. ⚠ THE BIG ONE: the playbook would have deleted the mining integration. bitcoin.conf on the node carries a section that was hand-added and was missing from the template entirely: blockmaxsize=3985000 blockmaxweight=3985000 blocknotify=killall -USR1 datum_gateway maxmempool=1000 blockreconstructionextratxn=1000000 blocknotify is how datum_gateway learns a new block landed. Running the old playbook would have stripped all of it and solo mining would have carried on against a stale template - a silent failure that costs money rather than raising an error. Also dbcache 200 -> 3528 (hand-tuned down; the calculation wants 90% of RAM) and logging moved off the file. All now reconciled, dbcache behind bitcoin_dbcache_mb_override. bitcoin-knots and datum-gateway are ONE SYSTEM. Noted in the README. AND MY OWN FIX MADE IT MORE DANGEROUS. The `Restart bitcoind` handler was guarded by uptime_kuma_enabled, so it had been inert: bitcoin.conf and the systemd unit both notify it and neither could restart anything - a config change applied to disk, reported success, and never took effect. Ungating that is right, but it converts "wrong config sitting inertly on disk" into "node restarted onto a config that breaks mining". The ungating had to land WITH the template reconciliation, not before it. It also raises the bar permanently: any residual template/live difference now restarts a Bitcoin node on every run. Four rounds of --check --diff to reach changed=0 - the DATUM section, an explanatory comment that was rendering into the deployed config (now a {# #} Jinja comment), a "# Pruning (optional)" comment the live file had, and one trailing blank line. The build path is 32 tasks all guarded by `not bitcoind_binary_exists.stat.exists`, so a converged host skips the 30-60 minute compile and both `state: absent` deletions. Those target /opt/bitcoin-knots/{source,bitcoin-}; the chain is in /mnt/knots_data and is never touched. Signature-verification tasks copied verbatim. The health check timer had last fired 2026-08-09 while reporting active/enabled - same OnBootSec + OnUnitActiveSec dead chain as fulcrum. The role runs the check once after enabling to supply the reference the timer schedules from. Ownership parity checked mechanically against `git show HEAD:`, keyed by TASK NAME rather than path - keying by path gave a false positive, because bitcoin_knots_source_dir is created with ownership and later removed with state: absent, so whichever task comes last wins and that differs between one file and five. 13/13 match. bitcoin_p2p_port and the tailscale hostname moved to services_config.yml for the socket-proxy play on the edge host. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 350 +++---- ansible/infra_secrets.yml | 350 +++---- ansible/roles/bitcoin_knots/README.md | 85 ++ ansible/roles/bitcoin_knots/defaults/main.yml | 71 ++ ansible/roles/bitcoin_knots/handlers/main.yml | 14 + ansible/roles/bitcoin_knots/tasks/build.yml | 222 +++++ .../roles/bitcoin_knots/tasks/configure.yml | 20 + .../roles/bitcoin_knots/tasks/healthcheck.yml | 56 ++ ansible/roles/bitcoin_knots/tasks/install.yml | 104 +++ ansible/roles/bitcoin_knots/tasks/main.yml | 8 + ansible/roles/bitcoin_knots/tasks/service.yml | 37 + .../bitcoin_knots/templates/bitcoin.conf.j2 | 67 ++ .../templates/bitcoind.service.j2 | 17 + .../templates/healthcheck.service.j2 | 14 + .../bitcoin_knots/templates/healthcheck.sh.j2 | 62 ++ .../templates/healthcheck.timer.j2 | 11 + .../bitcoin-knots/bitcoin_knots_vars.yml | 38 - .../deploy_bitcoin_knots_playbook.yml | 881 +----------------- ansible/services_config.yml | 6 + 19 files changed, 1163 insertions(+), 1250 deletions(-) create mode 100644 ansible/roles/bitcoin_knots/README.md create mode 100644 ansible/roles/bitcoin_knots/defaults/main.yml create mode 100644 ansible/roles/bitcoin_knots/handlers/main.yml create mode 100644 ansible/roles/bitcoin_knots/tasks/build.yml create mode 100644 ansible/roles/bitcoin_knots/tasks/configure.yml create mode 100644 ansible/roles/bitcoin_knots/tasks/healthcheck.yml create mode 100644 ansible/roles/bitcoin_knots/tasks/install.yml create mode 100644 ansible/roles/bitcoin_knots/tasks/main.yml create mode 100644 ansible/roles/bitcoin_knots/tasks/service.yml create mode 100644 ansible/roles/bitcoin_knots/templates/bitcoin.conf.j2 create mode 100644 ansible/roles/bitcoin_knots/templates/bitcoind.service.j2 create mode 100644 ansible/roles/bitcoin_knots/templates/healthcheck.service.j2 create mode 100644 ansible/roles/bitcoin_knots/templates/healthcheck.sh.j2 create mode 100644 ansible/roles/bitcoin_knots/templates/healthcheck.timer.j2 delete mode 100644 ansible/services/bitcoin-knots/bitcoin_knots_vars.yml diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index 6ac207c..a0e560a 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,174 +1,178 @@ $ANSIBLE_VAULT;1.1;AES256 -38656563383931366464306463373265623631353331376532333134313463303635656162373437 -3735623062343764316262613966353338326535313161360a323831623936373966633535396632 -63303961636566646338373464336637323830663932653039306362653832306163313938396135 -3633663532373362390a333233366636666437626365373732633361653833363264623138386661 -35636665666230326164343339633834616134666531343839623134343163333864333834303862 -37313864303335326133343666333733363333663336356234633332323262616530356661316363 -62306338303966306632303531613161386135346439333137393062343938303134656436656539 -64326134363664353165616436353635633265643161383335393633383563656231336139346138 -38366238653236623664313662626566373631633136613864373032316539646332363035343865 -32666664363737373962396162313438303030613264366232373030316335366534666531646239 -62306630316432346131383738373764313263363039376435653062313136356531383534343831 -37336536313564323336613639386138303562666437376238376630623665373230653232396261 -34333433343361643065643032366433643137396231343331326362626434643365356131613766 -34613039306130653535653330623333333631653538643536616530373538386332346132363739 -65653135306663336163336263376332326230616238666365653663663462306238366466663366 -31633833333465653534613863643465323631336661393366356363386462623434636631623738 -30376265663739333336393664626539353531653637303464316562613436373739353039653430 -36353034643839323131353033643866303333306435653532656239653061656235393536306463 -31626162393462356238323737313933633465623566646161363964393164393762626531326138 -61343336373065313232343064646463346638613661393263326265366539623861623537643164 -32663365313161316464333137383337656662323137356635636637326163343562323965666166 -30396531326364656461613232653361313061623835663663643861306334386539626530386666 -63306331313466623163366332326564353639383961656362633435363065313830613764306237 -37303530643266646464653263306432313766633835343739636261623464366536346665636135 -31333638306237623535623439613636363937333430633831306162333466323137323932663432 -35616135363134306231323933303635336332376163303131656131376131393465353434353330 -38623339313964313730663263373963306337373938393863323062366534613131623764386638 -65616463376633663864653363643135623663316439346336376366346166303962633366636565 -65373335623962613131336364396133623738613364383639356533316138383333353464616639 -62316161656633666431356462313634383239363464613031313232393131643936333237616364 -36646530646239626662386231313239396238323139383037653337376134316535373464613235 -35346439646234623462623262336461326434346632373966626464313633343266616463643764 -37633830653230666631383134313738326265343738386631346261343439356163353262626161 -37343134663764396439633535616537386638633731643164383064333266643830356563383066 -34356161623361623366363435653739383438396363643338353736323139313661373031396132 -61373834303439663939376339633832313639643332383239313337666433623435613161306139 -62633162346338663335663661643938316539316139356165346461326433646366306134356338 -35613934623666386463356536303362626466663062643236346163323337616232323535396265 -36636165656366396136666533626433363562353537343430333361643635313335336435626265 -36663233383265636530313332336562623833373930626136316265393634653466333732666563 -36383037336566373566313661363062346431343534396533656135326534646161346639383336 -35306633303836346365383533316661613630326533393836353036636132636663653530656336 -35656364613130633761666632333633333137656637353362353337626266616238336266306636 -61393932343736306465666338626365613531386362376565383738366230333465663137343363 -64336661343563643136626362303937653632316230356531323231653063306538616633616265 -64306332396639323461653136356662376236363861643466633466666265646538653833373138 -38623130633037333666623862636165666333366335643765613834383533343436626638356536 -32333939326532396339666637386237303730363332643861316236613634353230353565316239 -30356135323237633963343461376233656333633636303662333365333735383930656538356238 -31373862613534386464333865653361363663373664663234346636316262356137613037353135 -37656534383164363131613834633062656437373066646336666533313033643130636361383038 -61336366343230303036336563663537623230643061383732623865383134366535353365346363 -33373331323831396462633031353665363033346536306334366237646363636633353732626166 -66363938663861616461613562646530656366333166303264363030306139343161313938613361 -66376263326131626132646530643439633634303539613863623965373432363863326130616536 -64626336633238613034353166353664363230393732366465346363303537336161623834333265 -61346365393964393365656565323331303265616630613766656239656438316539636632373038 -62313139386233633732643765383039646534326236333134613531313437346161343638376432 -66336432363739633231303138383338313164313930373866366431316638353761323531333930 -31353136326630363038626135383939393639306466383832373630623565383766663736363937 -38656132303738333863623262656131396366313663666461666231626364646433613137333062 -32336661373236666235306534616535323061663634383763386339353732646636306563323632 -34313433653137613930363631373666613330363761333639373630306537343633643561366231 -38363833363139666631333431646262656239316138373339326533623437383336353139646431 -31313833363733653534333133633636616636343635393039666233383661666363373263313631 -34363064353135383261633666316462356231616163373634383730386536613462346535383331 -61336632643363306435323866346665626464366163343735616130613335663235376132396663 -35363635663664626638663331366131646333623533613065323731643366656562353138363335 -34653863323337653331306232306133323666356464323932323562376632333439303537306534 -66346462363535353961383061323762343237393535626638633365396566616636616236396462 -38353433393631393338646338383830663331663538336465366161653333373736623063666437 -34633365383736653230316631636666386664643731323434613761613139666432636236653137 -30393666663261616332323466353161636632643534386131306632313236653063633531356564 -66353964393634336530396162393862653837376436623738663135303632383166613333336165 -62333434613432626264653865353035643563663435316233376465383961303534646339373632 -31343530353636343238623565653566663936633763323861383231366533303164316339653637 -36393364616365666335386539653439653735323162396131643931333866613862633437333438 -34663233356461383537343738343134626532323361653666633733353939333065363131316133 -63616632336130653962636236336264363565303839653531326264386331316264623038326266 -62336233303332333638613461366231343935323036366536383639373837616638616362643331 -35663331653133343661346635613036643065303637613964666237353166386666313439383936 -31663730356636333061326535316638366531656430353633383663336535663830323063356666 -62333762393865333132653161346361313832643163356265616133373663363836303662613935 -63623662313830623434303530303833663063336164363836643531323239653966366233373266 -35373037323132653833646432356435333834346138636133633339613165303362336265303065 -31343931386136323362376431336164343931343934643031336435303738393764613332393561 -31363333623165643363646638336435633539626635656536323030646130323734643134336562 -62353233366533363131313462656239366337343532336332323563343931376237306231656532 -63326439303930326563323238613538386435393735356438316536373064643465306362346135 -63646532343334323538343838653764303332386139313239376332656664316361343536663766 -61363065343862373165366563366363376362303862633037373763663132663631346562313161 -62636635303738393763373232643862386135386537333637636631626638363232323036316366 -38383235366333363264653163313937333566633066356663383466393535333535313939326266 -30626531386333626334366533613366353433396438313838346339396433666436663961643438 -62633336633066616163626232326334633265356532663535323264373166316133386466653131 -66623261333964623235386433346236353338363961323230356633613864303735386530303165 -30646265346566626562643330333233636164386432613436383034623337323531613735656133 -64653232343839653661303662336435653439663732303638323732333064646333313732356339 -31303934613938326365656161383262363633373265616466616562313237363763613265376664 -38636534646237616332653531323036313430643862393362383630653339343931376164373735 -39323136653730316462643964643939396662333462653934623136643839373864613335303463 -61643765656431363734616435343564393663346535636134383637653538646336666164336339 -31353230656364633936643338613563306639316464653864396230626263306338323639323238 -35396334613039343734303232363962653164636561633164303132366234376632653339376538 -37366638376666353733633132636131653462613732656364316530306333313431316164663839 -34646366393835353338373763333666646535363634633864306433316135376663343833613435 -37356636346234623735663361336366363264646535336566363933653832613039663161663762 -65663234666437363834646163643733356136396464633830336533366337303438313665336166 -33376331373032316464353037383433623334643965656434636134333166323238666438653337 -32306234636661656639306464653238326536376661363036626561386630326239616137376565 -32666438356161336437646562373534326632306165666439666130313036373262633164376362 -66646639663663323134626636313130343362343966653564306265623630326664336534333037 -65626562666337653436373130613230656535386264303132373634353861376238386435366636 -38316637613036333836356536306132626631326562363535613835313432353133373138613264 -62396261303239396364613039346135623437336565663034313264643535366336653363336330 -31623539396435643132383363353861323064336430653936623438303933356562613335643861 -39663437333836383331663830626433386431383338373261353266386531373931383632306135 -36316339366539333730656638386635333733363764613364303563323665396336613930346233 -34333433323030643262373636663838343131303637613237376463386662303235623338393063 -66313166343730383961363765663463303266383938653638393830613662643362386361626132 -66306533313461663131613565343534303735366232383164333837363330626534643363626164 -33326134376263333530663631383930623066313035373135616665323564323033323639623235 -35383538356434663035306137366437613537373531316436336366646165346330383538386331 -64666232343961343261393339356531363636303165393532313632613363316665633233316330 -35623939613463333164643736313531666531626232386361663435653938343536363236616431 -61343430363861626339336166316135663234366132313762616230393239663931656536306663 -31343366663063396139336237373361333030353333623064336262313465316538623438313265 -62656232613435643437373233396566623537353038316262643239653432373863626530323334 -35633132353463356634633237386134346532353930343339303337643637373932666665313561 -39346531383663663131623863383832346237356231386263656566653631363836323132333135 -37316135613730386362666564643037653366373464323431303332306364373432646338333031 -31613533633664343038376533306235393030646236346232386236653138636234666561646263 -31386633343265663335323236346435653261363763366234346263626663643964393962343766 -63656237326563653633383461333364396165626138343461353963656131663265313835326465 -65643164316537333531363139623732376364346363613363663066663733616665346233633662 -35343065316437373136656235323466343732613335393761313061353432623664363462383562 -38333139636336366462326565383631633864633165613332343833396237336561346164656664 -63316463323233383434663061633564666264313261353632316164643035306234393732333435 -66623663316334373431633734323232313436366364623037303965366663386237343439613433 -65383763393061633431303736656238396364343765363463333433313836386666613964623634 -30643065626437653966393237326433373936653861306536363834623363613830343036363766 -64386234653331363564373065343139623965646562373933333162386264353832303164336362 -33613036666139396533333862316366396464646465346263623730313336663935356139663762 -65623466323239383732636165333564356534333939316539663366363065623631303933663863 -31393637643761303731646638323133636464623366363032656335643435373064666463366365 -65653831666262616165306537623534653763636633633137653733323234346266363630306434 -64616633326266636134343536363365393033663934663965323537633136356535353436316634 -61643636303237363033313637396533303761303536373836343332636466396339353065386539 -38366530623332316439393633356235386665636364313739643432623032303363613666656264 -31353063633862366638363838663932386131366434646462313130366633643430373230313338 -33383362313230373137393934323561343064623065353063653535366266373237656530303335 -38616263613232353632333361383030626133323261323033396638666231396638353537383338 -33393466663938356363623438366265336330313039313666343462356331626265383565623364 -66323835303365366333623733343664626431343663623465326235363430356661616166346630 -30336266383830336461663130616363663637653964366634663930393066633134626530653766 -34393439363332383831383865393464656165643331333661643138663133643133626236333363 -34643438353037346334316233653034313737616134653762353239636234383233336338633031 -32386332656530313361393532623661393538633463393861636539393666626362653031636566 -34616265353736633539313032346334313339313934323962313463376664373336666236346633 -31306563633966653134343937323839316137356164373338343965643934636631646666386361 -62363165663838626538363966396161363366303162356664313437613034636531356638333237 -35613833366462653834363261653365356336643635363831383930323566623262353865653233 -63333138633466653165666236356430666532353639326261613462636161636634613235353165 -38316566306533373138336334643465616339383739653639303632303661656135313039303632 -66613966633036653461396330663630396663623732346261653265343233393730306537363032 -66653737306431343435316433646164303338353865653336303731636663303863363662333863 -33323334656631326137646563616236393035336539646563643064663564383736343961656435 -36306136353037616634383866303636383531616136633230346538336563656366393466613262 -32663266323333393764616561316636356664346265353433653262313239326264306434383030 -34356263323331313364653966623630326433343863643839313165313063626261646339323837 -64613838353334613661346662346636313432393837386139386136613366353038343962366639 -32626265313764613733313334353432346434323439326637383462313863383165303963353665 -34313635623766313361633030616439343433353735326433383563656435393563 +35333539613236336636383331373761643732663835346539653531336662636432346132636539 +3738343939626363303936646239613531306461646431630a656362666266356534303466323962 +64626235663062393436356165323364333735396530343435373730613765346466386336653030 +6234333131383036330a663039613363656536353164323265383130333463646638396437356332 +65653330383235386563623231646535343461613833333765333336366338376264666139303635 +65666139643363356264353234343736623731656534636662616466313764633763303035343432 +32623965326532353733656533356566393834343338303833356334346533383531393435363461 +64306561663434353935386165393361613330643163323564383835386438346664323663663564 +35623032646362346262366333626461343364356136373231313635353665643838666336313766 +62663730653030326534393838343139313239656530316662323638396230323436396431306135 +64386263616633323061373331376536343261313330653061333662373034353564623934376235 +65363432316630636132613935663466366365613265326163383832613631303631616464633161 +64373661313034653338646537343164623630643039353533376537636264653261393464633765 +38376436366262373663303137343861383932356330356239326464666233303064666634623135 +66373430383061643463386265646666333962366366336534646635633365356334343238303930 +65316432386134353838333033386164643165366662393365623839616265646538333235343237 +63373236623963393937613333623630363032643365613561653638323030373435353639366433 +34326334643330336564323439656630313331623762373038343133613763643838643039656338 +61323933643764306339383733623939346466303636616339396534643434633936393563373934 +66646632336133386163613135363963363338373338323732333563643164313662633639633130 +32383263396666333639626366316630346333643039666431323164356364346566343237316166 +62343464373564303533653864653164626239383737303235353638346636356538383661643931 +38646538323431383638653532303163313032396164353139333461363066393566303237386239 +37613236376636316136346266613530363732353232313037383830313461393136306233363931 +35663963613630616530383865633330636434626230333563313362623265633637333261653632 +63303130323737633365646265393736613136383536383133663538663466643631626139663865 +34333938363430393763653937653561376362353633303962643637666162666538316366343132 +31646163626465643463326333303365313138626366363861663530343362366663323161656362 +31623361323033353238623462333031356265343239613834353864333365626636316531313238 +66343633393865316535316262376338346234633737366132313938356438383066356465303239 +64326261333935343337363331373366636438623966373439653432663030376236643664643865 +34393738343865336263666133613765353162353839393633346439356665613936643863323464 +64633462366565373532333336376666336263396362303639393061346438363035613534356334 +37376166363030393237373464356139343336333939346662336332336639396432346665373137 +63306539383830656363383539303366643335356133353738373266646463346265333733613030 +63383036343663623366653763636135326561663739303631333936313161396335373962643662 +31386331623466666262396132636265323831646333353038386131303033626435316634666662 +36353539383961623535333730653537633932613366653563323339333738643131393338313931 +33373964366638323239386130306666316333613233666337623966303830323637306534653430 +62383634303736636166376563626537346161663432616265336364336264373638306364336331 +35373862653133333637636164343030623664333536656433366563376330356431343962656162 +31646536626631666636383033336130373565376430386639313135373765383437336538313266 +65383038616139663436323833613232626531326531303937613931373566646263643634343461 +63393365333839663231623532373634643136383333373166356635356666353837383331383334 +36636465316536313765326562336539663539613036373638616633353936383866663231356262 +65313965643263666163313638626335363333623833656439633464343864333465313231326164 +30646530363161633834306633643132306562363065323032383066316533373963383763333466 +33613230643633346566396165376665366361633733366261306637303964376231303365333165 +37366633643331393537633563626164613630396663326233383263343930333232656265326633 +62373539306331623332386637646333326435393933323632313166386365656561356536663738 +35363362663937363661636437336532646437623864643463346238636331643935333264663365 +35353063643662363939396638386531386265336566373835646435353736386666646531373361 +30646136386138323530306135666333386437356430643262363234616366376335633638303133 +62316562323463346263323934363937336631656666306237626438626133346566613662363831 +65623561373231663262313763623965663036376631663662616634353664663762386666353539 +63323237303362393762343832396463343534633432626532363534396261613132323633353938 +64613333316436313063313561656561326139623736373439363363393632353564343361396666 +32623938383737623836393536393838346131343762353463386339336361346266623663353262 +63316362363736376639333638326433383662663866303638376662616335653764663631616666 +66663738636162353262373034373334303562363236373232306364393335346139396663626634 +64373735643537383230616661323238333239386330313231333833663062393832396366373337 +66313966356261383039623630376138643261393062393030346131663839346437363766333732 +63303136303037643931663431363435353261343133386332666531383835663361376165383632 +38333963653732613462323435633936353637336538396531616437393131333631663335613931 +34343236386135316336353661636266396434386563656433633935336330366162363563313738 +39303561323739306362383765303131623265626332613265323264393333356165326238633031 +33366464626537313662333166343532373735303135306563393737663536363862336232613435 +32613031396562333939303566343834323164396165373165363932323065623839643035373539 +33363362316162643264623937353066303536623962373433333430373436643862616636613537 +66653864646463653435366531373039373663333964636163633965613438366339613437613731 +34353330616337313037353633626633633666346363663764316163653635363335343237616666 +34663461616539376463613635346536366334613536326564626363663661313038636562303031 +36653837393233333234613131633735313739333263663532623563623231323239343936333831 +64313661376164323738316136623261353538323565383865376339366466393033373065613835 +34393339323839653430653335623664373534356232386262666437393234616362353438386361 +65383362346239626438346135663064643031386335633533646433326631386439353731653663 +33326633653363313232356233373531633737383530396439643139316164626566326164333439 +33393163376162623938383264653337633361643633306666646163633337366264353939353236 +39383035653036613463643633633234623439626365633761643038656339373362333238373033 +36626563373563363139386336653764623661633766343965346634653135306137393361386539 +33656261333137656632353739346462326364386330333536636634666538643461373766316165 +31383238353730356339353934663037373865363238373339346238623364383631313537386231 +66643937316361333262643230343965616663393430303661623161323361643835623862313564 +30646631396564383932633437303464666166636434303330303934613038393130666561343837 +34376164323633336432333664366631326239383463303366363763383062643565643464633634 +66313264666562626339326133623630366131333232383333333961323234366633313231376134 +62653962303630366330303537383639306263316630313862646334393531363439346233313462 +34353937303230373931616339393861323138333237633861326165346631323931633234346630 +35656666623261616564343733343764383032373733306466383763376564663665663731303539 +37373863613135326634333039333932633461613165623938623338623565313332376136343436 +36303037653935383630376362326335653466343233613833666361633661626437663563303331 +32363331663334616338643239356133396662633231396436636630343934306333343539646132 +65366263646563386563643562316337343130643661633730663664663564316535613531353165 +32666436326530323961663736306162306633393366306532613266323633373330383634656535 +38626138633335393266656638613765666665313561363231636261643936333033383533373366 +61626365636635653834306539303530383330346630363766383734346332326638316561303763 +63646639386366616230333263313130346663343663623536376664303364623064373562326466 +38383163326637633937663161393538616339326263306463353530303630333937326138363239 +62633535356230353439623234316562653237613130383832343735643661623033316264653362 +37633065313135656637616137653963623331323863363364663761663364663565623535343366 +65313164643461383436623038316466396632626661373438643533336234323132613866393134 +32366363613338646131663738643236613562663262333936656461323336303139616434333334 +38653133633232343164313661623466643061383161303564376638373138646431343634393962 +31306564396262383438643937353563353663353536373032353263636430363235313736653333 +39303336316438616365613961653737633234323131356339623265373463336662366336396364 +38333962373164383630363866343361376262316462333936616230316536396363316163653463 +37306461346136323431323265383030643232633062386433336261316333323030663234663063 +65623434613662306137396361616330333233386630313935653861333761323565646237343430 +38613139663035373563313530366632303232643030643466346531373333363337643831663736 +62636538343035643464396461326363646537393138613166323762346539333033633661383531 +62653634636666343661353565363266303430623131643733633232626230623466376337366137 +65653565626366373337626632346338396533343138393431346333346161353838386434643863 +33383935636132343764656232393335316431633937323533313338366461376438653136363231 +34643363356664373765376465333534653137383961386262633165636666353464323261396465 +37353064353530393037303930303430376530623364633832636434313264393135383536306438 +35316436616536396335393863363863653763303666653161383630336437653663636434366635 +38633633396132623663323137643338303038393061633961353064383765353735393432313766 +61343735363538316134626434326264383932393964646136306537313932316238326362646337 +66363234363334373737623361616235383231303834636536393836373466323265653339326239 +39333135396464346232386362376335333234383231343533393264346335376133313434393633 +34623039373134633032613838616532303130353833323330646165396631313864643831333763 +35633831626465316533346431616362366331633937393037666538633936343964643735363733 +32393939303931393362656332386234323634376534653435343137353063333037633033616639 +62643037376335316662613064396339393639386336616362623963373265383231613063356335 +63633432623061366237636134663834336165306234643537373064633561613561363963323031 +36653439396163333439366534653764616361313633343763323938636638656461616261343536 +37653866346161333533323436323063626534313539363666636232653639343366663663633162 +37343539386163666464643466326562363566316630633530363235663732316463346232366362 +31303037316265333862656137646631616163623330616565303933623731393530333230616533 +33356638303132313864343938313336626265656362613633623639653537366664383265636235 +36303633343439643339653833646536323161323638613139613734396433656365613862303932 +31353764303836653434653637653732363464373238376637393766323661333735363336336662 +65623030633433333339306335626334396162626161373637646337313535666431386432396162 +63333761363462643335633136316139626665633838613533366633656461623064393963346366 +31656637653436653631306230323630333531613465663134393339363930623263373031363839 +35393931313234393761303364386161623065326661323566653535343735346165643961303662 +63323436663930663031366131333361316339356463333634383937633535616663313333313037 +63363363643764633735323331373931396534646131323166626537646638333230346362656262 +64303962653265316530663531613235323165306565613866373966633038353237323933316336 +34313566383234326233313435336665633239666531656536326230373536666565636336653964 +37666533613165666262656637376662656339636538643639356435356332323964643731353462 +30653463326462373734313535643436336432623965616637643537666432356264666338333364 +39353134633761633461346665353161656436303764343661343530333532343739623830393762 +64653831366263366664323534646163663761633335396363653739386431303265643865373830 +35663930396333303430333831666433616366383366333032383232366431616338356138393533 +64613863383631333231373833616230343930393531646535303333376232633162333136646431 +34333639663032656562646561653061313132383832386266613431343630386530636362636330 +62626537383961666664656363343830376662306137336433663162323233616466386438653361 +34343138396237633138326633383335393330653231336238666365353530393232323332333837 +63666462613637313738653136333131623436653033396562373462373665653266366237653932 +35353939316438626233323561336230333330336632323062363436666239613137316563613062 +62393138373036636461393364663934633439623837366536386132663933323139396435393932 +66333662343965323338623531633562666631343963653233333637646565343634373337643636 +64303566643537383633663736393037353635303335393436643634393337396434393430656131 +61643830323039633363326565636139663534326434666161666231623434373963396161326566 +65626636353537343465343062336239313537376462316161343039373032373237313462663032 +35666335626337343334356530616236303639383035333362323035353862303863656331643630 +64663864653066623938323162356631626537663463616464663134613836633033626364313732 +66393065393365343339376637313362343130396464626363346330383164356538363833373430 +62363030663639343934613233366232363637323165363434373262356336663037306163663636 +64346534663833643932373733316331323032373738313637646464363563653932363735316639 +39386433336435316331373330343065396433323637613566613835356436333631323833313866 +33373332323632393239633363356232346634303233373663646536666635663233396335353733 +33336364646233626564663938643230373532613133343439636363323033373461386463643763 +36656234386563666662373737633537656262396234326232396636306139393532623531313431 +65663338663639663462323932653661623436306431323464663430613363303463343236326165 +30626635366639373331663239623763373966323934623365316563383039383737383262353130 +31616463353364353036313863316165656231613664373237373933626331316638373566383139 +66626634313234306137376166613939383664653536386532643636336466623132646639363564 +38373963653337393539326162383535373535376334386439316330386136366337393139336264 +64656337383235383963346335306632653837383166653737343661323237383065656137393232 +66626261396261383466336361343730616336626435363536613730313635323761353335633936 +63626164316462666461343038636633366136353864323135346335383034306464373437333062 +35626464333537313639356662316366613833343365663132303266333538346562326562323434 +65323230653334336239616163643036653537643264656239303666343430333765316631643337 +65356333346534623466343439333932656332306564613065656537386232636662373665306465 +6563 diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml index 6bec8f7..8d15ec0 100644 --- a/ansible/infra_secrets.yml +++ b/ansible/infra_secrets.yml @@ -1,174 +1,178 @@ $ANSIBLE_VAULT;1.1;AES256 -30383238343938316437303961393961633538303835356661366465306161353738616335343563 -6135613063376562303539613535633463616433383239380a323163346131633838393836383539 -38623766303739623131613935363037323462353664386563356661393665306136323863656363 -3433313638363731630a336165343563653930663435333037653330386364336530373830346336 -34396361653637626137343033343432303438383761393765313164393631666230616537366636 -33313234636535306535336132373830383438636235636466623861363734636238393234316137 -36643439643130363462643765353061623436376463386139616435323136376366653932323766 -34393161353062386666626633393639623434653864383439343339363665636134636139326538 -36323966373230363938623434336161396235316532663739353837623238343862633032613264 -32626565313637306131393466653739383935646134346330333835393932396634323565313861 -38613864656139646261343835316562323863363836376266653139336638366161323035373265 -39366566333136646134353065653065333337303832363237653361353234386335643165376331 -35316339396130366163313130633561643062316164376262353131323237303861666263643536 -32333062316330313532343832336434653664656565616634643436656466373533376535396533 -35626636393135626539633365326364353462633462656364316361303732643632356230643333 -30396233616633326438393836383237623636383464333830343464343837383231653139386336 -65313337366665343936366430623162643032626430336466643230313530343832383730303263 -65386235633832643931656562303561666131646362303533323430336235643436653866366161 -64636365633731613365663833663264343233313037326537616164656231656332663461653635 -31323237656361666630323930336630626437393836313431353839636435363466376535633962 -63323462626662616139653665346132343736636361643930333366666237303438356364656464 -36663363386131646361316535653463366234663062323539646562633962653537383465366439 -36343635313934353063373734386137653239343538663037646362376434383234363361623165 -66663734313133343062363137323931643338326365346234313835303431613632623461366339 -64656138303166356136313962646133306536333465316239383532353362333932326639393363 -63393430373133316532633562386432663332623934376636306438623061353238333562313737 -63393338333534646265643462666163356365653937643634316266303730346333346465313662 -33326638643639326361386461663931613766373937656266313964336461633939333535646231 -35316331333537303830653336373530353364616161346232323130336137373737616534373435 -35383466666262333834323332393737613763303562643232616432316636386635383934373633 -34316462393064656633373562333039646135373332373063653539313230646466326233313536 -38346238623162353338656437336536663534643062376138643030363935336461333638363134 -30653738383233636531333165386131383061363466313736336632303136616261623137313065 -30323766393862663965623337303638366239663130643638383765383930353533323635316232 -34646339366162653663323964363863623962643166646130633236353563313061353434376334 -66313862313265653238633034616530343164303432666539363733666662653733666462303662 -65663638653534643132353361313130323763636338303836346430623664303732356636636232 -62396135343236646537386334373166303663353661323236336365336533613139373430333364 -65363863323335353663396366616530633836363837653963656636333264323434323365396135 -62613862386361616633333636663263376465613465633362333733326665326639663961313631 -32653735353563366236363936343438623665363834633436393232653737623436303237376530 -34353032633433363865303866643761343130303431326236353332356464396433323430613830 -38373035306137363562376539633863343737616339346463636236393537346163323731333733 -63656161306437343862643836633539326661653062383361613739366237623162653939366335 -30613466643665313830353332393365623933376638326534336664616561363461643237386533 -64376532306663373934356361663832323337393266356138343236663964666637343431663237 -62303531353266633164336363356336616332663730663933353730356365393866313662363730 -63346434663162313333663738393461386638343864616634323938323437363232356466303637 -32663630633466623839306537366465653164616631373564366132636165346231303065653036 -66333134336132376232393639616163393366343365336661633330316436656163396434343963 -30356331393432613561636562343138643062613738613764666232656639393532623433343138 -37623963303237663337626661656638636364323931643730343535303737656461663831336336 -65653031656633386562663837386130613635306438313764653830663966663232646663356132 -61613338663331666265393566613661653462323530623930373034363363363238663465303162 -39303465613435343934636365336262376431616136383538353862323235643835373638663832 -38303336306665616666353963653532313162336536336137613337653363323138626338333861 -36663364656463623337663362643936393664363937306135353464363061393837346161333435 -63343030333830626438653565623065383264616264303035613432313266643739333063653066 -39623863373632363337663065653865343433393333386633626237393162383338343038646165 -30396331643033386331333039663066356432613733336235353266356336326138323864396132 -38313164393833343339663363366438313931393232613935343063306233323061363066616366 -62303564643930643262666364346130363530363630306331393032623736643630376563636533 -62656338343734343739383565396339323138666265626235376235613963666566333862656533 -63626566343737636532363535656238316332363831303862313265383936613339616430376231 -32663363613135626463373166623063336135303437313031353337323939633732396139623731 -36653061313233363837396636643234666439663261316361346263343934393237626134656230 -66326135396661383432323939666662376163376233323339303939336363336531653163633464 -33346266346533346631393666316239343463366636613336313833663730323734613135343930 -62613130613933656237366262363433353434303763326564333965383662336661313037616430 -32643061313638373931646464383465626234323866386336313861336531383638363736633438 -38643662376364333564383963613339623964396666313263653336393666623932393738656665 -63653234316265316535303531646230326530616639653036653731373934316238313563313338 -62373162346333666634346362333232373234613430386237656337353661643839663465663265 -36303738636137633636646537613336386230316334613739383362656661643163373936663130 -37393361323837323032376464323961306335623835616134633735626164306539383936613064 -36653133666230336262616438343539346561363863653364343662356233633239623765383165 -30336635393763646432396437613838383733626565353737336165626333643339376432393366 -31386334326130346164633534366532363037316334373133326362303965346633633939333539 -35613435626232653733303266666431666133366430633538623931623231353235383364303765 -64376133653562336631326630363036623033303039363561643236653531663861336262393865 -37346633626334313736303831633236313562393363313133613836326137303435356435613337 -31613335336562376435393237326439613863383264316537353165373437333866666364316366 -36326332336335396430663332326562613235623338633930633437663162643433663065323234 -36613663356534366433383431386666643736323936383030396139326464623961666666323462 -35613335633666313330636161613836343861313861666530323935343961316631666566653565 -61363033306338323966376631343232343038393564616262323532626131643265303865396539 -62363433353666626666393331636537363535306534383634333262326438366463363138666230 -32386366303462323930343863333463383638653532386362323338306266393337326535363663 -36346366393631396163343930626537643161353038383934313331363834323431653364303939 -37326532643864343338626430363065353833353862353262336235336637663865653263663339 -62363366663838366236653364363064356137346463383933626533343461356162363062653932 -37663966373031363133326534623763643030356335386139363565613135353835343233636330 -66393532396530643532626336303737306134323537373135663435376632343331323663333737 -66316638396563626663313636386566363036383663656230336530376234376138303462323030 -61613931336364383265366234393933653432313562666238666639326430376238623365373838 -38363339616365623065376438323738306633666561326432303935393938373638613736303332 -35353833663431313462333362313732353534383630343030393564626139353139623530633332 -34653639386263313738613430323664383264646632633062643031303437636262623966313630 -30663533623334633164353964643536623566623132666166373039303361346534346337396430 -66633632633162316462636662363938656263663135326237303361386234623661333938616262 -62343731623838646133333135306438623538626639663863313337313862626535303937346138 -64623338336333323239643565336533323234353436616230373063663539613839646238643735 -32613163343637656337653139346463383734363636633635316261646133343731653133613238 -35613265633538323039663665353664333434373834393264653830343766343136653632363532 -65356331633262643635343861393535353532383137346534633661613963333531626463346562 -63323638376536626534353062646561373566396235386634646663616230633635386661333334 -34643833653065633033333839633736656161396236363364393838613038326532353130353063 -62373635356665643133353561633861643436613035636239373630373164386531343736643564 -32656536633631353931666137663234343861633865303436323434303636376634366665313234 -33336163396330383263346333323065616437383564303839636235313964343633626166623833 -62316436326532386338356563643038663737616338363636383032623665373862643631643331 -62336364366537366531663762393964633531333166303662356334383562383466303334663261 -63663638633736353934663530343934613837623963366232323637626466343138613165626634 -38663338386465353463633438356434353839303334343165386430623733393934343466333134 -35373137653635653966383865333637623930666562323236323434653063613464353362396164 -31353632353637326330316239393665653833393438663436376562326638336239396337633134 -39343031323931303061663837353931396263333637633264643966343633363966626135356333 -30643564666336636535663264396632636532616234316130313761333731626639333665393430 -37653765336533346564306139333436623466626664393735363865663733306161393634323161 -38653262303363313562363233643634623364396237316636623431656366393538633131366139 -64383833313537653137363961306538353230656164626663613462323962363631643336356561 -61343539393239623861373633386338663534616438333538613632356265376132356539653964 -61643234373639643933653963323631336433633863326263303938643166356330396339626637 -30353337626231333730386465326533346233626561373564376234343763616431313031656531 -65396363396436383337373133333231393139303734643533623233613136363031616430653530 -31383231333763646437646430346564623739376365656661363539386164346365633730623738 -65373334613138336263623262653833363566633638646139656433326165353637363131316230 -31653436353336636333656430656564356233626535333263616136626136363430353237366230 -37346533316562653764396262646661623232346335633966303833356332346133653535356638 -61386338363139343561636564396232643361333936656637363130323165666536613066663137 -64356233376662666432366364626639373534663865343865323363333239636461326665306537 -35666533643834373839326139623564316533303562306235386238313764666438656139393362 -61353337303761666538636533613839336237646331663231353831303663323739623362326363 -66646136383765356431306639383336623237346534623535613435346238353866316164333763 -62646665396233646632633535633332313934653964396562656464376635306434613932623664 -66303131616137616439353939663731306438383662393039333036663862336537356335306266 -66646538316432346132633764626361353561663937323039383138636331326232643932633932 -34353061396461646664633364303536353833646130303032656238313637363736643663356237 -34356161313463626361393031643332656464326166636333616162383431396662383261383033 -31303537326532323238363539613439626365613538643364646366326236353934376539383865 -32373731323133383335343639626133353232613033306331373237646364643830613532306237 -37663034613363363164363935633332633732653761623439366636646137326163306632316264 -35356164343462633938333331323836313539303961316532613032396632666138646265636337 -38313539633430646166313435376532323630343232393735346335363731363666663363663138 -38373465396461306665303331383266663836666434626137636632653266393239656631353833 -32626531363763303439356565666434333334383737316431353032333638366338383430313436 -62326331353834303261663730303066623062336638323433353565656638336235663734323530 -66396666306534393139656138666430373537653862343232386233313730313831643734353539 -35636361643230316665306563633836656261353338663538336666383462343132666361356233 -32623732646564316461373537373331363138653332393638656437643635376663303430373933 -31383837643033323334326562663135613537373661396166656130613963383265303961323230 -63376662656232393833373731653234316462333834356534303738323430336633646437323332 -34376263396662373931333938623330626438663336323066326239636535323562623936636335 -39373335363663643238316230356534643033393363383436613865336361653566626336666630 -37313739646337653436663665386666393063316263653833633239396632313734616262616166 -38303363646661373139343039353532363736336261356462366238343135313637636166643431 -38633236386535346166386434616234326164346661386534663834646464303038313561613034 -33656265653739383932323632613766333335363633393666396663306337333231316639633731 -35383062613633363065616535666638326138306431366130336238343263356362333535346435 -35643866386430333730663461353136313630323961633764616134333835643234656633303232 -35363333656436653563353336393066306139633937623037643566633166646264313864646364 -36636663613465656339613534613634633733616662346530356361383934633466616138383865 -66313930613030386163346363333836633139376637323337396434613761636434666164393534 -37346437356431313930333932303465326138393262643665643139646134386161613465666363 -32376464653737363130303761303838663061626636643534326235333062643735653837633938 -35336161306435323665653035313930646666366631353464626235313761386135303665636137 -62646466313662613630616234373933376138383037363933626437306466633832343238633964 -30346265653865623537323833393763343632323839353237363466336663396131333266396533 -65626561646233626366353862396564373463623563343961646662633133613333646435303030 -31323332626534383732643839353739653835333934373835653331663762376337393930363833 -36313033633239383830363861623538396463396163333138613437303535396265336664313639 -31653837623938393137643137306566386666653564373838373265306462303062623436653261 -31333138396664613962303965613536373239643166353831623565643662323564 +35383939623164356565323465316566346532343163646633323634396364376462313461366439 +3637616136336262303234323464343434643234383132300a333665333836653636393166376330 +36666466393762346333363930386334333863616265363661383639333561393061313938373366 +6562316232653761340a313631633931353736383636326339656432333065663032323033326334 +39646438633330323662643834383438656134376363653938356639356338633635663831323938 +62653533613861383536346230613161393834326637623636356265633237323331323565323339 +61306565663831613635656661373535396564376561653035616362623333386338666432346634 +33373033333737626534366164306436363932653231346134616163633634633131646165316330 +65393031613461623762383066626566373733666639333766643131396433306637653063356361 +35343231356534626661623331653365643230646566323666633031323734663933643864336264 +61616333353139626462623933343234386238393437623833363461323834306233653061363535 +64326339356435626265363235313965346165353434613962396465346265346539343764663733 +31343234356331333536663239336334653864393132666264383263333365663464663931626562 +36343430656262313962633965326539633834306263363036343562663132393537343863313961 +35356161386234326361626165343234393837643364616339666131343364386132333763623737 +39373035626332613566396532343535646235636661663263353338633763666363666363633465 +61353566646261376534323662363062623962643062386164666136623966363535356132656562 +65653465373366636533336539333035346537346261633933626334356631393461373361623537 +34326463373766393965653639313063383764313736626366653063663466646161313365306532 +63353566636433346237393834386537396232383961613932666537373565653338363435383938 +32333438623332626163336631663664396430643566376363366262656432666261383039326431 +63633036393634646237393665353666666239653533393532646464643934373535666137373764 +62646535613563363939313666636466623762636439323237623236383165666630363938393666 +34336439316665326265623335383434323837326661343163643833663330663834376432333134 +32333737643364373035626430393963373934316663626666333837666161616138633032393663 +37323432636435383462383061346332373862376336323363303031666336646161633539643033 +37633962356130613138343235633564393330373463613861313638303633326639626436613763 +61383236643439616232613331366535356130373365643064366533613330383661623833303231 +64353532646337643630373361303564386166306639623137643534376631393731393831383261 +36373934623337646234353165316563303336373437626464356463383733303338643936376164 +63303638313766333832323764666431313937633930623865353331616536653930653336343266 +30333931336165393634633231376266373933646661306138386631663539336139333265663631 +36333562396239623236336333343935663866363063353036663337646431356537373434666331 +39356635306263323365333263386331656636356436373862616434356462376134363034663532 +64333633663662623931343363393233306331303066303664323034373333656132386437366162 +37633830626362326637623139396631663834623436646364363439356361666661396236353739 +39333766396132616333323566393936363137333235643739333639646234313266303062323536 +38633964373238646534626566366438633364636461343861303437356562313966626235326537 +63316438373461313936343634373835306539333436393337656535323662663936656638353838 +38316234663432656539636239363861396139376264303336356630396163353234656562633362 +39613032653163313362643733393232316530613533343535376339636139393936616263346561 +30356134313831316364643439373135393836303330303737636137323035386432343963393165 +66366333306262353763383838376662633766636532643064373962366665313761646336303033 +32643633616162316537376435386536316431343136346264643163373033336463393130383562 +39623635316333383036336538626636363130353733336531396132623466363933333230616364 +30633364393334643866393332346239646235373766346562653061633732326537343634363065 +66646662336661386432306231313362636535316466656435366639383030376365313364666564 +65666535663532383662383833633261343934383535313163643133646663396135613761336337 +64653433366164373461643038353830626139306566643438393933396339616361623062346536 +31353039633437383862343865343133336665633436366137663033643834303835336633616361 +30363863626463333437626165333338366332653838663464633964363961373338336462376332 +63303738636539396635613938366336303462326564663937656231346639326438326561333430 +38646538353362303062376138636362333166373931663036646438663937326438316338653861 +62356365633966623638626138653064663835656431613262393532623731656132343864663966 +31373936333031646131663436323036366637326134336132373565316330663066366237336439 +62656562346639336339363363643933633566646363356565366336633863636230316634663936 +38626662653131623064373639383837366538633339313864346631633032356362373035303665 +61376332313532353665373131613335653732343839366365346330363638366265656139373838 +38633737653565366165306562353333353637326530636535616566323061386362333266383636 +61636365386232306338663962633235613435613466643032323363646232306431646164633639 +36383139313363613630616662613734636465646130363933376438633839363163343866323032 +35356531376663386331666238656537613164633634633934616462383331326633333235626337 +33393330303532626132393239633735623963363562656163383564306461353330376532396334 +30666361353862356636636632383234323565376235343862343564383830326533373861303437 +63613731643964613530343234323661636133643838316564356338366566303261336330616432 +32613535633630303132313933393833636433373539346261613661626135356363303930636138 +39323837383863306264633838636530393632363936633938626465656334336333333961666331 +66653463343333326563356337373732643333333065353436393736613038346636643535383466 +31656535633765386339633938666639396438326638303961303365323032333564636562343361 +65303662343633343364366265646432376230356362373465356539623632653663373631633633 +38666138363837356331343061373237303337656432326633646565353337393238363362396238 +66633966383237653064343333396661653932386536383536366632326234303161353166633836 +37313438393263366436613066386235343034366433343432346266343137346336646238613462 +33336339363037343533616166616565333330333531643364393437393637386330626437346135 +64396334656664336535373764666535373732353639326433646561616338323633623138373832 +30616363303834643532623362306362326230323737653563393838653234373861373933316538 +64376430376161316666396336373662393237383531353361393435386236636631613336386566 +31643637623333623364323734323961643863623532613432386662363361316134626533666339 +34656130633462616337323938626439626235306533336264343138356439633065313434616132 +63646664343962373365663738356233663135653965326534353665316531393964636132393932 +36633638636236326135323335646535396266666433396462623963656664336539386662373233 +64363566363863376466316338316562393461623262393130303762646334336463623632393238 +65656138396536373132363633313636643864353266356266626266643737393366326538353237 +31333961376236653263613930613432666531366131623834633632646332316133666530646139 +65616466646564646437333730383836656566643465303931326665343764396137643338373631 +62333266326265356562656430666538376134356239666635393430343234363231323938393331 +31613463613936656138623535303731333535323130303835366465383665353632636138653338 +35663132386634313538326130343435313131303934393434393736353531336261306265303264 +39376537636566333563393032336534333635346432663235623530666264323930646361313463 +39343437313932323064666632386137353834313930343636356561346535613730363865613433 +63316133663935373238373566326363343130376666613331666361326262353266653535343866 +64383931303037386165663439643232656139653832303239383235383637336664313864663935 +63366631303536336530366438333762393839656435366161353838303732386466356238333632 +34646463646535376139663136613562316166356138343939633935303532326536313033643639 +35653765326234333832366165363962376263383465313633313664336636616430313138313365 +31376535613130666562336533396636646161633932633030363437616337643761343439616338 +62333332376465306264663137613265363432343538643430333764326562613065663062613665 +63643030383163343231353964353264626434383762346639366335346465336365366131646131 +38633962316232663530376133383034303465313331393561393636386663623861373161376366 +37306533636463356232326430346538346136653264343839356662643435633337313232616561 +33633331346166303038656135303532643531383466346334383665613662333431336362333334 +66383336626463653136353861613832366338623631353930353432323338663834316436373230 +37393135303037663035303633643439653238663038626135623436303435303361393064353034 +35663230656434356434386462363834313137353865313831343261623466633861303336366334 +33626164313334306362393461313633376464343037646534666339386462666634363563386661 +37373138643432373834356335623864313561376131633161613035386166623836613339333563 +31613934363331633537663261653230306539616238313837393762323932363365326238353131 +31666564303666373163306237353136636430313639313931303339346335653539643265613230 +62643836353432316165336665373964353762376432373263323666393333393564323932333638 +33643763613739323134666139653862396436393836383232303564636261393161656163653830 +66653237313262616464656437343465623961633562376638626432663530356434396463663465 +39386261343933366666356664313339373231326566633165306632353535363436313864303936 +38613636326666303061616337666436383263363931323339646432373965363565333763343631 +37343566323732303830353537393565306434363130323335633837663762333431376539633765 +61363939626234386564306539636637333137626133363139353235366238326633623062303833 +64303832643830646430333061646333633039663634336263396563393437326135613932386634 +35393639373631323435373864306433373638613836396664633537323636653531616534316465 +63626661313733333431333665663165646234343438356134366131353234366466373133623639 +34656436306433363136666633616464373937633435616534376536393233313236356634323161 +63633632303562346566646166636664376431396162663765386532613832326339666130666536 +36663865633963613536623637303061626336343835646639633031366333393132663033396533 +38383862646631333262366237323234396465356531323363653464643361326362366231653739 +62643738303734643033323362656230616466353161663136623864313136386366376634653664 +31363930373530346333396536623037303162313934333634373033326432373566373032643964 +30663962353664353062303935356262363333366437383338373961616237663361326631353565 +37313936373033323131303161343139653565663466323830663536346665666232643163613562 +30643061653363663864383663326563306261663061326366383731366463356238373539383935 +65626464346366613638623730383631393066393233353130396636393438373565383266636439 +66633837626461623335316438393432303839633234323065396533343466373962633566643032 +35353939323038333434316232326138326266343662636130363565343463636364613263383438 +35653736333339353831383733393161356533343138336161383435333261613262306135393565 +66333134326265376433376538373531353966396138656265666266353737343936366132623633 +66663339616263623139666236373563363666356236383538316363326635396332373161373932 +31313430313065353762646538383361316563326262363833316434343965383032633139663634 +38383430336432373031636635366539663337666330663361303964653331376430303263613461 +39303832306463356465366237373338356636663637353835663336656331366638656433623462 +34623761366365366634353635333632373162643330633435303466326437626138663834323531 +61663433656537366662386134393761396162326265393465363131393333623731393534656338 +35356132323062336235326630353464393434633633343932613432623536343365626165626432 +31633732316364616463656430623763323039623835343164326536316334353666616237623535 +32303036653936363933353932323666636662393231306138306239653065623338336662613633 +34376631386235653161323331313661656366386635343264353535623439383734393363666239 +64303466393539353862626436643366306564336638346432373661336537353033613036393630 +35383962663363366665393938666266326464383936653539363838623338366466626439663438 +35333737356234363731396162353730643665666630366133633839303836313064306330373233 +62613363316238653534313739633033336134386264666362393061653434356637643736633863 +37333766646666626531323738633161366537306539303961373037656461323265353032666338 +30383834386333656334376361643338333662353461303932373132356133643934656466373864 +31306466396231363132636463333633383939376462386230363661656164666331323333363435 +33383262343238383265653232653538643964346236643031386164616364393732336637636239 +39663333646137313933313562626265663365633633653465396234396563656539646630346634 +63643461666365643961623936623364373334663162366537303163393665303633303562336361 +38323431346334623835396165653232663361383836663633386361616630633437623332383732 +33656139323339613164653433623237623939633230663735613063623131623632393535353066 +31633234613563616566396262343334313262366434306662613965316135323965643732356130 +31643437396264303831653762393930383431323062303032333666653761623866653762333663 +32633862383763633663313166313365333963636233386430366165643635633938306533663164 +35326561393265393830376434393235613265653764636538663165326363636538343630323763 +32396531613238313736393462313736393535613662353365636165373831626466386535623433 +63366338616637386336613435316262393561393139336331323132613835663065623637333833 +62663262653536313334623736343535323766643235303865653864613862663664386664643934 +37363832653530626339313336363639633863656633623838653564666135646333306264313765 +39613331313432636330383066653431383333323265623233633434343066323764633765663832 +34653434323631313261653234353437353762616466633463383835313034336234326666303864 +35333034376131323436613735383433393266613966363432646663366636303564666532646163 +38333963306665306438333034313933303337363331376663633333643161636236343662383361 +33333933346665613061346634666530323830653732356231306365613436323263316462373734 +65366462646532366638366235646533383235346463613164393630306636346232626264323932 +37343631656361633235333338613933396565376636343932306363623635373237366639633236 +62653537313134613239353365613239666466616630653336653738393032646263336561623861 +31333434646532393963346238613662343737323831393136383736636631393065306664616333 +32643666393261623262326562356631353432353231383161613964383566643466323962313238 +33393531363135636362326434383962633463633039333664353865383231353634633535333738 +62636432646461376431396638333231643164306432333132356131316538303366343333396564 +35373732323633666263663064356262323432653462333834636433613231356637626265653866 +61346366353165363639633466336463313462653137613035373430313334336262626439393539 +3663 diff --git a/ansible/roles/bitcoin_knots/README.md b/ansible/roles/bitcoin_knots/README.md new file mode 100644 index 0000000..8e35900 --- /dev/null +++ b/ansible/roles/bitcoin_knots/README.md @@ -0,0 +1,85 @@ +# `bitcoin_knots` + +Builds Bitcoin Knots from source with PGP + SHA256 verification of the release +tarball, runs it as a full node on `knots-box`, and keeps a health check on a +systemd timer. The second play in the calling playbook publishes the P2P port +from the edge host via `socket_proxy`. + +Converted from `deploy_bitcoin_knots_playbook.yml` (892 lines) under Plan 6. The +playbook is now 40 lines. + +## The build is guarded; the chain is never touched + +`build.yml` is 32 tasks, every one carrying +`when: not bitcoind_binary_exists.stat.exists`. On a host that already has the +binary the whole download / verify / 30-60 minute compile skips — **including the +two `state: absent` deletions**, which target `/opt/bitcoin-knots/source` and the +extracted build directory. + +The chain lives elsewhere and nothing here touches it: + +| | | +|---|---| +| `bitcoin_knots_dir` | `/opt/bitcoin-knots` — build tree, safe to delete | +| `bitcoin_data_dir` | `/var/lib/bitcoin` — config, logs, wallets | +| `bitcoin_large_data_dir` | `/mnt/knots_data` — **~875 GB of blockchain** | + +The signature-verification tasks are the security control of this role. They are +copied verbatim; do not "simplify" them. + +## ⚠ This node is half of the mining setup + +`bitcoin.conf` carries a DATUM Gateway section that was hand-added on the node +and was **missing from the playbook's template**: + +```ini +blockmaxsize=3985000 +blockmaxweight=3985000 +blocknotify=killall -USR1 datum_gateway +maxmempool=1000 +blockreconstructionextratxn=1000000 +``` + +`blocknotify` is how `datum_gateway` learns a new block landed. Running the old +playbook would have deleted all of it, and solo mining would have carried on +grinding against a stale template — a silent failure that costs money rather +than raising an error. The template now carries it behind +`bitcoin_datum_gateway_enabled`. + +**bitcoin-knots and datum-gateway are one system, not two services.** Changing +either config means thinking about both. + +## The restart handler, and why exactness matters now + +The hand-written `Restart bitcoind` handler carried +`when: uptime_kuma_enabled | default(false)`, so it had been inert since the +decommissioning: `bitcoin.conf` and the systemd unit both notify it and neither +could restart anything. A config change applied to disk, reported success, and +never took effect. + +It is ungated here — which raises the bar for the template. **Any** residual +difference between the template and the live file, down to a trailing newline, +means the task reports `changed` and restarts a Bitcoin node on every run. It +took four rounds of `--check --diff` to reach `changed=0`: the DATUM section, an +explanatory comment that was rendering into the deployed file (now a `{# #}` +Jinja comment), a `# Pruning (optional)` comment the live file had, and one +trailing blank line. + +## `dbcache` + +Computed as 90% of RAM unless `bitcoin_dbcache_mb_override` is set. The live node +was hand-tuned to **200 MB**; the calculation produces 3528. As with fulcrum, +`set_fact` outranks role defaults, so the *calculation* honours the override — a +value pinned only in `defaults/` is silently ignored. + +## Monitoring: one variable, no product knowledge + +The check tests bitcoind's RPC and records the answer in its exit code, which +systemd keeps: `systemctl is-failed bitcoin-knots-healthcheck.service`. Set +`healthcheck_push_url` to report anywhere that accepts an HTTP ping. + +The timer had last fired **2026-08-09** while still reporting `active` and +`enabled` — the same `OnBootSec` + `OnUnitActiveSec` dead chain as fulcrum, where +nothing re-arms it if the service does not run in a given boot. The role runs the +check once after enabling, which both smoke-tests it and supplies the reference +the timer schedules from. diff --git a/ansible/roles/bitcoin_knots/defaults/main.yml b/ansible/roles/bitcoin_knots/defaults/main.yml new file mode 100644 index 0000000..29d5aa0 --- /dev/null +++ b/ansible/roles/bitcoin_knots/defaults/main.yml @@ -0,0 +1,71 @@ +# Bitcoin Knots Configuration Variables + +# Version - REQUIRED: Specify exact version/tag to build +bitcoin_knots_version: "v29.2.knots20251110" # Must specify exact version/tag +bitcoin_knots_version_short: "29.2.knots20251110" # Version without 'v' prefix (for tarball URLs) + +# Directories +bitcoin_knots_dir: /opt/bitcoin-knots +bitcoin_knots_source_dir: "{{ bitcoin_knots_dir }}/source" +bitcoin_data_dir: /var/lib/bitcoin # Standard location for config, logs, wallets +bitcoin_large_data_dir: /mnt/knots_data # Custom location for blockchain data (blocks, chainstate) +bitcoin_conf_dir: /etc/bitcoin + +# Network +bitcoin_rpc_port: 8332 +# Shared with the socket-proxy play on the edge host, so it lives in +# services_config.yml rather than only here. +bitcoin_p2p_port: "{{ service_settings.bitcoin.p2p_port }}" +bitcoin_rpc_bind: "0.0.0.0" + +# Build options +bitcoin_build_jobs: 4 # Parallel build jobs (-j flag), adjust based on CPU cores +bitcoin_build_prefix: /usr/local + +# Configuration options +bitcoin_enable_txindex: true # Set to true if transaction index needed (REQUIRED for Electrum servers like Electrs/ElectrumX) +bitcoin_max_connections: 125 +# dbcache will be calculated as 90% of host RAM automatically in playbook + +# ZMQ Configuration +bitcoin_zmq_enabled: true +bitcoin_zmq_bind: "tcp://0.0.0.0" +bitcoin_zmq_port_rawblock: 28332 +bitcoin_zmq_port_rawtx: 28333 +bitcoin_zmq_port_hashblock: 28334 +bitcoin_zmq_port_hashtx: 28335 + +# Service user +bitcoin_user: bitcoin +bitcoin_group: bitcoin + +# --- Health check ---------------------------------------------------------- +# Checks bitcoind RPC and records the answer in its exit code, which systemd +# keeps: `systemctl is-failed bitcoin-knots-healthcheck.service`. +# +# WHERE TO REPORT HEALTH — the one place to plug in monitoring. Empty means +# check, exit honestly, report nowhere. Any endpoint accepting an HTTP ping +# works; nothing here is specific to a monitoring product. +healthcheck_push_url: "" + +# --- Logging ---------------------------------------------------------------- +# The live node logs to a file. Set to "" to use printtoconsole=1 (journald). +bitcoin_logfile: "{{ bitcoin_data_dir }}/debug.log" + +# --- dbcache ---------------------------------------------------------------- +# Computed as 90% of RAM unless this is set. The live node was hand-tuned to +# 200 MB; the calculation would have produced 3528. As with fulcrum, note that +# set_fact outranks role defaults, so the CALCULATION has to honour this - a +# value pinned only in defaults/ is silently ignored. +bitcoin_dbcache_mb_override: 200 + +# --- DATUM Gateway ---------------------------------------------------------- +# This node feeds block templates to datum_gateway on knots-box. These settings +# were hand-added to bitcoin.conf and were missing from the template, so a +# playbook run would have removed them and broken the mining setup. +bitcoin_datum_gateway_enabled: true +bitcoin_blockmaxsize: 3985000 +bitcoin_blockmaxweight: 3985000 +bitcoin_blocknotify: "killall -USR1 datum_gateway" +bitcoin_maxmempool: 1000 +bitcoin_blockreconstructionextratxn: 1000000 diff --git a/ansible/roles/bitcoin_knots/handlers/main.yml b/ansible/roles/bitcoin_knots/handlers/main.yml new file mode 100644 index 0000000..49ac219 --- /dev/null +++ b/ansible/roles/bitcoin_knots/handlers/main.yml @@ -0,0 +1,14 @@ +--- +# Ungated on purpose. The hand-written handler carried +# when: uptime_kuma_enabled | default(false) +# so it has been inert since the decommissioning. Two tasks notify it — +# bitcoin.conf and the systemd unit — and neither could actually restart +# bitcoind. A configuration change to a Bitcoin node therefore applied to disk, +# reported success, and silently never took effect. +# +# Restarting bitcoind re-opens the chainstate; it does not reindex. +- name: Restart bitcoind + systemd: + name: bitcoind + state: restarted + daemon_reload: yes diff --git a/ansible/roles/bitcoin_knots/tasks/build.yml b/ansible/roles/bitcoin_knots/tasks/build.yml new file mode 100644 index 0000000..5a1b900 --- /dev/null +++ b/ansible/roles/bitcoin_knots/tasks/build.yml @@ -0,0 +1,222 @@ +--- +# Every task here is guarded by `when: not bitcoind_binary_exists.stat.exists`, +# so on a host that already has the binary the whole download / verify / build +# sequence skips — including the two `state: absent` deletions, which target +# /opt/bitcoin-knots/{source,bitcoin-} and never the chain data in +# /mnt/knots_data. +- name: Check if bitcoind binary already exists + stat: + path: "{{ bitcoin_build_prefix }}/bin/bitcoind" + register: bitcoind_binary_exists + changed_when: false + +- name: Install gnupg for signature verification + apt: + name: gnupg + state: present + when: not bitcoind_binary_exists.stat.exists + +- name: Import Luke Dashjr's Bitcoin Knots signing key + command: gpg --keyserver hkps://keyserver.ubuntu.com --recv-keys 90C8019E36C2E964 + register: key_import + changed_when: "'already in secret keyring' not in key_import.stdout and 'already in public keyring' not in key_import.stdout" + when: not bitcoind_binary_exists.stat.exists + failed_when: key_import.rc != 0 + +- name: Display imported key fingerprint + command: gpg --fingerprint 90C8019E36C2E964 + register: key_fingerprint + changed_when: false + when: not bitcoind_binary_exists.stat.exists + +- name: Download SHA256SUMS file + get_url: + url: "https://bitcoinknots.org/files/{{ bitcoin_version_major }}.x/{{ bitcoin_knots_version_short }}/SHA256SUMS" + dest: "/tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS" + mode: '0644' + when: not bitcoind_binary_exists.stat.exists + +- name: Download SHA256SUMS.asc signature file + get_url: + url: "https://bitcoinknots.org/files/{{ bitcoin_version_major }}.x/{{ bitcoin_knots_version_short }}/SHA256SUMS.asc" + dest: "/tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS.asc" + mode: '0644' + when: not bitcoind_binary_exists.stat.exists + +- name: Verify PGP signature on SHA256SUMS file + command: gpg --verify /tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS.asc /tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS + register: sha256sums_verification + changed_when: false + failed_when: false # Don't fail here - check for 'Good signature' in next task + when: not bitcoind_binary_exists.stat.exists + + +- name: Display SHA256SUMS verification result + debug: + msg: "{{ sha256sums_verification.stdout_lines + sha256sums_verification.stderr_lines }}" + when: not bitcoind_binary_exists.stat.exists + +- name: Fail if SHA256SUMS signature verification failed + fail: + msg: "SHA256SUMS signature verification failed. Aborting build." + when: not bitcoind_binary_exists.stat.exists and ('Good signature' not in sha256sums_verification.stdout and 'Good signature' not in sha256sums_verification.stderr) + +- name: Remove any existing tarball to force fresh download + file: + path: /tmp/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz + state: absent + when: not bitcoind_binary_exists.stat.exists + +- name: Download Bitcoin Knots source tarball + get_url: + url: "{{ bitcoin_source_tarball_url }}" + dest: "/tmp/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz" + mode: '0644' + validate_certs: yes + force: yes + when: not bitcoind_binary_exists.stat.exists + +- name: Calculate SHA256 checksum of downloaded tarball + command: sha256sum /tmp/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz + register: tarball_checksum + changed_when: false + when: not bitcoind_binary_exists.stat.exists + +- name: Extract expected checksum from SHA256SUMS file + shell: grep "bitcoin-{{ bitcoin_knots_version_short }}.tar.gz" /tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS | awk '{print $1}' + register: expected_checksum + changed_when: false + when: not bitcoind_binary_exists.stat.exists + failed_when: expected_checksum.stdout == "" + +- name: Display checksum comparison + debug: + msg: + - "Expected: {{ expected_checksum.stdout | trim }}" + - "Actual: {{ tarball_checksum.stdout.split()[0] }}" + when: not bitcoind_binary_exists.stat.exists + +- name: Verify tarball checksum matches SHA256SUMS + fail: + msg: "Tarball checksum mismatch! Expected {{ expected_checksum.stdout | trim }}, got {{ tarball_checksum.stdout.split()[0] }}" + when: not bitcoind_binary_exists.stat.exists and expected_checksum.stdout | trim != tarball_checksum.stdout.split()[0] + +- name: Remove existing source directory if it exists (to force fresh extraction) + file: + path: "{{ bitcoin_knots_source_dir }}" + state: absent + when: not bitcoind_binary_exists.stat.exists + +- name: Remove extracted directory if it exists (from previous runs) + file: + path: "{{ bitcoin_knots_dir }}/bitcoin-{{ bitcoin_knots_version_short }}" + state: absent + when: not bitcoind_binary_exists.stat.exists + +- name: Extract verified source tarball + unarchive: + src: /tmp/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz + dest: "{{ bitcoin_knots_dir }}" + remote_src: yes + when: not bitcoind_binary_exists.stat.exists + +- name: Check if extracted directory exists + stat: + path: "{{ bitcoin_knots_dir }}/bitcoin-{{ bitcoin_knots_version_short }}" + register: extracted_dir_stat + changed_when: false + when: not bitcoind_binary_exists.stat.exists + +- name: Rename extracted directory to expected name + command: mv "{{ bitcoin_knots_dir }}/bitcoin-{{ bitcoin_knots_version_short }}" "{{ bitcoin_knots_source_dir }}" + when: not bitcoind_binary_exists.stat.exists and extracted_dir_stat.stat.exists + +- name: Check if CMakeLists.txt exists + stat: + path: "{{ bitcoin_knots_source_dir }}/CMakeLists.txt" + register: cmake_exists + changed_when: false + when: not bitcoind_binary_exists.stat.exists + +- name: Create CMake build directory + file: + path: "{{ bitcoin_knots_source_dir }}/build" + state: directory + mode: '0755' + when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) + +- name: Configure Bitcoin Knots build with CMake + command: > + cmake + -DCMAKE_INSTALL_PREFIX={{ bitcoin_build_prefix }} + -DBUILD_BITCOIN_WALLET=OFF + -DCMAKE_BUILD_TYPE=Release + -DWITH_ZMQ=ON + .. + args: + chdir: "{{ bitcoin_knots_source_dir }}/build" + when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) + register: configure_result + changed_when: true + +- name: Verify CMake enabled ZMQ + shell: | + set -e + cd "{{ bitcoin_knots_source_dir }}/build" + cmake -LAH .. | grep -iE 'ZMQ|WITH_ZMQ|ENABLE_ZMQ|USE_ZMQ' + when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) + register: zmq_check + changed_when: false + +- name: Fail if CMakeLists.txt not found + fail: + msg: "CMakeLists.txt not found in {{ bitcoin_knots_source_dir }}. Cannot build Bitcoin Knots." + when: not bitcoind_binary_exists.stat.exists and not (cmake_exists.stat.exists | default(false)) + +- name: Build Bitcoin Knots with CMake (this may take 30-60+ minutes) + command: cmake --build . -j{{ bitcoin_build_jobs }} + args: + chdir: "{{ bitcoin_knots_source_dir }}/build" + when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) + async: 3600 + poll: 0 + register: build_result + changed_when: true + +- name: Check build status + async_status: + jid: "{{ build_result.ansible_job_id }}" + register: build_job_result + until: build_job_result.finished + retries: 120 + delay: 60 + when: not bitcoind_binary_exists.stat.exists and build_result.ansible_job_id is defined + +- name: Fail if build failed + fail: + msg: "Bitcoin Knots build failed: {{ build_job_result.msg }}" + when: not bitcoind_binary_exists.stat.exists and build_result.ansible_job_id is defined and build_job_result.failed | default(false) + +- name: Install Bitcoin Knots binaries + command: cmake --install . + args: + chdir: "{{ bitcoin_knots_source_dir }}/build" + when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) + changed_when: true + +- name: Verify bitcoind binary exists + stat: + path: "{{ bitcoin_build_prefix }}/bin/bitcoind" + register: bitcoind_installed + changed_when: false + +- name: Verify bitcoin-cli binary exists + stat: + path: "{{ bitcoin_build_prefix }}/bin/bitcoin-cli" + register: bitcoin_cli_installed + changed_when: false + +- name: Fail if binaries not found + fail: + msg: "Bitcoin Knots binaries not found after installation" + when: not bitcoind_installed.stat.exists or not bitcoin_cli_installed.stat.exists diff --git a/ansible/roles/bitcoin_knots/tasks/configure.yml b/ansible/roles/bitcoin_knots/tasks/configure.yml new file mode 100644 index 0000000..52fd3fd --- /dev/null +++ b/ansible/roles/bitcoin_knots/tasks/configure.yml @@ -0,0 +1,20 @@ +--- +# Ownership copied verbatim from the playbook this replaces; verified +# mechanically against `git show HEAD:` rather than retyped from memory. +- name: Create bitcoin.conf configuration file + ansible.builtin.template: + src: bitcoin.conf.j2 + dest: "{{ bitcoin_conf_dir }}/bitcoin.conf" + owner: "{{ bitcoin_user }}" + group: "{{ bitcoin_group }}" + mode: '0640' + notify: Restart bitcoind + +- name: Create systemd service file for bitcoind + ansible.builtin.template: + src: bitcoind.service.j2 + dest: /etc/systemd/system/bitcoind.service + owner: root + group: root + mode: '0644' + notify: Restart bitcoind diff --git a/ansible/roles/bitcoin_knots/tasks/healthcheck.yml b/ansible/roles/bitcoin_knots/tasks/healthcheck.yml new file mode 100644 index 0000000..34b5529 --- /dev/null +++ b/ansible/roles/bitcoin_knots/tasks/healthcheck.yml @@ -0,0 +1,56 @@ +--- +# Everything here answers "is bitcoind healthy" and records the answer. The +# Uptime Kuma specifics that used to follow — an embedded Python script creating +# monitors over the API, a /tmp credentials file, push-URL extraction and a +# systemd Environment= rewrite — are gone. Where it reports is now one variable, +# healthcheck_push_url. See the role README. +- name: Install curl for health check script + apt: + name: curl + state: present + +- name: Create Bitcoin Knots health check script + ansible.builtin.template: + src: healthcheck.sh.j2 + dest: /usr/local/bin/bitcoin-knots-healthcheck-push.sh + owner: root + group: root + mode: '0755' + validate: "bash -n %s" + +- name: Create systemd service for Bitcoin Knots health check + ansible.builtin.template: + src: healthcheck.service.j2 + dest: /etc/systemd/system/bitcoin-knots-healthcheck.service + owner: root + group: root + mode: '0644' + +- name: Create systemd timer for Bitcoin Knots health check + ansible.builtin.template: + src: healthcheck.timer.j2 + dest: /etc/systemd/system/bitcoin-knots-healthcheck.timer + owner: root + group: root + mode: '0644' + +- name: Reload systemd daemon for health check + systemd: + daemon_reload: yes + +- name: Enable and restart the Bitcoin Knots health check timer + systemd: + name: bitcoin-knots-healthcheck.timer + enabled: yes + state: restarted + daemon_reload: yes + +# Runs the check once, which is both a smoke test and the thing that actually +# arms the timer. This timer is OnBootSec + OnUnitActiveSec with no OnCalendar: +# OnBootSec elapses once, and OnUnitActiveSec needs the SERVICE to have run this +# boot to have anything to schedule from. Restarting the timer does not supply +# that reference; running the service does. The live timer had last fired on +# 2026-08-09 while still reporting `active` and `enabled`. +- name: Run the Bitcoin Knots health check once to arm the timer + command: systemctl start bitcoin-knots-healthcheck.service + changed_when: false diff --git a/ansible/roles/bitcoin_knots/tasks/install.yml b/ansible/roles/bitcoin_knots/tasks/install.yml new file mode 100644 index 0000000..3003b6e --- /dev/null +++ b/ansible/roles/bitcoin_knots/tasks/install.yml @@ -0,0 +1,104 @@ +--- +- name: Calculate dbcache as a share of system RAM + set_fact: + bitcoin_dbcache_mb: "{{ (ansible_memtotal_mb | float * 0.9) | int }}" + when: bitcoin_dbcache_mb_override | string | length == 0 + +- name: Use the explicit dbcache override + set_fact: + bitcoin_dbcache_mb: "{{ bitcoin_dbcache_mb_override }}" + when: bitcoin_dbcache_mb_override | string | length > 0 + changed_when: false + +- name: Display calculated dbcache value + debug: + msg: "Setting dbcache to {{ bitcoin_dbcache_mb }} MB (90% of {{ ansible_memtotal_mb }} MB total RAM)" + + +- name: Install build dependencies + apt: + name: + - build-essential + - libtool + - autotools-dev + - automake + - pkg-config + - bsdmainutils + - python3 + - python3-pip + - libevent-dev + - libboost-system-dev + - libboost-filesystem-dev + - libboost-test-dev + - libboost-thread-dev + - libboost-chrono-dev + - libboost-program-options-dev + - libboost-dev + - libssl-dev + - libdb-dev + - libminiupnpc-dev + - libzmq3-dev + - libnatpmp-dev + - libsqlite3-dev + - git + - curl + - wget + - cmake + state: present + update_cache: yes + +- name: Create bitcoin group + group: + name: "{{ bitcoin_group }}" + system: yes + state: present + +- name: Create bitcoin user + user: + name: "{{ bitcoin_user }}" + group: "{{ bitcoin_group }}" + system: yes + shell: /usr/sbin/nologin + home: "{{ bitcoin_data_dir }}" + create_home: yes + state: present + +- name: Create bitcoin-knots directory + file: + path: "{{ bitcoin_knots_dir }}" + state: directory + owner: root + group: root + mode: '0755' + +- name: Create bitcoin-knots source directory + file: + path: "{{ bitcoin_knots_source_dir }}" + state: directory + owner: root + group: root + mode: '0755' + +- name: Create bitcoin data directory (for config, logs, wallets) + file: + path: "{{ bitcoin_data_dir }}" + state: directory + owner: "{{ bitcoin_user }}" + group: "{{ bitcoin_group }}" + mode: '0750' + +- name: Create bitcoin large data directory (for blockchain) + file: + path: "{{ bitcoin_large_data_dir }}" + state: directory + owner: "{{ bitcoin_user }}" + group: "{{ bitcoin_group }}" + mode: '0750' + +- name: Create bitcoin config directory + file: + path: "{{ bitcoin_conf_dir }}" + state: directory + owner: root + group: root + mode: '0755' diff --git a/ansible/roles/bitcoin_knots/tasks/main.yml b/ansible/roles/bitcoin_knots/tasks/main.yml new file mode 100644 index 0000000..ab869a7 --- /dev/null +++ b/ansible/roles/bitcoin_knots/tasks/main.yml @@ -0,0 +1,8 @@ +--- +# import_tasks, not include_tasks: static imports stay visible to --list-tasks, +# which is how this conversion was verified against the playbook it replaced. +- ansible.builtin.import_tasks: install.yml +- ansible.builtin.import_tasks: build.yml +- ansible.builtin.import_tasks: configure.yml +- ansible.builtin.import_tasks: service.yml +- ansible.builtin.import_tasks: healthcheck.yml diff --git a/ansible/roles/bitcoin_knots/tasks/service.yml b/ansible/roles/bitcoin_knots/tasks/service.yml new file mode 100644 index 0000000..6f6149e --- /dev/null +++ b/ansible/roles/bitcoin_knots/tasks/service.yml @@ -0,0 +1,37 @@ +--- +- name: Reload systemd daemon + systemd: + daemon_reload: yes + +- name: Enable and start bitcoind service + systemd: + name: bitcoind + enabled: yes + state: started + +- name: Wait for bitcoind RPC to be available + uri: + url: "http://{{ bitcoin_rpc_bind }}:{{ bitcoin_rpc_port }}" + method: POST + body_format: json + body: + jsonrpc: "1.0" + id: "healthcheck" + method: "getblockchaininfo" + params: [] + user: "{{ bitcoin_rpc_user }}" + password: "{{ bitcoin_rpc_password }}" + status_code: 200 + timeout: 10 + register: rpc_check + until: rpc_check.status == 200 + retries: 30 + delay: 5 + ignore_errors: yes + +- name: Display RPC connection status + debug: + msg: "Bitcoin Knots RPC is {{ 'available' if rpc_check.status == 200 else 'not yet available' }}" + +# ═════════════════════════════════════════════════════════════════════════ +# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. diff --git a/ansible/roles/bitcoin_knots/templates/bitcoin.conf.j2 b/ansible/roles/bitcoin_knots/templates/bitcoin.conf.j2 new file mode 100644 index 0000000..5277a3c --- /dev/null +++ b/ansible/roles/bitcoin_knots/templates/bitcoin.conf.j2 @@ -0,0 +1,67 @@ +# Bitcoin Knots Configuration +# Generated by Ansible + +# Data directory (blockchain storage) +datadir={{ bitcoin_large_data_dir }} + +# RPC Configuration +server=1 +rpcuser={{ bitcoin_rpc_user }} +rpcpassword={{ bitcoin_rpc_password }} +rpcbind={{ bitcoin_rpc_bind }} +rpcport={{ bitcoin_rpc_port }} +rpcallowip=0.0.0.0/0 + +# Network Configuration +listen=1 +port={{ bitcoin_p2p_port }} +maxconnections={{ bitcoin_max_connections }} + +# Performance +dbcache={{ bitcoin_dbcache_mb }} + +# Transaction Index (optional) +{% if bitcoin_enable_txindex %} +txindex=1 +{% endif %} + +{# The live node carries this comment and the template never produced it, so a + run would have silently deleted it. Harmless in itself, but matching it keeps + this task at `ok` - which means any future `changed` here is a real signal + rather than known noise. #} +# Pruning (optional) + +# Logging +logtimestamps=1 +{% if bitcoin_logfile %} +logfile={{ bitcoin_logfile }} +{% else %} +printtoconsole=1 +{% endif %} + +# ZMQ Configuration +{% if bitcoin_zmq_enabled | default(false) %} +zmqpubrawblock={{ bitcoin_zmq_bind }}:{{ bitcoin_zmq_port_rawblock }} +zmqpubrawtx={{ bitcoin_zmq_bind }}:{{ bitcoin_zmq_port_rawtx }} +zmqpubhashblock={{ bitcoin_zmq_bind }}:{{ bitcoin_zmq_port_hashblock }} +zmqpubhashtx={{ bitcoin_zmq_bind }}:{{ bitcoin_zmq_port_hashtx }} +{% endif %} + +# Security +disablewallet=1 +{% if bitcoin_datum_gateway_enabled %} + +{# These were hand-added on the node and were NOT in this template, so running + the playbook would have stripped them. blocknotify is how datum_gateway + learns a new block landed; without it solo mining keeps grinding on a stale + template - a silent failure that costs money rather than raising an error. + Kept as a Jinja comment so the explanation stays in the repo and out of the + deployed config. #} +# Specific for DATUM gateway +blockmaxsize={{ bitcoin_blockmaxsize }} +blockmaxweight={{ bitcoin_blockmaxweight }} +blocknotify={{ bitcoin_blocknotify }} +maxmempool={{ bitcoin_maxmempool }} +blockreconstructionextratxn={{ bitcoin_blockreconstructionextratxn }} + +{% endif %} diff --git a/ansible/roles/bitcoin_knots/templates/bitcoind.service.j2 b/ansible/roles/bitcoin_knots/templates/bitcoind.service.j2 new file mode 100644 index 0000000..ac1140f --- /dev/null +++ b/ansible/roles/bitcoin_knots/templates/bitcoind.service.j2 @@ -0,0 +1,17 @@ +[Unit] +Description=Bitcoin Knots daemon +After=network.target + +[Service] +Type=simple +User={{ bitcoin_user }} +Group={{ bitcoin_group }} +ExecStart={{ bitcoin_build_prefix }}/bin/bitcoind -conf={{ bitcoin_conf_dir }}/bitcoin.conf +Restart=always +RestartSec=10 +TimeoutStopSec=600 +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/bitcoin_knots/templates/healthcheck.service.j2 b/ansible/roles/bitcoin_knots/templates/healthcheck.service.j2 new file mode 100644 index 0000000..a2ba83d --- /dev/null +++ b/ansible/roles/bitcoin_knots/templates/healthcheck.service.j2 @@ -0,0 +1,14 @@ +[Unit] +Description=Bitcoin Knots Health Check +After=network.target bitcoind.service + +[Service] +Type=oneshot +User=root +ExecStart=/usr/local/bin/bitcoin-knots-healthcheck-push.sh +Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/bitcoin_knots/templates/healthcheck.sh.j2 b/ansible/roles/bitcoin_knots/templates/healthcheck.sh.j2 new file mode 100644 index 0000000..4e6ea9d --- /dev/null +++ b/ansible/roles/bitcoin_knots/templates/healthcheck.sh.j2 @@ -0,0 +1,62 @@ +#!/bin/bash +# Bitcoin Knots health check — managed by Ansible (roles/bitcoin_knots) +# +# The exit code is the answer and systemd keeps it: +# systemctl is-failed bitcoin-knots-healthcheck.service +# Reporting anywhere else is optional and generic. +# +# + +RPC_HOST="{{ bitcoin_rpc_bind }}" +RPC_PORT={{ bitcoin_rpc_port }} +RPC_USER="{{ bitcoin_rpc_user }}" +RPC_PASSWORD="{{ bitcoin_rpc_password }}" +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" + +# Check if bitcoind RPC is responding +check_bitcoind() { + local response + response=$(curl -s --max-time 30 \ + --user "${RPC_USER}:${RPC_PASSWORD}" \ + --data-binary '{"jsonrpc":"1.0","id":"healthcheck","method":"getblockchaininfo","params":[]}' \ + --header 'Content-Type: application/json' \ + "http://${RPC_HOST}:${RPC_PORT}" 2>&1) + + if [ $? -eq 0 ]; then + # Check if response contains a non-null error + # Successful responses have "error": null, failures have "error": {...} + if echo "$response" | grep -q '"error":null\|"error": null'; then + return 0 + else + return 1 + fi + else + return 1 + fi +} + +report() { + local status=$1 + local msg=$2 + + # No push URL is normal, not an error: the exit code below is still a + # complete answer for anything reading unit state. + [ -n "$PUSH_URL" ] || return 0 + + # URL encode spaces in message + local encoded_msg="${msg// /%20}" + + if ! curl -s --max-time 10 --retry 2 -o /dev/null \ + "${PUSH_URL}?status=${status}&msg=${encoded_msg}&ping="; then + return 1 + fi +} + +# Main health check +if check_bitcoind; then + report "up" "OK" + exit 0 +else + report "down" "bitcoind RPC not responding" + exit 1 +fi diff --git a/ansible/roles/bitcoin_knots/templates/healthcheck.timer.j2 b/ansible/roles/bitcoin_knots/templates/healthcheck.timer.j2 new file mode 100644 index 0000000..d5857ab --- /dev/null +++ b/ansible/roles/bitcoin_knots/templates/healthcheck.timer.j2 @@ -0,0 +1,11 @@ +[Unit] +Description=Bitcoin Knots Health Check Timer +Requires=bitcoind.service + +[Timer] +OnBootSec=1min +OnUnitActiveSec=1min +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/services/bitcoin-knots/bitcoin_knots_vars.yml b/ansible/services/bitcoin-knots/bitcoin_knots_vars.yml deleted file mode 100644 index c9bd7ca..0000000 --- a/ansible/services/bitcoin-knots/bitcoin_knots_vars.yml +++ /dev/null @@ -1,38 +0,0 @@ -# Bitcoin Knots Configuration Variables - -# Version - REQUIRED: Specify exact version/tag to build -bitcoin_knots_version: "v29.2.knots20251110" # Must specify exact version/tag -bitcoin_knots_version_short: "29.2.knots20251110" # Version without 'v' prefix (for tarball URLs) - -# Directories -bitcoin_knots_dir: /opt/bitcoin-knots -bitcoin_knots_source_dir: "{{ bitcoin_knots_dir }}/source" -bitcoin_data_dir: /var/lib/bitcoin # Standard location for config, logs, wallets -bitcoin_large_data_dir: /mnt/knots_data # Custom location for blockchain data (blocks, chainstate) -bitcoin_conf_dir: /etc/bitcoin - -# Network -bitcoin_rpc_port: 8332 -bitcoin_p2p_port: 8333 -bitcoin_rpc_bind: "0.0.0.0" - -# Build options -bitcoin_build_jobs: 4 # Parallel build jobs (-j flag), adjust based on CPU cores -bitcoin_build_prefix: /usr/local - -# Configuration options -bitcoin_enable_txindex: true # Set to true if transaction index needed (REQUIRED for Electrum servers like Electrs/ElectrumX) -bitcoin_max_connections: 125 -# dbcache will be calculated as 90% of host RAM automatically in playbook - -# ZMQ Configuration -bitcoin_zmq_enabled: true -bitcoin_zmq_bind: "tcp://0.0.0.0" -bitcoin_zmq_port_rawblock: 28332 -bitcoin_zmq_port_rawtx: 28333 -bitcoin_zmq_port_hashblock: 28334 -bitcoin_zmq_port_hashtx: 28335 - -# Service user -bitcoin_user: bitcoin -bitcoin_group: bitcoin diff --git a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml index 1ae5823..bc073bd 100644 --- a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +++ b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml @@ -1,3 +1,11 @@ +--- +# Bitcoin Knots: full node built from source, with PGP signature and SHA256 +# verification of the release tarball. The build is guarded by a binary-exists +# check, so a converged host skips the whole 30-60 minute compile. +# +# The chain lives in bitcoin_large_data_dir (/mnt/knots_data, ~875 GB). Nothing +# here touches it; the only `state: absent` tasks target the build tree under +# /opt/bitcoin-knots and run only when the binary is missing. - name: Build and Deploy Bitcoin Knots from Source hosts: bitcoin become: yes @@ -5,740 +13,12 @@ - ../../infra_vars.yml - ../../services_config.yml - ../../infra_secrets.yml - - ./bitcoin_knots_vars.yml vars: - bitcoin_repo_url: "https://github.com/bitcoinknots/bitcoin.git" - bitcoin_sigs_base_url: "https://raw.githubusercontent.com/bitcoinknots/guix.sigs/knots" - bitcoin_version_major: "{{ bitcoin_knots_version_short | regex_replace('^(\\d+)\\..*', '\\1') }}" - bitcoin_source_tarball_url: "https://bitcoinknots.org/files/{{ bitcoin_version_major }}.x/{{ bitcoin_knots_version_short }}/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - - name: Calculate 90% of system RAM for dbcache - set_fact: - bitcoin_dbcache_mb: "{{ (ansible_memtotal_mb | float * 0.9) | int }}" - changed_when: false - - - name: Display calculated dbcache value - debug: - msg: "Setting dbcache to {{ bitcoin_dbcache_mb }} MB (90% of {{ ansible_memtotal_mb }} MB total RAM)" - - - - name: Install build dependencies - apt: - name: - - build-essential - - libtool - - autotools-dev - - automake - - pkg-config - - bsdmainutils - - python3 - - python3-pip - - libevent-dev - - libboost-system-dev - - libboost-filesystem-dev - - libboost-test-dev - - libboost-thread-dev - - libboost-chrono-dev - - libboost-program-options-dev - - libboost-dev - - libssl-dev - - libdb-dev - - libminiupnpc-dev - - libzmq3-dev - - libnatpmp-dev - - libsqlite3-dev - - git - - curl - - wget - - cmake - state: present - update_cache: yes - - - name: Create bitcoin group - group: - name: "{{ bitcoin_group }}" - system: yes - state: present - - - name: Create bitcoin user - user: - name: "{{ bitcoin_user }}" - group: "{{ bitcoin_group }}" - system: yes - shell: /usr/sbin/nologin - home: "{{ bitcoin_data_dir }}" - create_home: yes - state: present - - - name: Create bitcoin-knots directory - file: - path: "{{ bitcoin_knots_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create bitcoin-knots source directory - file: - path: "{{ bitcoin_knots_source_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create bitcoin data directory (for config, logs, wallets) - file: - path: "{{ bitcoin_data_dir }}" - state: directory - owner: "{{ bitcoin_user }}" - group: "{{ bitcoin_group }}" - mode: '0750' - - - name: Create bitcoin large data directory (for blockchain) - file: - path: "{{ bitcoin_large_data_dir }}" - state: directory - owner: "{{ bitcoin_user }}" - group: "{{ bitcoin_group }}" - mode: '0750' - - - name: Create bitcoin config directory - file: - path: "{{ bitcoin_conf_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Check if bitcoind binary already exists - stat: - path: "{{ bitcoin_build_prefix }}/bin/bitcoind" - register: bitcoind_binary_exists - changed_when: false - - - name: Install gnupg for signature verification - apt: - name: gnupg - state: present - when: not bitcoind_binary_exists.stat.exists - - - name: Import Luke Dashjr's Bitcoin Knots signing key - command: gpg --keyserver hkps://keyserver.ubuntu.com --recv-keys 90C8019E36C2E964 - register: key_import - changed_when: "'already in secret keyring' not in key_import.stdout and 'already in public keyring' not in key_import.stdout" - when: not bitcoind_binary_exists.stat.exists - failed_when: key_import.rc != 0 - - - name: Display imported key fingerprint - command: gpg --fingerprint 90C8019E36C2E964 - register: key_fingerprint - changed_when: false - when: not bitcoind_binary_exists.stat.exists - - - name: Download SHA256SUMS file - get_url: - url: "https://bitcoinknots.org/files/{{ bitcoin_version_major }}.x/{{ bitcoin_knots_version_short }}/SHA256SUMS" - dest: "/tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS" - mode: '0644' - when: not bitcoind_binary_exists.stat.exists - - - name: Download SHA256SUMS.asc signature file - get_url: - url: "https://bitcoinknots.org/files/{{ bitcoin_version_major }}.x/{{ bitcoin_knots_version_short }}/SHA256SUMS.asc" - dest: "/tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS.asc" - mode: '0644' - when: not bitcoind_binary_exists.stat.exists - - - name: Verify PGP signature on SHA256SUMS file - command: gpg --verify /tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS.asc /tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS - register: sha256sums_verification - changed_when: false - failed_when: false # Don't fail here - check for 'Good signature' in next task - when: not bitcoind_binary_exists.stat.exists - - - - name: Display SHA256SUMS verification result - debug: - msg: "{{ sha256sums_verification.stdout_lines + sha256sums_verification.stderr_lines }}" - when: not bitcoind_binary_exists.stat.exists - - - name: Fail if SHA256SUMS signature verification failed - fail: - msg: "SHA256SUMS signature verification failed. Aborting build." - when: not bitcoind_binary_exists.stat.exists and ('Good signature' not in sha256sums_verification.stdout and 'Good signature' not in sha256sums_verification.stderr) - - - name: Remove any existing tarball to force fresh download - file: - path: /tmp/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz - state: absent - when: not bitcoind_binary_exists.stat.exists - - - name: Download Bitcoin Knots source tarball - get_url: - url: "{{ bitcoin_source_tarball_url }}" - dest: "/tmp/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz" - mode: '0644' - validate_certs: yes - force: yes - when: not bitcoind_binary_exists.stat.exists - - - name: Calculate SHA256 checksum of downloaded tarball - command: sha256sum /tmp/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz - register: tarball_checksum - changed_when: false - when: not bitcoind_binary_exists.stat.exists - - - name: Extract expected checksum from SHA256SUMS file - shell: grep "bitcoin-{{ bitcoin_knots_version_short }}.tar.gz" /tmp/bitcoin-knots-{{ bitcoin_knots_version_short }}-SHA256SUMS | awk '{print $1}' - register: expected_checksum - changed_when: false - when: not bitcoind_binary_exists.stat.exists - failed_when: expected_checksum.stdout == "" - - - name: Display checksum comparison - debug: - msg: - - "Expected: {{ expected_checksum.stdout | trim }}" - - "Actual: {{ tarball_checksum.stdout.split()[0] }}" - when: not bitcoind_binary_exists.stat.exists - - - name: Verify tarball checksum matches SHA256SUMS - fail: - msg: "Tarball checksum mismatch! Expected {{ expected_checksum.stdout | trim }}, got {{ tarball_checksum.stdout.split()[0] }}" - when: not bitcoind_binary_exists.stat.exists and expected_checksum.stdout | trim != tarball_checksum.stdout.split()[0] - - - name: Remove existing source directory if it exists (to force fresh extraction) - file: - path: "{{ bitcoin_knots_source_dir }}" - state: absent - when: not bitcoind_binary_exists.stat.exists - - - name: Remove extracted directory if it exists (from previous runs) - file: - path: "{{ bitcoin_knots_dir }}/bitcoin-{{ bitcoin_knots_version_short }}" - state: absent - when: not bitcoind_binary_exists.stat.exists - - - name: Extract verified source tarball - unarchive: - src: /tmp/bitcoin-{{ bitcoin_knots_version_short }}.tar.gz - dest: "{{ bitcoin_knots_dir }}" - remote_src: yes - when: not bitcoind_binary_exists.stat.exists - - - name: Check if extracted directory exists - stat: - path: "{{ bitcoin_knots_dir }}/bitcoin-{{ bitcoin_knots_version_short }}" - register: extracted_dir_stat - changed_when: false - when: not bitcoind_binary_exists.stat.exists - - - name: Rename extracted directory to expected name - command: mv "{{ bitcoin_knots_dir }}/bitcoin-{{ bitcoin_knots_version_short }}" "{{ bitcoin_knots_source_dir }}" - when: not bitcoind_binary_exists.stat.exists and extracted_dir_stat.stat.exists - - - name: Check if CMakeLists.txt exists - stat: - path: "{{ bitcoin_knots_source_dir }}/CMakeLists.txt" - register: cmake_exists - changed_when: false - when: not bitcoind_binary_exists.stat.exists - - - name: Create CMake build directory - file: - path: "{{ bitcoin_knots_source_dir }}/build" - state: directory - mode: '0755' - when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) - - - name: Configure Bitcoin Knots build with CMake - command: > - cmake - -DCMAKE_INSTALL_PREFIX={{ bitcoin_build_prefix }} - -DBUILD_BITCOIN_WALLET=OFF - -DCMAKE_BUILD_TYPE=Release - -DWITH_ZMQ=ON - .. - args: - chdir: "{{ bitcoin_knots_source_dir }}/build" - when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) - register: configure_result - changed_when: true - - - name: Verify CMake enabled ZMQ - shell: | - set -e - cd "{{ bitcoin_knots_source_dir }}/build" - cmake -LAH .. | grep -iE 'ZMQ|WITH_ZMQ|ENABLE_ZMQ|USE_ZMQ' - when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) - register: zmq_check - changed_when: false - - - name: Fail if CMakeLists.txt not found - fail: - msg: "CMakeLists.txt not found in {{ bitcoin_knots_source_dir }}. Cannot build Bitcoin Knots." - when: not bitcoind_binary_exists.stat.exists and not (cmake_exists.stat.exists | default(false)) - - - name: Build Bitcoin Knots with CMake (this may take 30-60+ minutes) - command: cmake --build . -j{{ bitcoin_build_jobs }} - args: - chdir: "{{ bitcoin_knots_source_dir }}/build" - when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) - async: 3600 - poll: 0 - register: build_result - changed_when: true - - - name: Check build status - async_status: - jid: "{{ build_result.ansible_job_id }}" - register: build_job_result - until: build_job_result.finished - retries: 120 - delay: 60 - when: not bitcoind_binary_exists.stat.exists and build_result.ansible_job_id is defined - - - name: Fail if build failed - fail: - msg: "Bitcoin Knots build failed: {{ build_job_result.msg }}" - when: not bitcoind_binary_exists.stat.exists and build_result.ansible_job_id is defined and build_job_result.failed | default(false) - - - name: Install Bitcoin Knots binaries - command: cmake --install . - args: - chdir: "{{ bitcoin_knots_source_dir }}/build" - when: not bitcoind_binary_exists.stat.exists and cmake_exists.stat.exists | default(false) - changed_when: true - - - name: Verify bitcoind binary exists - stat: - path: "{{ bitcoin_build_prefix }}/bin/bitcoind" - register: bitcoind_installed - changed_when: false - - - name: Verify bitcoin-cli binary exists - stat: - path: "{{ bitcoin_build_prefix }}/bin/bitcoin-cli" - register: bitcoin_cli_installed - changed_when: false - - - name: Fail if binaries not found - fail: - msg: "Bitcoin Knots binaries not found after installation" - when: not bitcoind_installed.stat.exists or not bitcoin_cli_installed.stat.exists - - - name: Create bitcoin.conf configuration file - copy: - dest: "{{ bitcoin_conf_dir }}/bitcoin.conf" - content: | - # Bitcoin Knots Configuration - # Generated by Ansible - - # Data directory (blockchain storage) - datadir={{ bitcoin_large_data_dir }} - - # RPC Configuration - server=1 - rpcuser={{ bitcoin_rpc_user }} - rpcpassword={{ bitcoin_rpc_password }} - rpcbind={{ bitcoin_rpc_bind }} - rpcport={{ bitcoin_rpc_port }} - rpcallowip=0.0.0.0/0 - - # Network Configuration - listen=1 - port={{ bitcoin_p2p_port }} - maxconnections={{ bitcoin_max_connections }} - - # Performance - dbcache={{ bitcoin_dbcache_mb }} - - # Transaction Index (optional) - {% if bitcoin_enable_txindex %} - txindex=1 - {% endif %} - - # Logging (to journald via systemd) - logtimestamps=1 - printtoconsole=1 - - # ZMQ Configuration - {% if bitcoin_zmq_enabled | default(false) %} - zmqpubrawblock={{ bitcoin_zmq_bind }}:{{ bitcoin_zmq_port_rawblock }} - zmqpubrawtx={{ bitcoin_zmq_bind }}:{{ bitcoin_zmq_port_rawtx }} - zmqpubhashblock={{ bitcoin_zmq_bind }}:{{ bitcoin_zmq_port_hashblock }} - zmqpubhashtx={{ bitcoin_zmq_bind }}:{{ bitcoin_zmq_port_hashtx }} - {% endif %} - - # Security - disablewallet=1 - owner: "{{ bitcoin_user }}" - group: "{{ bitcoin_group }}" - mode: '0640' - notify: Restart bitcoind - - - name: Create systemd service file for bitcoind - copy: - dest: /etc/systemd/system/bitcoind.service - content: | - [Unit] - Description=Bitcoin Knots daemon - After=network.target - - [Service] - Type=simple - User={{ bitcoin_user }} - Group={{ bitcoin_group }} - ExecStart={{ bitcoin_build_prefix }}/bin/bitcoind -conf={{ bitcoin_conf_dir }}/bitcoin.conf - Restart=always - RestartSec=10 - TimeoutStopSec=600 - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - notify: Restart bitcoind - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start bitcoind service - systemd: - name: bitcoind - enabled: yes - state: started - - - name: Wait for bitcoind RPC to be available - uri: - url: "http://{{ bitcoin_rpc_bind }}:{{ bitcoin_rpc_port }}" - method: POST - body_format: json - body: - jsonrpc: "1.0" - id: "healthcheck" - method: "getblockchaininfo" - params: [] - user: "{{ bitcoin_rpc_user }}" - password: "{{ bitcoin_rpc_password }}" - status_code: 200 - timeout: 10 - register: rpc_check - until: rpc_check.status == 200 - retries: 30 - delay: 5 - ignore_errors: yes - - - name: Display RPC connection status - debug: - msg: "Bitcoin Knots RPC is {{ 'available' if rpc_check.status == 200 else 'not yet available' }}" - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create Bitcoin Knots health check and push script - when: uptime_kuma_enabled | default(false) - copy: - dest: /usr/local/bin/bitcoin-knots-healthcheck-push.sh - content: | - #!/bin/bash - # - # Bitcoin Knots Health Check and Push to Uptime Kuma - # Checks if bitcoind RPC is responding and pushes status to Uptime Kuma - # - - RPC_HOST="{{ bitcoin_rpc_bind }}" - RPC_PORT={{ bitcoin_rpc_port }} - RPC_USER="{{ bitcoin_rpc_user }}" - RPC_PASSWORD="{{ bitcoin_rpc_password }}" - UPTIME_KUMA_PUSH_URL="${UPTIME_KUMA_PUSH_URL}" - - # Check if bitcoind RPC is responding - check_bitcoind() { - local response - response=$(curl -s --max-time 30 \ - --user "${RPC_USER}:${RPC_PASSWORD}" \ - --data-binary '{"jsonrpc":"1.0","id":"healthcheck","method":"getblockchaininfo","params":[]}' \ - --header 'Content-Type: application/json' \ - "http://${RPC_HOST}:${RPC_PORT}" 2>&1) - - if [ $? -eq 0 ]; then - # Check if response contains a non-null error - # Successful responses have "error": null, failures have "error": {...} - if echo "$response" | grep -q '"error":null\|"error": null'; then - return 0 - else - return 1 - fi - else - return 1 - fi - } - - # Push status to Uptime Kuma - push_to_uptime_kuma() { - local status=$1 - local msg=$2 - - if [ -z "$UPTIME_KUMA_PUSH_URL" ]; then - echo "ERROR: UPTIME_KUMA_PUSH_URL not set" - return 1 - fi - - # URL encode spaces in message - local encoded_msg="${msg// /%20}" - - if ! curl -s --max-time 10 --retry 2 -o /dev/null \ - "${UPTIME_KUMA_PUSH_URL}?status=${status}&msg=${encoded_msg}&ping="; then - echo "ERROR: Failed to push to Uptime Kuma" - return 1 - fi - } - - # Main health check - if check_bitcoind; then - push_to_uptime_kuma "up" "OK" - exit 0 - else - push_to_uptime_kuma "down" "bitcoind RPC not responding" - exit 1 - fi - owner: root - group: root - mode: '0755' - - - name: Install curl for health check script - apt: - name: curl - state: present - - - name: Create systemd timer for Bitcoin Knots health check - copy: - dest: /etc/systemd/system/bitcoin-knots-healthcheck.timer - content: | - [Unit] - Description=Bitcoin Knots Health Check Timer - Requires=bitcoind.service - - [Timer] - OnBootSec=1min - OnUnitActiveSec=1min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Create systemd service for Bitcoin Knots health check - when: uptime_kuma_enabled | default(false) - copy: - dest: /etc/systemd/system/bitcoin-knots-healthcheck.service - content: | - [Unit] - Description=Bitcoin Knots Health Check and Push to Uptime Kuma - After=network.target bitcoind.service - - [Service] - Type=oneshot - User=root - ExecStart=/usr/local/bin/bitcoin-knots-healthcheck-push.sh - Environment=UPTIME_KUMA_PUSH_URL= - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon for health check - systemd: - daemon_reload: yes - - - name: Enable and start Bitcoin Knots health check timer - systemd: - name: bitcoin-knots-healthcheck.timer - enabled: yes - state: started - - - name: Create Uptime Kuma push monitor setup script for Bitcoin Knots - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_bitcoin_knots_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - # Load configs - with open('/tmp/ansible_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_name = config['monitor_name'] - - # Connect to Uptime Kuma - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - push_token = existing_monitor.get('pushToken') or existing_monitor.get('push_token') - if not push_token: - raise ValueError("Could not find push token for monitor") - push_url = f"{url}/api/push/{push_token}" - print(f"Push URL: {push_url}") - else: - print(f"Creating push monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.PUSH, - name=monitor_name, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - monitors = api.get_monitors() - new_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - if new_monitor: - push_token = new_monitor.get('pushToken') or new_monitor.get('push_token') - if not push_token: - raise ValueError("Could not find push token for new monitor") - push_url = f"{url}/api/push/{push_token}" - print(f"Push URL: {push_url}") - - api.disconnect() - print("SUCCESS") - - except Exception as e: - error_msg = str(e) if str(e) else repr(e) - print(f"ERROR: {error_msg}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_name: "Bitcoin Knots" - mode: '0644' - - - name: Run Uptime Kuma push monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_bitcoin_knots_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Extract push URL from monitor setup output - set_fact: - uptime_kuma_push_url: "{{ monitor_setup.stdout | regex_search('Push URL: (https?://[^\\s]+)', '\\1') | first | default('') }}" - delegate_to: localhost - become: no - when: monitor_setup.stdout is defined - - - name: Display extracted push URL - debug: - msg: "Uptime Kuma Push URL: {{ uptime_kuma_push_url }}" - when: uptime_kuma_push_url | default('') != '' - - - name: Set push URL in systemd service environment - lineinfile: - path: /etc/systemd/system/bitcoin-knots-healthcheck.service - regexp: '^Environment=UPTIME_KUMA_PUSH_URL=' - line: "Environment=UPTIME_KUMA_PUSH_URL={{ uptime_kuma_push_url }}" - state: present - insertafter: '^\[Service\]' - when: uptime_kuma_push_url | default('') != '' - - - name: Reload systemd daemon after push URL update - systemd: - daemon_reload: yes - when: uptime_kuma_push_url | default('') != '' - - - name: Restart health check timer to pick up new environment - systemd: - name: bitcoin-knots-healthcheck.timer - state: restarted - when: uptime_kuma_push_url | default('') != '' - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_bitcoin_knots_monitor.py - - /tmp/ansible_config.yml - - handlers: - - name: Restart bitcoind - when: uptime_kuma_enabled | default(false) - systemd: - name: bitcoind - state: restarted - + # Preserves the push URL this check has been reporting to. The role knows + # nothing about Uptime Kuma — this is just "a URL that accepts a ping". + healthcheck_push_url: "{{ healthcheck_push_urls.bitcoin_knots | default('') }}" + roles: + - bitcoin_knots - name: Setup public Bitcoin P2P forwarding on the edge host hosts: edge @@ -746,12 +26,6 @@ vars_files: - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - - ./bitcoin_knots_vars.yml - vars: - bitcoin_tailscale_hostname: "knots-box" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - tasks: - name: Expose Bitcoin P2P through a socket proxy ansible.builtin.include_role: @@ -759,134 +33,9 @@ vars: socket_proxy_name: bitcoin-p2p socket_proxy_description: "Bitcoin P2P" - socket_proxy_listen_port: "{{ bitcoin_p2p_port }}" - socket_proxy_upstream_host: "{{ bitcoin_tailscale_hostname }}" - # These four were added by hand on vipy and were NOT in this playbook; - # writing the unit without them would have dropped FreeBind, which lets - # the socket bind before the address is up. + socket_proxy_listen_port: "{{ service_settings.bitcoin.p2p_port }}" + socket_proxy_upstream_host: "{{ service_settings.bitcoin.tailscale_hostname }}" socket_proxy_documentation: "https://github.com/bitcoin/bitcoin" socket_proxy_free_bind: true socket_proxy_timeout_stop_sec: 5 socket_proxy_log_to_journal: true - - - name: Display public endpoint - when: uptime_kuma_enabled | default(false) - debug: - msg: "Bitcoin P2P public endpoint: {{ ansible_host }}:{{ bitcoin_p2p_port }}" - - # =========================================== - # Uptime Kuma TCP Monitor for Public P2P - # =========================================== - - name: Create Uptime Kuma TCP monitor setup script for Bitcoin P2P - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_bitcoin_p2p_tcp_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_bitcoin_p2p_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_host = config['monitor_host'] - monitor_port = config['monitor_port'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - api.add_monitor(type='group', name='services') - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing = next((m for m in monitors if m.get('name') == monitor_name), None) - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing: - print(f"Monitor '{monitor_name}' already exists (ID: {existing['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating TCP monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.PORT, - name=monitor_name, - hostname=monitor_host, - port=monitor_port, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for TCP monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_bitcoin_p2p_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_host: "{{ ansible_host }}" - monitor_port: {{ bitcoin_p2p_port }} - monitor_name: "Bitcoin Knots P2P Public" - mode: '0644' - - - name: Run Uptime Kuma TCP monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_bitcoin_p2p_tcp_monitor.py - delegate_to: localhost - become: no - register: tcp_monitor_setup - changed_when: "'SUCCESS' in tcp_monitor_setup.stdout" - ignore_errors: yes - - - name: Display TCP monitor setup output - debug: - msg: "{{ tcp_monitor_setup.stdout_lines }}" - when: tcp_monitor_setup.stdout is defined - - - name: Clean up TCP monitor temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_bitcoin_p2p_tcp_monitor.py - - /tmp/ansible_bitcoin_p2p_config.yml - diff --git a/ansible/services_config.yml b/ansible/services_config.yml index d9a9c13..e342a79 100644 --- a/ansible/services_config.yml +++ b/ansible/services_config.yml @@ -46,6 +46,12 @@ service_settings: # from the edge host. A role default cannot serve the second play, so it # lives here rather than in roles/mempool/defaults. frontend_port: 8080 + bitcoin: + # The P2P port is needed on two hosts: the bitcoin_knots role deploys the + # node on knots-box, and the socket-proxy play publishes the port from the + # edge host. A role default cannot reach that second play. + p2p_port: 8333 + tailscale_hostname: knots-box fulcrum: # Same shape as mempool: the fulcrum role deploys on fulcrum-box, and the # socket-proxy play publishes the SSL port from the edge host. A role default From a0c23ae766eb7ac2e0513dcdb7ce05786f1c6aca Mon Sep 17 00:00:00 2001 From: counterweight Date: Sun, 13 Sep 2026 18:25:41 +0200 Subject: [PATCH 53/67] datum-gateway: convert to a role, de-Uptime-Kuma the health check 802-line playbook becomes 68 lines (three plays: the role, the Caddy dashboard, the Stratum socket proxy) plus a 345-line role. datum_gateway_vars.yml is deleted; its content is the role's defaults. Verified after a real run with zero miners connected: datum-gateway restarted cleanly onto the reformatted config, deployed config.json semantically identical to what was there (pool_address bc1qvrj3g84..., pool_pass_* false, ports unchanged), health check timer firing, and the Knots side untouched - bitcoind still up since 2026-08-19 with blocknotify intact. TWO PIECES OF DRIFT WHERE THE NODE WAS RIGHT, both confirmed with the operator: - datum_mining_address: the vault held bc1qdse9dsg... while the node had been mining to bc1qvrj3g... since 2026-08-08. This is WHERE BLOCK REWARDS ARE PAID. And unlike fulcrum and bitcoin-knots, the `Restart datum-gateway` handler here was never gated, so the stale value would have applied immediately rather than sitting inert on disk. - pool_pass_workers / pool_pass_full_users: false on the node, true in the vars file. Both corrected in the vault and role defaults with notes recording why. Comparing this config needs semantics, not text: the live file is single-line JSON and the template renders pretty-printed, so a textual diff is pure noise. Rendering it and comparing parsed JSON is what surfaced both differences. config.json carries bitcoind.rpcpassword and api.admin_password, and --diff prints rendered content - so `--check --diff` put them on the terminal. The task now sets diff: false by default (-e datum_reveal_config=true to opt in). Those two should be rotated. I also mis-reported pool_pass_workers/pool_pass_full_users as exposed credentials because my masking matched "pass" in the key name. They are BOOLEANS, and mining.pool_address is a Bitcoin address, public by nature. Only the two real passwords above were exposed. `Configure cmake build` and `Compile datum_gateway` are bare command: tasks with no changed_when, so they recompile on every run. The build is reproducible - Install datum_gateway binary sees identical content and leaves the installed binary's timestamp alone - but it is wasted work each time. Documented as the idempotent floor. Ownership parity checked mechanically against `git show HEAD:` keyed by task name: 7/7 match, 9 Kuma tasks dropped. This completes Plan 6 Stage 2: all six services in the list are roles. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 367 ++++---- ansible/infra_secrets.yml | 367 ++++---- ansible/roles/datum_gateway/README.md | 73 ++ .../datum_gateway/defaults/main.yml} | 19 +- ansible/roles/datum_gateway/handlers/main.yml | 15 + .../roles/datum_gateway/tasks/configure.yml | 25 + .../roles/datum_gateway/tasks/healthcheck.yml | 50 ++ ansible/roles/datum_gateway/tasks/install.yml | 74 ++ ansible/roles/datum_gateway/tasks/main.yml | 6 + ansible/roles/datum_gateway/tasks/service.yml | 16 + .../datum_gateway/templates/config.json.j2 | 35 + .../templates/datum-gateway.service.j2 | 22 + .../templates/healthcheck.service.j2 | 14 + .../datum_gateway/templates/healthcheck.sh.j2 | 32 + .../templates/healthcheck.timer.j2 | 10 + .../deploy_datum_gateway_playbook.yml | 780 +----------------- ansible/services_config.yml | 7 + 17 files changed, 797 insertions(+), 1115 deletions(-) create mode 100644 ansible/roles/datum_gateway/README.md rename ansible/{services/datum-gateway/datum_gateway_vars.yml => roles/datum_gateway/defaults/main.yml} (57%) create mode 100644 ansible/roles/datum_gateway/handlers/main.yml create mode 100644 ansible/roles/datum_gateway/tasks/configure.yml create mode 100644 ansible/roles/datum_gateway/tasks/healthcheck.yml create mode 100644 ansible/roles/datum_gateway/tasks/install.yml create mode 100644 ansible/roles/datum_gateway/tasks/main.yml create mode 100644 ansible/roles/datum_gateway/tasks/service.yml create mode 100644 ansible/roles/datum_gateway/templates/config.json.j2 create mode 100644 ansible/roles/datum_gateway/templates/datum-gateway.service.j2 create mode 100644 ansible/roles/datum_gateway/templates/healthcheck.service.j2 create mode 100644 ansible/roles/datum_gateway/templates/healthcheck.sh.j2 create mode 100644 ansible/roles/datum_gateway/templates/healthcheck.timer.j2 diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index a0e560a..a2b0fcf 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,178 +1,191 @@ $ANSIBLE_VAULT;1.1;AES256 -35333539613236336636383331373761643732663835346539653531336662636432346132636539 -3738343939626363303936646239613531306461646431630a656362666266356534303466323962 -64626235663062393436356165323364333735396530343435373730613765346466386336653030 -6234333131383036330a663039613363656536353164323265383130333463646638396437356332 -65653330383235386563623231646535343461613833333765333336366338376264666139303635 -65666139643363356264353234343736623731656534636662616466313764633763303035343432 -32623965326532353733656533356566393834343338303833356334346533383531393435363461 -64306561663434353935386165393361613330643163323564383835386438346664323663663564 -35623032646362346262366333626461343364356136373231313635353665643838666336313766 -62663730653030326534393838343139313239656530316662323638396230323436396431306135 -64386263616633323061373331376536343261313330653061333662373034353564623934376235 -65363432316630636132613935663466366365613265326163383832613631303631616464633161 -64373661313034653338646537343164623630643039353533376537636264653261393464633765 -38376436366262373663303137343861383932356330356239326464666233303064666634623135 -66373430383061643463386265646666333962366366336534646635633365356334343238303930 -65316432386134353838333033386164643165366662393365623839616265646538333235343237 -63373236623963393937613333623630363032643365613561653638323030373435353639366433 -34326334643330336564323439656630313331623762373038343133613763643838643039656338 -61323933643764306339383733623939346466303636616339396534643434633936393563373934 -66646632336133386163613135363963363338373338323732333563643164313662633639633130 -32383263396666333639626366316630346333643039666431323164356364346566343237316166 -62343464373564303533653864653164626239383737303235353638346636356538383661643931 -38646538323431383638653532303163313032396164353139333461363066393566303237386239 -37613236376636316136346266613530363732353232313037383830313461393136306233363931 -35663963613630616530383865633330636434626230333563313362623265633637333261653632 -63303130323737633365646265393736613136383536383133663538663466643631626139663865 -34333938363430393763653937653561376362353633303962643637666162666538316366343132 -31646163626465643463326333303365313138626366363861663530343362366663323161656362 -31623361323033353238623462333031356265343239613834353864333365626636316531313238 -66343633393865316535316262376338346234633737366132313938356438383066356465303239 -64326261333935343337363331373366636438623966373439653432663030376236643664643865 -34393738343865336263666133613765353162353839393633346439356665613936643863323464 -64633462366565373532333336376666336263396362303639393061346438363035613534356334 -37376166363030393237373464356139343336333939346662336332336639396432346665373137 -63306539383830656363383539303366643335356133353738373266646463346265333733613030 -63383036343663623366653763636135326561663739303631333936313161396335373962643662 -31386331623466666262396132636265323831646333353038386131303033626435316634666662 -36353539383961623535333730653537633932613366653563323339333738643131393338313931 -33373964366638323239386130306666316333613233666337623966303830323637306534653430 -62383634303736636166376563626537346161663432616265336364336264373638306364336331 -35373862653133333637636164343030623664333536656433366563376330356431343962656162 -31646536626631666636383033336130373565376430386639313135373765383437336538313266 -65383038616139663436323833613232626531326531303937613931373566646263643634343461 -63393365333839663231623532373634643136383333373166356635356666353837383331383334 -36636465316536313765326562336539663539613036373638616633353936383866663231356262 -65313965643263666163313638626335363333623833656439633464343864333465313231326164 -30646530363161633834306633643132306562363065323032383066316533373963383763333466 -33613230643633346566396165376665366361633733366261306637303964376231303365333165 -37366633643331393537633563626164613630396663326233383263343930333232656265326633 -62373539306331623332386637646333326435393933323632313166386365656561356536663738 -35363362663937363661636437336532646437623864643463346238636331643935333264663365 -35353063643662363939396638386531386265336566373835646435353736386666646531373361 -30646136386138323530306135666333386437356430643262363234616366376335633638303133 -62316562323463346263323934363937336631656666306237626438626133346566613662363831 -65623561373231663262313763623965663036376631663662616634353664663762386666353539 -63323237303362393762343832396463343534633432626532363534396261613132323633353938 -64613333316436313063313561656561326139623736373439363363393632353564343361396666 -32623938383737623836393536393838346131343762353463386339336361346266623663353262 -63316362363736376639333638326433383662663866303638376662616335653764663631616666 -66663738636162353262373034373334303562363236373232306364393335346139396663626634 -64373735643537383230616661323238333239386330313231333833663062393832396366373337 -66313966356261383039623630376138643261393062393030346131663839346437363766333732 -63303136303037643931663431363435353261343133386332666531383835663361376165383632 -38333963653732613462323435633936353637336538396531616437393131333631663335613931 -34343236386135316336353661636266396434386563656433633935336330366162363563313738 -39303561323739306362383765303131623265626332613265323264393333356165326238633031 -33366464626537313662333166343532373735303135306563393737663536363862336232613435 -32613031396562333939303566343834323164396165373165363932323065623839643035373539 -33363362316162643264623937353066303536623962373433333430373436643862616636613537 -66653864646463653435366531373039373663333964636163633965613438366339613437613731 -34353330616337313037353633626633633666346363663764316163653635363335343237616666 -34663461616539376463613635346536366334613536326564626363663661313038636562303031 -36653837393233333234613131633735313739333263663532623563623231323239343936333831 -64313661376164323738316136623261353538323565383865376339366466393033373065613835 -34393339323839653430653335623664373534356232386262666437393234616362353438386361 -65383362346239626438346135663064643031386335633533646433326631386439353731653663 -33326633653363313232356233373531633737383530396439643139316164626566326164333439 -33393163376162623938383264653337633361643633306666646163633337366264353939353236 -39383035653036613463643633633234623439626365633761643038656339373362333238373033 -36626563373563363139386336653764623661633766343965346634653135306137393361386539 -33656261333137656632353739346462326364386330333536636634666538643461373766316165 -31383238353730356339353934663037373865363238373339346238623364383631313537386231 -66643937316361333262643230343965616663393430303661623161323361643835623862313564 -30646631396564383932633437303464666166636434303330303934613038393130666561343837 -34376164323633336432333664366631326239383463303366363763383062643565643464633634 -66313264666562626339326133623630366131333232383333333961323234366633313231376134 -62653962303630366330303537383639306263316630313862646334393531363439346233313462 -34353937303230373931616339393861323138333237633861326165346631323931633234346630 -35656666623261616564343733343764383032373733306466383763376564663665663731303539 -37373863613135326634333039333932633461613165623938623338623565313332376136343436 -36303037653935383630376362326335653466343233613833666361633661626437663563303331 -32363331663334616338643239356133396662633231396436636630343934306333343539646132 -65366263646563386563643562316337343130643661633730663664663564316535613531353165 -32666436326530323961663736306162306633393366306532613266323633373330383634656535 -38626138633335393266656638613765666665313561363231636261643936333033383533373366 -61626365636635653834306539303530383330346630363766383734346332326638316561303763 -63646639386366616230333263313130346663343663623536376664303364623064373562326466 -38383163326637633937663161393538616339326263306463353530303630333937326138363239 -62633535356230353439623234316562653237613130383832343735643661623033316264653362 -37633065313135656637616137653963623331323863363364663761663364663565623535343366 -65313164643461383436623038316466396632626661373438643533336234323132613866393134 -32366363613338646131663738643236613562663262333936656461323336303139616434333334 -38653133633232343164313661623466643061383161303564376638373138646431343634393962 -31306564396262383438643937353563353663353536373032353263636430363235313736653333 -39303336316438616365613961653737633234323131356339623265373463336662366336396364 -38333962373164383630363866343361376262316462333936616230316536396363316163653463 -37306461346136323431323265383030643232633062386433336261316333323030663234663063 -65623434613662306137396361616330333233386630313935653861333761323565646237343430 -38613139663035373563313530366632303232643030643466346531373333363337643831663736 -62636538343035643464396461326363646537393138613166323762346539333033633661383531 -62653634636666343661353565363266303430623131643733633232626230623466376337366137 -65653565626366373337626632346338396533343138393431346333346161353838386434643863 -33383935636132343764656232393335316431633937323533313338366461376438653136363231 -34643363356664373765376465333534653137383961386262633165636666353464323261396465 -37353064353530393037303930303430376530623364633832636434313264393135383536306438 -35316436616536396335393863363863653763303666653161383630336437653663636434366635 -38633633396132623663323137643338303038393061633961353064383765353735393432313766 -61343735363538316134626434326264383932393964646136306537313932316238326362646337 -66363234363334373737623361616235383231303834636536393836373466323265653339326239 -39333135396464346232386362376335333234383231343533393264346335376133313434393633 -34623039373134633032613838616532303130353833323330646165396631313864643831333763 -35633831626465316533346431616362366331633937393037666538633936343964643735363733 -32393939303931393362656332386234323634376534653435343137353063333037633033616639 -62643037376335316662613064396339393639386336616362623963373265383231613063356335 -63633432623061366237636134663834336165306234643537373064633561613561363963323031 -36653439396163333439366534653764616361313633343763323938636638656461616261343536 -37653866346161333533323436323063626534313539363666636232653639343366663663633162 -37343539386163666464643466326562363566316630633530363235663732316463346232366362 -31303037316265333862656137646631616163623330616565303933623731393530333230616533 -33356638303132313864343938313336626265656362613633623639653537366664383265636235 -36303633343439643339653833646536323161323638613139613734396433656365613862303932 -31353764303836653434653637653732363464373238376637393766323661333735363336336662 -65623030633433333339306335626334396162626161373637646337313535666431386432396162 -63333761363462643335633136316139626665633838613533366633656461623064393963346366 -31656637653436653631306230323630333531613465663134393339363930623263373031363839 -35393931313234393761303364386161623065326661323566653535343735346165643961303662 -63323436663930663031366131333361316339356463333634383937633535616663313333313037 -63363363643764633735323331373931396534646131323166626537646638333230346362656262 -64303962653265316530663531613235323165306565613866373966633038353237323933316336 -34313566383234326233313435336665633239666531656536326230373536666565636336653964 -37666533613165666262656637376662656339636538643639356435356332323964643731353462 -30653463326462373734313535643436336432623965616637643537666432356264666338333364 -39353134633761633461346665353161656436303764343661343530333532343739623830393762 -64653831366263366664323534646163663761633335396363653739386431303265643865373830 -35663930396333303430333831666433616366383366333032383232366431616338356138393533 -64613863383631333231373833616230343930393531646535303333376232633162333136646431 -34333639663032656562646561653061313132383832386266613431343630386530636362636330 -62626537383961666664656363343830376662306137336433663162323233616466386438653361 -34343138396237633138326633383335393330653231336238666365353530393232323332333837 -63666462613637313738653136333131623436653033396562373462373665653266366237653932 -35353939316438626233323561336230333330336632323062363436666239613137316563613062 -62393138373036636461393364663934633439623837366536386132663933323139396435393932 -66333662343965323338623531633562666631343963653233333637646565343634373337643636 -64303566643537383633663736393037353635303335393436643634393337396434393430656131 -61643830323039633363326565636139663534326434666161666231623434373963396161326566 -65626636353537343465343062336239313537376462316161343039373032373237313462663032 -35666335626337343334356530616236303639383035333362323035353862303863656331643630 -64663864653066623938323162356631626537663463616464663134613836633033626364313732 -66393065393365343339376637313362343130396464626363346330383164356538363833373430 -62363030663639343934613233366232363637323165363434373262356336663037306163663636 -64346534663833643932373733316331323032373738313637646464363563653932363735316639 -39386433336435316331373330343065396433323637613566613835356436333631323833313866 -33373332323632393239633363356232346634303233373663646536666635663233396335353733 -33336364646233626564663938643230373532613133343439636363323033373461386463643763 -36656234386563666662373737633537656262396234326232396636306139393532623531313431 -65663338663639663462323932653661623436306431323464663430613363303463343236326165 -30626635366639373331663239623763373966323934623365316563383039383737383262353130 -31616463353364353036313863316165656231613664373237373933626331316638373566383139 -66626634313234306137376166613939383664653536386532643636336466623132646639363564 -38373963653337393539326162383535373535376334386439316330386136366337393139336264 -64656337383235383963346335306632653837383166653737343661323237383065656137393232 -66626261396261383466336361343730616336626435363536613730313635323761353335633936 -63626164316462666461343038636633366136353864323135346335383034306464373437333062 -35626464333537313639356662316366613833343365663132303266333538346562326562323434 -65323230653334336239616163643036653537643264656239303666343430333765316631643337 -65356333346534623466343439333932656332306564613065656537386232636662373665306465 -6563 +32393162393333376133353565666661663262306566653830666366303432343431613035666166 +3638303965376639353538363166386236363033633165660a616638323233306330663432383233 +37316365353864356130653431323361303137383935356535326434326465336463633166376330 +3731336431306635640a633233353865303466393561396431366133643332653738663164643739 +36323936383031623134353531343930343964393438656238353836306538386633373464306563 +39323735646131373736616638336638343564376433386562326364353135373639316661653233 +66656161366438323966323435343130323265306262663035346563643965613230353563636561 +62653732633263363833343535626563663263356665383864356363393865396133346438663363 +30653332653935376264633165626463623033386633336435353636303238393332376163366532 +32663136383164636262376461303937326561383537383563366663313264316234633366303435 +62313965623731363262303837636239323466356362636236396232636666386365393064376538 +34333136656139653866346563376134316662343536656531363038343061373834313638393936 +64663165323635346133333066393662323163303462326561653735343331373138343565363932 +66396230356565346239346533616261633765316263653738393839343433336631363736643133 +34636533326362663365363933633664386266633539643733316163623933396262393235323962 +31623865343734643435343134373237346466356362316434376133306633653866343437656632 +34656261343861346463323932626135326666393565626537666365636537303131366434336231 +65363737376533616638356565373531643933653061303661373431306433333439346365356337 +61333635616137363137366331643965633261393538333631613830646436303432373062313434 +38373638613364313864663331386635633731663930373063633430303465366532656338356364 +65343631336338346438653239656134633239613037663261336134386266363562646134343630 +38346634613362326630613132623262333232373366343333376130333239323861353936666561 +31363638323461313236623238626366353636356339363633626162363539333139373366616630 +39653231623538663766666339666166333433646432663664336339666233363535346161636461 +37313066643964343664663336346332343463323937656635636239623265316431306330393137 +31336261343731616362303963376331316233303030613035653863383539376131333865333066 +65346639643233346661646563386364363334316661613531643938356439383766373132313030 +36363331383738333332643935613766393965333462383638653239316333383636613363303530 +32666536356332373737396131663531653938396161626266633062646664613963653630623032 +37373862636531643033303335383361333435353964376361376337613662366536326561623233 +38616637613136323538356666613531656635633166383739653239393266316661326634386563 +31326562313062343730643531663463383437626637313439333336616530363031636661366635 +38653637616235613136353138666665313031323735373364663164616636313461616236646239 +36613563316133316466313933303032316330353338636231383739613530666337663363303439 +38333061353966356538376566646138343831623131366466346338666461363066643162616638 +33393231383734386235653233616232346666323962636339313037643636346561363266353464 +38333563306663373664373833636265653135343465333134393732383936353665313165323339 +38323061323035643166356634373736376231663632346263336230303564323766333632353131 +64646164646166336435353064386166313236306165343930656332633365333538383133393037 +61623665396632333462393337653436316634656530303432316238353461643231393238396161 +33313736383463356538393836306132383738393338353036303164363938646532396635313034 +36303862313531353165653034313135393065306662326130353732353461333533383439663866 +30343539353539643036323363303638646338366164303239363766316562633936653330393466 +36393133303162613531643832303239303038323034383133313531633765643234666366643365 +38653831376334656466363564626636303931633434633337343430616230353938663539366434 +64633737623066653632313739373766333463363838366562333835326464653635653635663064 +37376466643566643737313265306237323663363761653538636462613636623666396531333063 +66663037306365383831316363633666666162356461623363613163373632383736353233393363 +39343962623835343335636262306663633266326363313735393766623236343133653032373265 +34636665633939613837666665383735346135373634643266393531386666666137626231396638 +38333166353863333938383163353431636630633037623139356437626464626330346138393832 +34326339373363313763343032343431643165616235376165323264383437666266376261616135 +35373730643837363331626138666565353033316132613631623931353264393837613436343334 +34393137376661343731653934386132326435343931383937366164356336313630663665643263 +64316265643139316433313663363130373766653337636237636638623231386161386238353664 +66663832383137333463396230316431353338353431323133643238363335653734323562306639 +31363162643630616336343362346132313134663236653664653461616566653433323763656333 +64333865626533323834323130343434626134303136643136326366316134366363306433303666 +32663064326235366330303030373163633836306435323630316563376161643331636161303365 +64313433326239386433656139313030323235333538666632663432313439393837666133313566 +62633666333964353031366234636134333737613939663463333632386361316239636339366565 +64393930313665343366623738313561306234316666316532373839653465336339393838326636 +61343638666263643434363862306364663162613637323931623661303932346262363037363963 +34303739646236643830336139656230346436396339373936393130323364633436363039393132 +32643365396134383666353935643530346130613563373731633133396365373364393466396339 +33366663663963376565313630663937333630376661356365356362643639656538336162363831 +62653764306333333137383435626639613035396232303434666163343430313933303839343631 +36373861363333613932653334326238633431663537313863333830323737373463383662323239 +65316161623933663134306430636435653661386333396262613737306537356266616338643637 +37636662373762393538333261623666356532633436383835393763386438393433336339613266 +34613832353363303132326664393064363961623130653439383031363163316132656665333337 +64613736353130376635646432343663653033376539616262383361343866386261333262653333 +33333136363366303564643633306634366361646261396230663231373231323163343139646262 +37333436393332316138643732303862646631643864613434373038393535333036366136363066 +32616331373761613766393933646536363137623238633865336466633639366562643236316163 +65383837303164323036353137393231396234313630623837663537346335343430343064373364 +36333365653936656539303634663066303362383737363130316663656162393733646338393432 +39643733666335376266396465623438656131373737613939656238323837653865316331353739 +33613334313037616533666230376536396230383334653338626663663261646234356237633665 +38383731663939353735646637646637623161353330653134363765376663356164646230353632 +63393634313633663365626634663137396265346265613163636538323031383064613165366435 +31396262336362346130303133373234336263653832633936316664666430393935393064356464 +38323663336434663963393637636234646636656636386263333438626337356230383138613439 +30343838633463663631326462646233656233313735383136343633633832393434366436656531 +34336132356464353232636664666432656261313537373635323430393934656633643961633465 +34353034663032613735343135323361343234663363306432393439636331346337656135316639 +61333235366465343466366139356536616334376362643464323230663037383837303064353833 +30653038383832393237316239633535366665363631363231373232353261663362336361616664 +33656131643263373465353730646561656464363535333361666631663433323638376364633831 +31306666343430366330623338616461626135363862656630336566306432626564616234346433 +64353639353162306464313433323439356130653065316134626435363938656331623831623333 +31656130316565663066363839343466303265396561636665623739356562383366303835636561 +39356433623230366234393332613364393530653731333133666165666538623662623666623061 +38356364333939666139626463633133386336633330336165386237616537353862373666373130 +32333135396634313066386637316330326633326465653463653839363361393833386264346530 +30363962323338333932623130643162386633386532393031626630386364366362306564356537 +38333461643836613036643239393636623832363966636131323131373332336434343062613764 +66313236613439616338373263336535363134343562356231613166623432393536343435346565 +37383966333665633666356462663538343930366163653964626266363938613436373261353339 +31366636626338333432663730633035353935623437353231636331633064303430303765623236 +36386434633437393066333336393534356438343562326665326563353732376566626630636165 +61356139366232333737376136663266666139336664623034323536663261366636396631393162 +64663834366465383562386632343537303263353565373534336430343266356630333665363439 +32663334336565633533623533336365313035643234363930323431393533623933626637393866 +31653432336635363938663039613630623239656566623839653062643034393032316235373363 +37643863373636666639386337666332633432366539306463346138323865306465643834343138 +31363264343038346161636563653962623130373836366666366135353430666266386361346564 +64343630623038626437623532353066336665636332316238323364633162653230393862663933 +65643466313139326566613862363833393534343536656132653730623333643732313438653261 +37633031303633666531663636616336313638306261663532366236386536616465666162623665 +36316231666166613264303938616262396234343464383433653331633466303530623334666566 +32623663653132616334646634326435363832346366366139336661366336353931326232383332 +35666566663763646164393962396535333630393366373737653562373937303965306265323562 +32623730663861323363663138343138613634313839323232313130353138643134616265636437 +62613062616661383262643632303839356535373232383265363565663830363736383730636430 +35666332333961303165653662336261343837366430626530306236656332393465363934366162 +32383439363934396336633430626366653963383134326636613333373536393234353938383631 +64616632393434666135656339656266613064313237346634353831336664666463373864633231 +38663039363632373733646534636262613132633739306531663439313836313761656139353930 +32306232626361616339383931646338313532366161356634373965656139356137616238623463 +37323336323131653335636565393133313162616130343039346536353333343635383234646338 +33363838646265386438353266623732306633643362333866613537313064356631393135653637 +63306432643164333365343461353166646462373131383935653533613433306539633865373363 +30666239373563646132663734616637383965343539643330356531376661633933396233616462 +66383739376236386362623531303531343735653438303031343530363330386434666137616239 +37626333323766626130623039633462616330383934393830623430393466656164373437346166 +64633234633661663736626335323061313431313638363439653736326563393362373736613036 +64636536363865626361313465326365316564353761353834623235343939376231393831333362 +36303532356266306434656538336663383163306631663836313532343335363966663137373033 +32303036393739653530653162653038373663613536366162353863326663636431623033306232 +63353437653936383933303733663235383835623364653637376566353134666239633836613365 +36313362653237373934633831636161323161643762623761643433373466383132613034663538 +31373561336232326463373761623537663833366534353937336565316133623566633138373230 +63643838653330626565363966373735656334313834363034633066393934653331366161396136 +34646536343230306534663661646132306537663132393265663636356638636332363730383961 +30396633333962353835636134343539346332376566646335353533303331363434623635306463 +33636535343938396533623961343465353131623338626233656632333733313732363633646130 +66613230306233383435326531363632313239306566373261373031663136393334626133303761 +66363263306139343837626433656237376631646533653963333834633138353234396631333239 +31323161373435626233643763643965656465633034373061356263346635376633343836333330 +38343139663938363036336161613364323161633163643061336136316663313666336161303733 +39646139663131313537303735653665333265323531353839393432656361633761373433613064 +38653233346433353133306266623534386538393466323334643464306632663336373166633661 +35313831616663353831636663616339643631303163343637656132313236616563373861396135 +65633234396439373431663931343134393739633733393631663766653530373031306363623462 +36383063336666326232373638333430316535363630623630303433393737393431633832316232 +61343363366634663237316666636165353838343261613231373363626633323235306162653165 +64383934636634653134366438383138346233383366386463626663313033313537383031333631 +35633834633639323038666637313431326564346566376333353931336232313739313239373566 +33613862616634663630336532616530643165366534613230623133653931383064363632623538 +62616130653135366533613337343030633663313030646630373633636339386661663762393539 +30396337323932663032346530303034313636326630643966363566643965633362633631333839 +63346461646230663337303537366233613532313032386264303736366262343937393261303264 +64323666306532346530313438323338643830353832373633653334316432353666383134313465 +64313962343464396131636466613130643530383738663564323734323534386435353231306463 +37633734306134386539666165366531656234333863653535323730323537613537363834363536 +30633065376437323364373538653765646433373733356163346531646665616134626533323466 +39343030616264646233633838343233396636626461616561646664326534653762363531396463 +31366163363136316338356135653565316363666537376438666136646635373366633338356363 +66663336643366366436643265633034626239323437396165653263653434313833393032393436 +37643334633639633265393865363731346561366564383430303332343734623130383563376661 +38386135653163343163653930396261376539646238636139616236303130363561326364663962 +33623739363235333633653865623436636437616435376439326662613037313730346266623432 +33363361666138346365313234623631383264356532643433353535386562366237373734623935 +32326666323363353132383230303633383264616138656363636261393734613133386462633436 +30303830336365333664376565313734666132373066396137663431333435373738366464313863 +30393132313937353237636437343935613861313635643637383237643531333638376533316364 +62646631306231346437663234363832393162383738356235663561376333656333646365323433 +66386561373935316662663130626632323930623161383964613830303538383836396135383266 +39613536316132653666396161363164333864343963366634356632383836656662663366613461 +31343964386333653631323733633866333531366336303062353837383366613434333136633830 +65653065646633613864643266323037373231373936646465356663383838373536623461393765 +36386565393038626661623264333036623561663362633161363064383263633931373265366539 +30373137383364326466666532313266343361633436316439623564653065353534373561313265 +39383630383666633138616666633031656663333734636131363338616135363136393366333033 +33396664613130383161333862633766656637383063643230393661323233306364626230393937 +38616438313664383262353735636138316538653930626639653232393531333738313564303731 +32306238336166333934336437303636643135346238386165616539306432336639343130353935 +66383336663738336562366630653330646330363063643632363837666233656236323237383732 +36316536613866623532633233393931333339656365343064393862623662306634336466383764 +65303737373430316238336139663637383133646438633833346339643239623836343138393933 +33636337363130383064366463313439386164316133666238346237313761333437663730373061 +65363733353863396163346534306233633764326462666234373732626232623361626438376336 +65653437353832616537653338356234666238663533373831633065356562373832666238363464 +31616139386533616132313534373534656335666230656264636531306163376164333265333634 +39366539326633613966366464633535363432303238343533653131396335306639623035393434 +65366134316636663238663862366165306132323339393334306435306465373636306337623664 +36666139613361306234366237356634366563666438346537646564653464376330303165316464 +65333030656639313263316637383233373630666536636335313337656266656164326465623030 +3264653937336633613533613866386139613665366337373262 diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml index 8d15ec0..f7fa9d5 100644 --- a/ansible/infra_secrets.yml +++ b/ansible/infra_secrets.yml @@ -1,178 +1,191 @@ $ANSIBLE_VAULT;1.1;AES256 -35383939623164356565323465316566346532343163646633323634396364376462313461366439 -3637616136336262303234323464343434643234383132300a333665333836653636393166376330 -36666466393762346333363930386334333863616265363661383639333561393061313938373366 -6562316232653761340a313631633931353736383636326339656432333065663032323033326334 -39646438633330323662643834383438656134376363653938356639356338633635663831323938 -62653533613861383536346230613161393834326637623636356265633237323331323565323339 -61306565663831613635656661373535396564376561653035616362623333386338666432346634 -33373033333737626534366164306436363932653231346134616163633634633131646165316330 -65393031613461623762383066626566373733666639333766643131396433306637653063356361 -35343231356534626661623331653365643230646566323666633031323734663933643864336264 -61616333353139626462623933343234386238393437623833363461323834306233653061363535 -64326339356435626265363235313965346165353434613962396465346265346539343764663733 -31343234356331333536663239336334653864393132666264383263333365663464663931626562 -36343430656262313962633965326539633834306263363036343562663132393537343863313961 -35356161386234326361626165343234393837643364616339666131343364386132333763623737 -39373035626332613566396532343535646235636661663263353338633763666363666363633465 -61353566646261376534323662363062623962643062386164666136623966363535356132656562 -65653465373366636533336539333035346537346261633933626334356631393461373361623537 -34326463373766393965653639313063383764313736626366653063663466646161313365306532 -63353566636433346237393834386537396232383961613932666537373565653338363435383938 -32333438623332626163336631663664396430643566376363366262656432666261383039326431 -63633036393634646237393665353666666239653533393532646464643934373535666137373764 -62646535613563363939313666636466623762636439323237623236383165666630363938393666 -34336439316665326265623335383434323837326661343163643833663330663834376432333134 -32333737643364373035626430393963373934316663626666333837666161616138633032393663 -37323432636435383462383061346332373862376336323363303031666336646161633539643033 -37633962356130613138343235633564393330373463613861313638303633326639626436613763 -61383236643439616232613331366535356130373365643064366533613330383661623833303231 -64353532646337643630373361303564386166306639623137643534376631393731393831383261 -36373934623337646234353165316563303336373437626464356463383733303338643936376164 -63303638313766333832323764666431313937633930623865353331616536653930653336343266 -30333931336165393634633231376266373933646661306138386631663539336139333265663631 -36333562396239623236336333343935663866363063353036663337646431356537373434666331 -39356635306263323365333263386331656636356436373862616434356462376134363034663532 -64333633663662623931343363393233306331303066303664323034373333656132386437366162 -37633830626362326637623139396631663834623436646364363439356361666661396236353739 -39333766396132616333323566393936363137333235643739333639646234313266303062323536 -38633964373238646534626566366438633364636461343861303437356562313966626235326537 -63316438373461313936343634373835306539333436393337656535323662663936656638353838 -38316234663432656539636239363861396139376264303336356630396163353234656562633362 -39613032653163313362643733393232316530613533343535376339636139393936616263346561 -30356134313831316364643439373135393836303330303737636137323035386432343963393165 -66366333306262353763383838376662633766636532643064373962366665313761646336303033 -32643633616162316537376435386536316431343136346264643163373033336463393130383562 -39623635316333383036336538626636363130353733336531396132623466363933333230616364 -30633364393334643866393332346239646235373766346562653061633732326537343634363065 -66646662336661386432306231313362636535316466656435366639383030376365313364666564 -65666535663532383662383833633261343934383535313163643133646663396135613761336337 -64653433366164373461643038353830626139306566643438393933396339616361623062346536 -31353039633437383862343865343133336665633436366137663033643834303835336633616361 -30363863626463333437626165333338366332653838663464633964363961373338336462376332 -63303738636539396635613938366336303462326564663937656231346639326438326561333430 -38646538353362303062376138636362333166373931663036646438663937326438316338653861 -62356365633966623638626138653064663835656431613262393532623731656132343864663966 -31373936333031646131663436323036366637326134336132373565316330663066366237336439 -62656562346639336339363363643933633566646363356565366336633863636230316634663936 -38626662653131623064373639383837366538633339313864346631633032356362373035303665 -61376332313532353665373131613335653732343839366365346330363638366265656139373838 -38633737653565366165306562353333353637326530636535616566323061386362333266383636 -61636365386232306338663962633235613435613466643032323363646232306431646164633639 -36383139313363613630616662613734636465646130363933376438633839363163343866323032 -35356531376663386331666238656537613164633634633934616462383331326633333235626337 -33393330303532626132393239633735623963363562656163383564306461353330376532396334 -30666361353862356636636632383234323565376235343862343564383830326533373861303437 -63613731643964613530343234323661636133643838316564356338366566303261336330616432 -32613535633630303132313933393833636433373539346261613661626135356363303930636138 -39323837383863306264633838636530393632363936633938626465656334336333333961666331 -66653463343333326563356337373732643333333065353436393736613038346636643535383466 -31656535633765386339633938666639396438326638303961303365323032333564636562343361 -65303662343633343364366265646432376230356362373465356539623632653663373631633633 -38666138363837356331343061373237303337656432326633646565353337393238363362396238 -66633966383237653064343333396661653932386536383536366632326234303161353166633836 -37313438393263366436613066386235343034366433343432346266343137346336646238613462 -33336339363037343533616166616565333330333531643364393437393637386330626437346135 -64396334656664336535373764666535373732353639326433646561616338323633623138373832 -30616363303834643532623362306362326230323737653563393838653234373861373933316538 -64376430376161316666396336373662393237383531353361393435386236636631613336386566 -31643637623333623364323734323961643863623532613432386662363361316134626533666339 -34656130633462616337323938626439626235306533336264343138356439633065313434616132 -63646664343962373365663738356233663135653965326534353665316531393964636132393932 -36633638636236326135323335646535396266666433396462623963656664336539386662373233 -64363566363863376466316338316562393461623262393130303762646334336463623632393238 -65656138396536373132363633313636643864353266356266626266643737393366326538353237 -31333961376236653263613930613432666531366131623834633632646332316133666530646139 -65616466646564646437333730383836656566643465303931326665343764396137643338373631 -62333266326265356562656430666538376134356239666635393430343234363231323938393331 -31613463613936656138623535303731333535323130303835366465383665353632636138653338 -35663132386634313538326130343435313131303934393434393736353531336261306265303264 -39376537636566333563393032336534333635346432663235623530666264323930646361313463 -39343437313932323064666632386137353834313930343636356561346535613730363865613433 -63316133663935373238373566326363343130376666613331666361326262353266653535343866 -64383931303037386165663439643232656139653832303239383235383637336664313864663935 -63366631303536336530366438333762393839656435366161353838303732386466356238333632 -34646463646535376139663136613562316166356138343939633935303532326536313033643639 -35653765326234333832366165363962376263383465313633313664336636616430313138313365 -31376535613130666562336533396636646161633932633030363437616337643761343439616338 -62333332376465306264663137613265363432343538643430333764326562613065663062613665 -63643030383163343231353964353264626434383762346639366335346465336365366131646131 -38633962316232663530376133383034303465313331393561393636386663623861373161376366 -37306533636463356232326430346538346136653264343839356662643435633337313232616561 -33633331346166303038656135303532643531383466346334383665613662333431336362333334 -66383336626463653136353861613832366338623631353930353432323338663834316436373230 -37393135303037663035303633643439653238663038626135623436303435303361393064353034 -35663230656434356434386462363834313137353865313831343261623466633861303336366334 -33626164313334306362393461313633376464343037646534666339386462666634363563386661 -37373138643432373834356335623864313561376131633161613035386166623836613339333563 -31613934363331633537663261653230306539616238313837393762323932363365326238353131 -31666564303666373163306237353136636430313639313931303339346335653539643265613230 -62643836353432316165336665373964353762376432373263323666393333393564323932333638 -33643763613739323134666139653862396436393836383232303564636261393161656163653830 -66653237313262616464656437343465623961633562376638626432663530356434396463663465 -39386261343933366666356664313339373231326566633165306632353535363436313864303936 -38613636326666303061616337666436383263363931323339646432373965363565333763343631 -37343566323732303830353537393565306434363130323335633837663762333431376539633765 -61363939626234386564306539636637333137626133363139353235366238326633623062303833 -64303832643830646430333061646333633039663634336263396563393437326135613932386634 -35393639373631323435373864306433373638613836396664633537323636653531616534316465 -63626661313733333431333665663165646234343438356134366131353234366466373133623639 -34656436306433363136666633616464373937633435616534376536393233313236356634323161 -63633632303562346566646166636664376431396162663765386532613832326339666130666536 -36663865633963613536623637303061626336343835646639633031366333393132663033396533 -38383862646631333262366237323234396465356531323363653464643361326362366231653739 -62643738303734643033323362656230616466353161663136623864313136386366376634653664 -31363930373530346333396536623037303162313934333634373033326432373566373032643964 -30663962353664353062303935356262363333366437383338373961616237663361326631353565 -37313936373033323131303161343139653565663466323830663536346665666232643163613562 -30643061653363663864383663326563306261663061326366383731366463356238373539383935 -65626464346366613638623730383631393066393233353130396636393438373565383266636439 -66633837626461623335316438393432303839633234323065396533343466373962633566643032 -35353939323038333434316232326138326266343662636130363565343463636364613263383438 -35653736333339353831383733393161356533343138336161383435333261613262306135393565 -66333134326265376433376538373531353966396138656265666266353737343936366132623633 -66663339616263623139666236373563363666356236383538316363326635396332373161373932 -31313430313065353762646538383361316563326262363833316434343965383032633139663634 -38383430336432373031636635366539663337666330663361303964653331376430303263613461 -39303832306463356465366237373338356636663637353835663336656331366638656433623462 -34623761366365366634353635333632373162643330633435303466326437626138663834323531 -61663433656537366662386134393761396162326265393465363131393333623731393534656338 -35356132323062336235326630353464393434633633343932613432623536343365626165626432 -31633732316364616463656430623763323039623835343164326536316334353666616237623535 -32303036653936363933353932323666636662393231306138306239653065623338336662613633 -34376631386235653161323331313661656366386635343264353535623439383734393363666239 -64303466393539353862626436643366306564336638346432373661336537353033613036393630 -35383962663363366665393938666266326464383936653539363838623338366466626439663438 -35333737356234363731396162353730643665666630366133633839303836313064306330373233 -62613363316238653534313739633033336134386264666362393061653434356637643736633863 -37333766646666626531323738633161366537306539303961373037656461323265353032666338 -30383834386333656334376361643338333662353461303932373132356133643934656466373864 -31306466396231363132636463333633383939376462386230363661656164666331323333363435 -33383262343238383265653232653538643964346236643031386164616364393732336637636239 -39663333646137313933313562626265663365633633653465396234396563656539646630346634 -63643461666365643961623936623364373334663162366537303163393665303633303562336361 -38323431346334623835396165653232663361383836663633386361616630633437623332383732 -33656139323339613164653433623237623939633230663735613063623131623632393535353066 -31633234613563616566396262343334313262366434306662613965316135323965643732356130 -31643437396264303831653762393930383431323062303032333666653761623866653762333663 -32633862383763633663313166313365333963636233386430366165643635633938306533663164 -35326561393265393830376434393235613265653764636538663165326363636538343630323763 -32396531613238313736393462313736393535613662353365636165373831626466386535623433 -63366338616637386336613435316262393561393139336331323132613835663065623637333833 -62663262653536313334623736343535323766643235303865653864613862663664386664643934 -37363832653530626339313336363639633863656633623838653564666135646333306264313765 -39613331313432636330383066653431383333323265623233633434343066323764633765663832 -34653434323631313261653234353437353762616466633463383835313034336234326666303864 -35333034376131323436613735383433393266613966363432646663366636303564666532646163 -38333963306665306438333034313933303337363331376663633333643161636236343662383361 -33333933346665613061346634666530323830653732356231306365613436323263316462373734 -65366462646532366638366235646533383235346463613164393630306636346232626264323932 -37343631656361633235333338613933396565376636343932306363623635373237366639633236 -62653537313134613239353365613239666466616630653336653738393032646263336561623861 -31333434646532393963346238613662343737323831393136383736636631393065306664616333 -32643666393261623262326562356631353432353231383161613964383566643466323962313238 -33393531363135636362326434383962633463633039333664353865383231353634633535333738 -62636432646461376431396638333231643164306432333132356131316538303366343333396564 -35373732323633666263663064356262323432653462333834636433613231356637626265653866 -61346366353165363639633466336463313462653137613035373430313334336262626439393539 -3663 +35396230336432656135363562623433303661376462666131373963633839386238636162613065 +6563613965386631663235303662636261323635393031620a633032313564306337666631646130 +32623664343966653866643438613965303936616234313036303061336139633535653333636439 +3432666530356639640a353631666335653863326232323061663136646332316535656437336235 +31623433376561633931626537333162323331336338326332633630343530656364323536303336 +62383763643033383138333365336130356264626661663732313561323465366437323164393733 +38333532376239373861613566333466316234623564316434383434393064323330663438393139 +33323263323038376135623439376132663637653634613465653865623163366365633861336161 +31323337656463396464303963353236383638646438646531393934656137303063343238363331 +61656262313037306531343437306361646637373864356662386333346134626637346363383931 +39386565643335633863396662363264633432646336383235303466616262663533303965623831 +39333563353431636438646639376663666439343161396163656463366333373362313466626465 +36376665653561636135383531336236643461303261393534373331633934346438323035663637 +63343634376532386532633435616661656336313966363566323937653533613665363461306638 +65393366346662356264353433386534326537363436313836373735616663303131643063313064 +61396634623262616661333365313466636332653261666539653135656166383933373436333032 +64393634396531656536323436373061626531663436313630373865323838376639386430376331 +64306138653931383335366130386334386236363062363631333465303635303531386261383662 +65343236663334356631306439613263653162626536303163343463353431363766643636653166 +33613439643634323331323165386337316463383232646232616439306565353866646635353163 +30343965316234666164633863333632376561383638323639353361313263616561613335316135 +62326466623134313236316430303861356335346633333337373639343766393639353431653061 +39363266663732393434326539353032666161623265633464626237613364396239626163343466 +36303364326638346230326632333636643866663937373864326161663765363535333761373138 +36333034623532616261613336326134626334313063353165303035383166303733393939303231 +30643731353963303839623265336663663436643839666366656661313232623036643232646465 +33666539343565363464373530323033353330343139666239623663646166373733363736633335 +61316134646234653638373639356665396663376366366366383831643938633933656261316431 +65346334653864663863333239383131326665316234363536356234373266653064353562623934 +61656632343235376235323466316562336337326637386534393436643664396634326261643736 +65376264376430333461353336656539636133343933323966323431376638653161316535323862 +65383661373763336638333461303433373365633161623333326535343932366537633362663431 +38663630336139313831656634653736633162353038333438643533633962633166633330333939 +38636266316162303039303737666265633230373364316362393562383532393539376133633137 +31366562666337313866343662363732333932643866646164633964323435393139633766376165 +39616262383366376164383334353633313431393830663962356165633734306337666330386235 +61303664303231346133353130666132623533343938613962363334663466313833633430393337 +38393631326534316363333966366138313836303830393763376633356234353335353366323962 +66366362386338353830346232303032666265343138313836633539333763626132653065656136 +35653634616638346138613464393464636239633762346236613030663766363839373134323730 +34373936323838633035373764373461636533333033646439386665653539653066616135333664 +38643964613966623833653233343464333464633134353431396536656432353738333761363932 +35646436323565656463613433353563373936316130373565636638333237373038613435633765 +63393538613531323063383434643739383134643863313739343863396437333461343936653666 +30656161656638316534633264626338666431333035633432346666636235343438303936363738 +61633838303238346336316365623062356464393463386237366664656134626231633831313964 +39393663626337366264663462663532636365623834386337333538633833633337646138373331 +30633639313633643233386664643033386136306265393233313236343166373838393165316234 +36366339373730343933343661353462656165643361626335663661343666313762613534313934 +37333764363565316163656334613631633938646564323535336164396438626261346365306233 +39383338303063343839633161623631343165336466383838633935323466613831313137313732 +34303961306638363233346436363432363366666162326438623962653138643661336233303464 +38626464326331623735313335356637376633373065366235303834343661633866303965646236 +34373066323464396363663562363239343261643930306632356566656364373231323039316339 +63656630356566343131363935396630313632383137323263363936383862623534623532636563 +61616236346237373031343034646231663833613132356362303532633062356638383266303438 +38303937376166663333653766633165306631366333323831333534613438363063323839633263 +32363132626335656434303130386330626530346463616136393137333863393965346639366261 +35313166356437393163663866626437383633323031373931643439623866353162646566303739 +61373530353264663436643937323431393762623933383233383833663235656434356639353664 +31336230666464396533633735313632663064333436306235303334653064303237303737636132 +62663461383761303863663538626262313038343739656262306564653034353031383234666534 +39643765393634323937656263626463373865643561376235663131306431306564363463346266 +61313835623933336436636131316130346337623835366366343735613562616137643466616631 +65386234656235383639353432333663626538303564303065383064636638333466366663346535 +62373034353430393733656163303933393332373030646630646638373861393463663434303264 +31626439663239373732336330356131613862343836363232326663643436643264323531366530 +62346532633565373961666464626431386431623038313464396361643934646332373735383339 +31343333623739623437356431623439333134333932363734663734636239623930663234656562 +63633561646362643738306632343732636166303636383939626135356536356134656262356636 +36346338643433393462303162343737643330656230643339386536393738353161303831663338 +65346262346136363565333261656164356664323533626537363562313935653766623535313361 +64343637376530333833343935633338303738376664323363643131383938333636636365333134 +33306363396233613164363461616238616364633461326666343366343534663166646439316437 +30396236333236393466323639663930636333663261373563346364613366616336643335646139 +39613632663131323432663938373537663335643934616333656662346432353766326631643034 +35316436653765386235366464373565386463333137343039653732386235633162373834383730 +66323139363137333835613132383364333437643332613930313463343539393736333336363039 +37333136373933353932353637663238363336613133626237393436353762316463633437626130 +33373762643736643465353030653538653262363531643935363836396631653431663231373636 +33656637336535646538383364393165313461616565316330653362643037333635386433626566 +64346166393130636661313731343436383139663831323031346663613761363637353537373632 +37316135333930386632613039356465353761383433333465333033646433373235323932343333 +39643532306133383665643838303366613631386439303862613963656434616461316236383233 +39356665383463356435646630313932626263643037623038613662383536663831316537613038 +63336262346361326664333432386465376634383534373562643238323462633334626366343761 +65303264316332656663616161373335313863396363643462656661613766306632636138623234 +35623666313130633836366161623862333161376464326166616539633835613861343038363736 +62646661393837353266663962313836326162306665323230353764633965336634373934366230 +31633364373164653163626162353934333933303332623564653362396665303961393039343565 +31386632363536373561643838613465633462346539343266613562393236626166383332653335 +38633432333535633532363239653332386565626232653963653930623735633632356264653463 +38376238393961653834616134376462363862633462333265663932653338656362363461613533 +33666137333639616237306263366264623236343732393134643266353331323634343462613438 +64616464316535363766666331373638313566643132316131336337393131346334393632633366 +32633964616461366435643234326234366434363131643437346131396139323331626237396365 +61666437643335396432373932653066663862623035646365343561616630663835663161303537 +63653134303134633861393036653464613265646135353061343663623964333436663834396336 +30313063336131626534396438393731616233333365393965383863323461303561356332363763 +66313863373164663734663462393431653864623764656536366266643231313765363430393134 +33306565646532306639346635323461313736663533303266643433306634326565383232643939 +65623738643739353363653266353534323332363037373834366135643537313739353263636638 +32653234353538386461323632613536366434613262653538333661313333363539393664363363 +64656461356661613161383562653465353866623961653232386265613864343562356639336339 +30653131623130633332343931343537326636356134303336306139333736303763303533323735 +37636131323530336264356264383263343038316264646634356338353761383033646631396364 +62303330386362623436343161353731623435356331373666613335653761653366356433373564 +34623062613439376432303765333763346662393866326337613061616138396430326133376139 +34663337393938343563363266386536616334623964633132303361393137623936623434393562 +37383038666166643537316439323333373431383661643863653034323330366565306630643161 +34643931373966306539623361323261633061363837643538373830343730646534366435636335 +35303862396563333835636431636433393933363533643636343265613638343933373438323563 +64613937666330343038343437616131303630323961306135386264363361383163393836353963 +35303064333065373533366535366361393436393535343965333865646137366166376530346436 +62386432316264663532333465623265313734333337323038613066663533356638666662323334 +35353165633564666130353962386439373863353535373236653338646633356531373230373262 +38666637663661396538333031653966616365353961653530613465623362393932643064323035 +39636534663464363064323862336633663038393834663333353235353562363934373739656664 +65653362346632613862393133316362366665646130326232313138323237646664383565366466 +34666231346433656235363362313434313438396361383231656135306330303035393530393530 +61643331626464646266326239336565613562353462323834393731326339383461313164633930 +31653633373633333539373962383866663739653135336463623837643165323831366562326536 +63613233383533636338646466363962313933393035356339316162616436306138383262633430 +37356366636662363231366532393763633564366461383162613539633734626563303864316233 +61653339393465626631373664306165353964373966623733386532636233346535326132366261 +34663036343665316234303439313064643939663833386633366264653163353234356237343837 +39613134623138616264613732383065623434303336393866633836306432663562393730343833 +35633139343130343236653931346266613665616230633032626665663031313262306530396364 +30663863326632633261346363633838653339393239376462386631303865636364616238646663 +37663033363433333531353332383866383065363337643265636264353562663635393766303265 +39323832663261653831616561333639373838316138356333646439356535333531626530633238 +33623039666433633436363365666431623039353065386564653139336138346136333238663866 +63356237373130343163303936643562373037613361663834386562353337663061313934643430 +32646137636139646263336439663730393737613061663933386165343034386161343333623834 +33323163626266363136376536616362636631353264363736663366636266653238656362633934 +30373664643933613637346363343964346463363036313961633961303832323834376337666630 +38363862613136396663626436303237396435333464303938303562333034613532373134353233 +38653436336461386139303864393338656661653137646137336332616435656638393231323838 +62663430663330653963313939663362656430303761656633343837306265366332393030363930 +30346439393837363333636136353235353463333563656332633236663033633835363531633261 +61643632396666663934346139373235346261626339386431333064333632333034333031666439 +66336561303536656462343363653136616562306362333662643938616437663938393933633463 +38636339356565663532663235326266386332353764656639363165336634653639336463616362 +66393131656631646133376461343438386162306430613561323630613865626235303438346233 +33373839323732666437363333346336663963323066353561396338383633626138323337383032 +35663132623364666537343036303136336333633864323834633265343936366631363865333962 +63306433333035363565303866333234363038663934366361656539333963373932306535313638 +61626661336330393265353064616263396533316361616534343566613833363061346230333036 +62626130383630306361646531666436613461333865343131626465323361373163336539343163 +39313036373236646162333563393561646663326631613333646238626463396364343061383233 +31373365396535326131616237323636623364633263386564613763623636623138386430376364 +39326637386237616332363531393534636262393336366335306665636466346134633365386665 +33646266386635636361383937616630376530643237396366666231313930373939313566383938 +64646436613039633835363966343838656461313834643438356166356566663861663833666137 +32666437626637323766303962303164393565626362656430353439383839313166303966366433 +64383730336566633336303966376366343332633366373662353337663965616333363033393765 +36306536316135303164306164313239333938646630633864336237346161303963376531373464 +34373465366231336365323030376235306662313433616433633962363564326262363631666136 +65613538656534366136303030303339383136323137633934633662646537353235646534313332 +31366365613533653339646131653239343663346239613934393666373536386635656534376165 +64613638643338613738353263633664636164353265616536626132346663636133393137616433 +33663635343063373465303936313731333765396137393235643566656332666634636462386461 +37323065373733333433316463663635653630656131313235386663306134373364393765356239 +38393964346361356135626331653161633536333463666463653936656531313463666632393061 +36383166633935343063663763663536326466646663383763333161613165393665383030396262 +34666631316262616131346233613035343236613838333036663661386464346136383737323036 +66366632313130646430616563653731346362343339363235616335393964633335326633376135 +34386666353864303161346263663265393731326332303231323934383838343934376339373139 +34663666363335663165636237303330306338393136653339633733663364623461303134303839 +62366337396363613737623138666565613530643861653830616332656435373233656266353461 +66616636393864643338616164316131623464613466396562386238663364623630303338643837 +64633433666139333864313862663832306230663339643933663530383939396230393032323861 +30376464336566303833306664613633333335326562396462326437343437323032363065363237 +61646238363436383134343434353534326537666431333430343065633035353437616366383039 +36616362353536396436623835656462353231653838336666636335666333653130633330353232 +64643733393034343139373733623365333961343139353532313230613134346365303061616663 +36636465353234316361376261303834333532623336646335306636353965613361366430643436 +33306135633538653034643361323235353566636661363436326133326238633964646263376166 +32333862623239333661393930346638663461633762653332386631616666366333333863336239 +63356265663437343037396464396365323336313038336636636562343262396466353333313765 +33396636396562323338383561616366323832616566323464653035366432653833633831633234 +30636237616332383434623338616639633732643134383637343162353234316639333733613033 +31306362623163373631633965623631643537393662626538663966373765383530343064313165 +66303337656537343539323565663636333139383033393462326331643234616138323236663630 +62653163393733613838633739393439343464653738616161663330633236626234313034346462 +31633539303139306361613839366234643535383961346130363433633033373434633565613063 +35323839386463383438663131303665386130653332366463363330363530653533313661363361 +63313539653239303861356638303236396138326131626564666337626332383462633732373465 +36353966383036663733333336643633666633346233363732633837656334316331313032323064 +3739636230343038353664353163363939363465326433613561 diff --git a/ansible/roles/datum_gateway/README.md b/ansible/roles/datum_gateway/README.md new file mode 100644 index 0000000..cdf35e5 --- /dev/null +++ b/ansible/roles/datum_gateway/README.md @@ -0,0 +1,73 @@ +# `datum_gateway` + +Builds and runs [DATUM Gateway](https://github.com/OCEAN-xyz/datum_gateway), the +solo/pooled mining gateway, on `knots-box`. The calling playbook adds two more +plays on the edge host: the dashboard via `caddy_site`, and the public Stratum +port via `socket_proxy`. + +Converted from `deploy_datum_gateway_playbook.yml` (802 lines) under Plan 6. The +playbook is now 68 lines and keeps all three plays. + +## ⚠ This is half of a system + +The Bitcoin Knots node on the same host feeds this gateway through +`blocknotify=killall -USR1 datum_gateway` in `bitcoin.conf` — see +`roles/bitcoin_knots/README.md`, where that line was found to be missing from the +template entirely. **Changing either config means thinking about both.** + +Interrupting Stratum costs mining shares. Check before any run that restarts it: + +```bash +ss -tn state established '( sport = :23334 )' +``` + +## Two pieces of drift where the node was right + +The repo and the node had diverged on values that matter, and the deployment +would have applied the repo's: + +| | node (correct) | repo said | +|---|---|---| +| `datum_mining_address` | `bc1qvrj3g84…` | `bc1qdse9dsg…` | +| `pool_pass_workers` / `_full_users` | `false` | `true` | + +The address is the one that would have hurt: **it is where block rewards are +paid**, and unlike fulcrum and bitcoin-knots the `Restart datum-gateway` handler +here was *never* gated, so the change would have applied immediately rather than +sitting inert. Both corrected in the vault and defaults, with notes. + +Verify semantics rather than text when touching `config.json` — render it and +compare parsed JSON, because the live file is single-line and the template is +pretty-printed, so a textual diff is all noise: + +```python +json.load(open('live.json')) == json.load(open('rendered.json')) +``` + +## `config.json` holds real secrets — diff is suppressed + +The file carries `bitcoind.rpcpassword` and `api.admin_password`. `--diff` +prints rendered content, so the task sets `diff: false` by default; pass +`-e datum_reveal_config=true` to opt in. + +Note `pool_pass_workers` / `pool_pass_full_users` are **booleans**, not +passwords, despite the names — they control DATUM's pool-password passthrough. +`mining.pool_address` is a Bitcoin address and public by nature. + +## Expect `changed` on the compile every run + +`Configure cmake build` and `Compile datum_gateway` are bare `command:` tasks +with no `changed_when`, so they always report changed and always re-run. The +build is reproducible — `Install datum_gateway binary` sees identical content and +does not replace it, so the installed binary keeps its original timestamp — but +the compile itself is wasted work on every run. That is the idempotent floor, not +drift. + +## Monitoring: one variable, no product knowledge + +The check tests the gateway API and records the answer in its exit code, which +systemd keeps: `systemctl is-failed datum-gateway-healthcheck.service`. Set +`healthcheck_push_url` to report anywhere accepting an HTTP ping. + +Unlike the other services here, only the health-check *timer* handler was gated +by `uptime_kuma_enabled`; the main deployment restart worked throughout. diff --git a/ansible/services/datum-gateway/datum_gateway_vars.yml b/ansible/roles/datum_gateway/defaults/main.yml similarity index 57% rename from ansible/services/datum-gateway/datum_gateway_vars.yml rename to ansible/roles/datum_gateway/defaults/main.yml index 48bb5b7..d3647ec 100644 --- a/ansible/services/datum-gateway/datum_gateway_vars.yml +++ b/ansible/roles/datum_gateway/defaults/main.yml @@ -14,8 +14,8 @@ datum_gateway_log_dir: /var/log/datum-gateway datum_gateway_bin_path: /usr/local/bin/datum_gateway # Ports -datum_gateway_stratum_port: 23334 # Miners connect here via Stratum v1 -datum_gateway_api_port: 7152 # Web dashboard / API +datum_gateway_stratum_port: "{{ service_settings.datum_gateway.stratum_port }}" +datum_gateway_api_port: "{{ service_settings.datum_gateway.api_port }}" # Stratum settings datum_vardiff_min: 524288 # Minimum share difficulty (must be power of 2; OCEAN floor overrides if higher) @@ -37,7 +37,18 @@ datum_bitcoin_rpc_url: "http://127.0.0.1:8332" datum_coinbase_tag_primary: "DATUM" datum_coinbase_tag_secondary: "BY ORDER OF BIP110" -datum_pool_pass_workers: true -datum_pool_pass_full_users: true +# Both false on the node; the vars file said true. Corrected 2026-09-13 to +# match reality, on the same basis as datum_mining_address: the running node +# is authoritative. These control DATUM's pool-password passthrough. +datum_pool_pass_workers: false +datum_pool_pass_full_users: false datum_pooled_mining_only: true + +# --- Health check ----------------------------------------------------------- +# Checks the DATUM Gateway API and records the answer in its exit code, which +# systemd keeps: `systemctl is-failed datum-gateway-healthcheck.service`. +# +# WHERE TO REPORT HEALTH — the one place to plug in monitoring. Empty means +# check, exit honestly, report nowhere. +healthcheck_push_url: "" diff --git a/ansible/roles/datum_gateway/handlers/main.yml b/ansible/roles/datum_gateway/handlers/main.yml new file mode 100644 index 0000000..b7cfe25 --- /dev/null +++ b/ansible/roles/datum_gateway/handlers/main.yml @@ -0,0 +1,15 @@ +--- +- name: Restart datum-gateway + systemd: + name: datum-gateway + state: restarted + daemon_reload: yes + +# Ungated. This one carried `when: uptime_kuma_enabled | default(false)` while +# the main Restart datum-gateway handler above did not — so on this service the +# deployment restart worked and only the health-check timer restart was dead. +- name: Restart datum-gateway health check timer + systemd: + name: datum-gateway-healthcheck.timer + state: restarted + daemon_reload: yes diff --git a/ansible/roles/datum_gateway/tasks/configure.yml b/ansible/roles/datum_gateway/tasks/configure.yml new file mode 100644 index 0000000..463dec0 --- /dev/null +++ b/ansible/roles/datum_gateway/tasks/configure.yml @@ -0,0 +1,25 @@ +--- +# Ownership copied verbatim from the playbook this replaces and verified +# mechanically against `git show HEAD:`. +- name: Write DATUM Gateway config.json + ansible.builtin.template: + src: config.json.j2 + dest: "{{ datum_gateway_config_dir }}/config.json" + owner: "{{ datum_gateway_user }}" + group: "{{ datum_gateway_group }}" + mode: '0640' + # config.json carries the bitcoind RPC password, the API admin password and + # the pool passwords. `--diff` prints rendered content, so running with --diff + # put all of them on the terminal and into any log capturing it. Suppressed by + # default; pass -e datum_reveal_config=true when you genuinely need the diff. + diff: "{{ datum_reveal_config | default(false) | bool }}" + notify: Restart datum-gateway + +- name: Create datum-gateway systemd service + ansible.builtin.template: + src: datum-gateway.service.j2 + dest: /etc/systemd/system/datum-gateway.service + owner: root + group: root + mode: '0644' + notify: Restart datum-gateway diff --git a/ansible/roles/datum_gateway/tasks/healthcheck.yml b/ansible/roles/datum_gateway/tasks/healthcheck.yml new file mode 100644 index 0000000..211b85b --- /dev/null +++ b/ansible/roles/datum_gateway/tasks/healthcheck.yml @@ -0,0 +1,50 @@ +--- +# Everything here answers "is DATUM Gateway healthy" and records the answer. The +# Uptime Kuma specifics that used to follow — an embedded Python script creating +# monitors over the API, a /tmp credentials file, a push-URL file read back and +# parsed, and a systemd Environment= rewrite — are gone. Where it reports is now +# one variable, healthcheck_push_url. +- name: Create DATUM Gateway health check script + ansible.builtin.template: + src: healthcheck.sh.j2 + dest: /usr/local/bin/datum-gateway-healthcheck-push.sh + owner: root + group: root + mode: '0755' + validate: "bash -n %s" + +- name: Create datum-gateway health check systemd service + ansible.builtin.template: + src: healthcheck.service.j2 + dest: /etc/systemd/system/datum-gateway-healthcheck.service + owner: root + group: root + mode: '0644' + notify: Restart datum-gateway health check timer + +- name: Create datum-gateway health check systemd timer + ansible.builtin.template: + src: healthcheck.timer.j2 + dest: /etc/systemd/system/datum-gateway-healthcheck.timer + owner: root + group: root + mode: '0644' + notify: Restart datum-gateway health check timer + +- name: Reload systemd daemon after health check units + systemd: + daemon_reload: yes + +# Ungated: enabling a timer is deployment, not monitoring. +- name: Enable and restart the datum-gateway health check timer + systemd: + name: datum-gateway-healthcheck.timer + enabled: yes + state: restarted + daemon_reload: yes + +# Arms the timer and smoke-tests the check. See roles/bitcoin_knots/README.md for +# why restarting the timer alone is not enough with OnBootSec + OnUnitActiveSec. +- name: Run the DATUM Gateway health check once to arm the timer + command: systemctl start datum-gateway-healthcheck.service + changed_when: false diff --git a/ansible/roles/datum_gateway/tasks/install.yml b/ansible/roles/datum_gateway/tasks/install.yml new file mode 100644 index 0000000..61dcadb --- /dev/null +++ b/ansible/roles/datum_gateway/tasks/install.yml @@ -0,0 +1,74 @@ +--- +- name: Install DATUM Gateway build dependencies + apt: + name: + - cmake + - build-essential + - git + - libjansson-dev + - libmicrohttpd-dev + - libsodium-dev + - libcurl4-openssl-dev + # Runtime-only (netcat for health check) + - netcat-openbsd + state: present + update_cache: yes + +# =========================================== +# System User and Directories +# =========================================== +- name: Create datum system user + user: + name: "{{ datum_gateway_user }}" + system: yes + shell: /usr/sbin/nologin + home: "{{ datum_gateway_dir }}" + create_home: no + comment: "DATUM Gateway" + +- name: Create DATUM Gateway directories + file: + path: "{{ item.path }}" + state: directory + owner: "{{ item.owner }}" + group: "{{ datum_gateway_group }}" + mode: "{{ item.mode }}" + loop: + - { path: "{{ datum_gateway_dir }}", owner: root, mode: "0755" } + - { path: "{{ datum_gateway_source_dir }}", owner: root, mode: "0755" } + - { path: "{{ datum_gateway_config_dir }}", owner: "{{ datum_gateway_user }}", mode: "0750" } + - { path: "{{ datum_gateway_log_dir }}", owner: "{{ datum_gateway_user }}", mode: "0750" } + +# =========================================== +# Build from Source +# =========================================== +- name: Clone DATUM Gateway repository at {{ datum_gateway_version }} + git: + repo: https://github.com/OCEAN-xyz/datum_gateway.git + dest: "{{ datum_gateway_source_dir }}" + version: "{{ datum_gateway_version }}" + force: yes + register: git_clone + +- name: Configure cmake build + command: cmake . -DCMAKE_BUILD_TYPE=Release + args: + chdir: "{{ datum_gateway_source_dir }}" + +- name: Compile datum_gateway + command: make -j{{ datum_gateway_build_jobs }} + args: + chdir: "{{ datum_gateway_source_dir }}" + +- name: Install datum_gateway binary + copy: + src: "{{ datum_gateway_source_dir }}/datum_gateway" + dest: "{{ datum_gateway_bin_path }}" + remote_src: yes + owner: root + group: root + mode: "0755" + notify: Restart datum-gateway + +# =========================================== +# Configuration diff --git a/ansible/roles/datum_gateway/tasks/main.yml b/ansible/roles/datum_gateway/tasks/main.yml new file mode 100644 index 0000000..57d9a45 --- /dev/null +++ b/ansible/roles/datum_gateway/tasks/main.yml @@ -0,0 +1,6 @@ +--- +# import_tasks, not include_tasks: static imports stay visible to --list-tasks. +- ansible.builtin.import_tasks: install.yml +- ansible.builtin.import_tasks: configure.yml +- ansible.builtin.import_tasks: service.yml +- ansible.builtin.import_tasks: healthcheck.yml diff --git a/ansible/roles/datum_gateway/tasks/service.yml b/ansible/roles/datum_gateway/tasks/service.yml new file mode 100644 index 0000000..c8581fd --- /dev/null +++ b/ansible/roles/datum_gateway/tasks/service.yml @@ -0,0 +1,16 @@ +--- +- name: Reload systemd daemon + systemd: + daemon_reload: yes + +- name: Enable and start datum-gateway + systemd: + name: datum-gateway + enabled: yes + state: started + +# =========================================== +# Health Check Script + Systemd Timer +# =========================================== +# ═════════════════════════════════════════════════════════════════════════ +# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. diff --git a/ansible/roles/datum_gateway/templates/config.json.j2 b/ansible/roles/datum_gateway/templates/config.json.j2 new file mode 100644 index 0000000..e17a386 --- /dev/null +++ b/ansible/roles/datum_gateway/templates/config.json.j2 @@ -0,0 +1,35 @@ +{ + "bitcoind": { + "rpcuser": "{{ bitcoin_rpc_user }}", + "rpcpassword": "{{ bitcoin_rpc_password }}", + "rpcurl": "{{ datum_bitcoin_rpc_url }}", + "notify_fallback": true + }, + "stratum": { + "listen_port": {{ datum_gateway_stratum_port }}, + "vardiff_min": {{ datum_vardiff_min }} + }, + "mining": { + "pool_address": "{{ datum_mining_address }}", + "coinbase_tag_primary": "{{ datum_coinbase_tag_primary }}", + "coinbase_tag_secondary": "{{ datum_coinbase_tag_secondary }}" + }, + "api": { + "admin_password": "{{ datum_gateway_admin_password }}", + "listen_port": {{ datum_gateway_api_port }}, + "modify_conf": false + }, + "logger": { + "log_to_console": true, + "log_to_file": true, + "log_file": "{{ datum_gateway_log_dir }}/datum_gateway.log", + "log_rotate_daily": true, + "log_level_console": 2, + "log_level_file": 1 + }, + "datum": { + "pool_pass_workers": {{ datum_pool_pass_workers | lower }}, + "pool_pass_full_users": {{ datum_pool_pass_full_users | lower }}, + "pooled_mining_only": {{ datum_pooled_mining_only | lower }} + } +} diff --git a/ansible/roles/datum_gateway/templates/datum-gateway.service.j2 b/ansible/roles/datum_gateway/templates/datum-gateway.service.j2 new file mode 100644 index 0000000..18f7a3c --- /dev/null +++ b/ansible/roles/datum_gateway/templates/datum-gateway.service.j2 @@ -0,0 +1,22 @@ +[Unit] +Description=DATUM Gateway - Bitcoin Mining Gateway +Documentation=https://github.com/OCEAN-xyz/datum_gateway +After=network.target bitcoind.service +Wants=bitcoind.service + +[Service] +User={{ datum_gateway_user }} +Group={{ datum_gateway_group }} +Type=simple +ExecStart={{ datum_gateway_bin_path }} --config {{ datum_gateway_config_dir }}/config.json +Restart=on-failure +RestartSec=10 +StandardOutput=journal +StandardError=journal + +# Prevent config from being read by other users +ReadWritePaths={{ datum_gateway_log_dir }} +ReadOnlyPaths={{ datum_gateway_config_dir }} + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/datum_gateway/templates/healthcheck.service.j2 b/ansible/roles/datum_gateway/templates/healthcheck.service.j2 new file mode 100644 index 0000000..5dcd3f5 --- /dev/null +++ b/ansible/roles/datum_gateway/templates/healthcheck.service.j2 @@ -0,0 +1,14 @@ +[Unit] +Description=DATUM Gateway Health Check +After=network.target datum-gateway.service + +[Service] +Type=oneshot +User=root +ExecStart=/usr/local/bin/datum-gateway-healthcheck-push.sh +Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/datum_gateway/templates/healthcheck.sh.j2 b/ansible/roles/datum_gateway/templates/healthcheck.sh.j2 new file mode 100644 index 0000000..ba8d069 --- /dev/null +++ b/ansible/roles/datum_gateway/templates/healthcheck.sh.j2 @@ -0,0 +1,32 @@ +#!/bin/bash +# DATUM Gateway health check — managed by Ansible (roles/datum_gateway) +# +# The exit code is the answer and systemd keeps it: +# systemctl is-failed datum-gateway-healthcheck.service +# Reporting anywhere else is optional and generic. +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +STRATUM_PORT={{ datum_gateway_stratum_port }} + +check_datum() { + # Service must be active and stratum port must be listening + systemctl is-active --quiet datum-gateway && \ + nc -z 127.0.0.1 "${STRATUM_PORT}" +} + +report() { + local status=$1 + local msg=$2 + # No push URL is normal, not an error: the exit code below is still a + # complete answer for anything reading unit state. + [ -n "$PUSH_URL" ] || return 0 + curl -s --max-time 10 --retry 2 -o /dev/null \ + "${PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true +} + +if check_datum; then + report "up" "OK" + exit 0 +else + report "down" "DATUM Gateway not responding" + exit 1 +fi diff --git a/ansible/roles/datum_gateway/templates/healthcheck.timer.j2 b/ansible/roles/datum_gateway/templates/healthcheck.timer.j2 new file mode 100644 index 0000000..c14d1d2 --- /dev/null +++ b/ansible/roles/datum_gateway/templates/healthcheck.timer.j2 @@ -0,0 +1,10 @@ +[Unit] +Description=DATUM Gateway Health Check Timer + +[Timer] +OnBootSec=2min +OnUnitActiveSec=1min +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index 0c53e9d..371f3fa 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -1,667 +1,59 @@ --- -# DATUM Gateway Deployment Playbook +# DATUM Gateway: solo/pooled mining gateway, built from source on knots-box. # -# Deploys DATUM Gateway (https://github.com/OCEAN-xyz/datum_gateway) on the -# Bitcoin Knots host so it has direct localhost RPC access to bitcoind. -# -# What this does: -# 1. Installs build deps and compiles datum_gateway from source -# 2. Creates a dedicated system user and config/log directories -# 3. Writes /etc/datum-gateway/config.json from vars/secrets -# 4. Patches bitcoin.conf with the required blockmaxsize/blocknotify lines -# 5. Creates and enables a systemd service -# 6. Creates a push-monitor health check script + systemd timer -# 7. Registers a push monitor in Uptime Kuma -# -# Separate play: adds a Caddy reverse proxy on vipy for the dashboard. -# -# Stratum port (23334) is bound on knots_box_local. Expose it to miners via -# a firewall rule, Tailscale, or a socket proxy on vipy — not handled here. -# -# Required secrets in infra_secrets.yml: -# datum_mining_address - Bitcoin address for block rewards -# datum_gateway_admin_password - Password for the /api admin endpoint -# bitcoin_rpc_user - Shared with the bitcoin-knots deployment -# bitcoin_rpc_password - Shared with the bitcoin-knots deployment - -- name: Deploy DATUM Gateway on knots_box_local +# This is HALF OF A SYSTEM. The Bitcoin Knots node on the same host feeds it via +# `blocknotify=killall -USR1 datum_gateway` in bitcoin.conf (see +# roles/bitcoin_knots/README.md). Changing either config means thinking about +# both. Interrupting Stratum costs mining shares, so check for connected miners +# before restarting: +# ss -tn state established '( sport = :23334 )' +- name: Deploy DATUM Gateway on the bitcoin host hosts: bitcoin become: yes vars_files: - ../../infra_vars.yml - ../../services_config.yml - ../../infra_secrets.yml - - ./datum_gateway_vars.yml vars: - datum_gateway_subdomain: "{{ subdomains.datum_gateway }}" - datum_gateway_domain: "{{ datum_gateway_subdomain }}.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" + # Preserves the push URL this check reports to. The role knows nothing about + # Uptime Kuma — this is just "a URL that accepts a ping". + healthcheck_push_url: "{{ healthcheck_push_urls.datum_gateway | default('') }}" + roles: + - datum_gateway - tasks: - # =========================================== - # Build Dependencies - # =========================================== - - name: Install DATUM Gateway build dependencies - apt: - name: - - cmake - - build-essential - - git - - libjansson-dev - - libmicrohttpd-dev - - libsodium-dev - - libcurl4-openssl-dev - # Runtime-only (netcat for health check) - - netcat-openbsd - state: present - update_cache: yes - - # =========================================== - # System User and Directories - # =========================================== - - name: Create datum system user - user: - name: "{{ datum_gateway_user }}" - system: yes - shell: /usr/sbin/nologin - home: "{{ datum_gateway_dir }}" - create_home: no - comment: "DATUM Gateway" - - - name: Create DATUM Gateway directories - file: - path: "{{ item.path }}" - state: directory - owner: "{{ item.owner }}" - group: "{{ datum_gateway_group }}" - mode: "{{ item.mode }}" - loop: - - { path: "{{ datum_gateway_dir }}", owner: root, mode: "0755" } - - { path: "{{ datum_gateway_source_dir }}", owner: root, mode: "0755" } - - { path: "{{ datum_gateway_config_dir }}", owner: "{{ datum_gateway_user }}", mode: "0750" } - - { path: "{{ datum_gateway_log_dir }}", owner: "{{ datum_gateway_user }}", mode: "0750" } - - # =========================================== - # Build from Source - # =========================================== - - name: Clone DATUM Gateway repository at {{ datum_gateway_version }} - git: - repo: https://github.com/OCEAN-xyz/datum_gateway.git - dest: "{{ datum_gateway_source_dir }}" - version: "{{ datum_gateway_version }}" - force: yes - register: git_clone - - - name: Configure cmake build - command: cmake . -DCMAKE_BUILD_TYPE=Release - args: - chdir: "{{ datum_gateway_source_dir }}" - - - name: Compile datum_gateway - command: make -j{{ datum_gateway_build_jobs }} - args: - chdir: "{{ datum_gateway_source_dir }}" - - - name: Install datum_gateway binary - copy: - src: "{{ datum_gateway_source_dir }}/datum_gateway" - dest: "{{ datum_gateway_bin_path }}" - remote_src: yes - owner: root - group: root - mode: "0755" - notify: Restart datum-gateway - - # =========================================== - # Configuration - # =========================================== - - name: Write DATUM Gateway config.json - copy: - dest: "{{ datum_gateway_config_dir }}/config.json" - content: | - { - "bitcoind": { - "rpcuser": "{{ bitcoin_rpc_user }}", - "rpcpassword": "{{ bitcoin_rpc_password }}", - "rpcurl": "{{ datum_bitcoin_rpc_url }}", - "notify_fallback": true - }, - "stratum": { - "listen_port": {{ datum_gateway_stratum_port }}, - "vardiff_min": {{ datum_vardiff_min }} - }, - "mining": { - "pool_address": "{{ datum_mining_address }}", - "coinbase_tag_primary": "{{ datum_coinbase_tag_primary }}", - "coinbase_tag_secondary": "{{ datum_coinbase_tag_secondary }}" - }, - "api": { - "admin_password": "{{ datum_gateway_admin_password }}", - "listen_port": {{ datum_gateway_api_port }}, - "modify_conf": false - }, - "logger": { - "log_to_console": true, - "log_to_file": true, - "log_file": "{{ datum_gateway_log_dir }}/datum_gateway.log", - "log_rotate_daily": true, - "log_level_console": 2, - "log_level_file": 1 - }, - "datum": { - "pool_pass_workers": {{ datum_pool_pass_workers | lower }}, - "pool_pass_full_users": {{ datum_pool_pass_full_users | lower }}, - "pooled_mining_only": {{ datum_pooled_mining_only | lower }} - } - } - owner: "{{ datum_gateway_user }}" - group: "{{ datum_gateway_group }}" - mode: "0640" - notify: Restart datum-gateway - - # =========================================== - # Systemd Service - # =========================================== - - name: Create datum-gateway systemd service - copy: - dest: /etc/systemd/system/datum-gateway.service - content: | - [Unit] - Description=DATUM Gateway - Bitcoin Mining Gateway - Documentation=https://github.com/OCEAN-xyz/datum_gateway - After=network.target bitcoind.service - Wants=bitcoind.service - - [Service] - User={{ datum_gateway_user }} - Group={{ datum_gateway_group }} - Type=simple - ExecStart={{ datum_gateway_bin_path }} --config {{ datum_gateway_config_dir }}/config.json - Restart=on-failure - RestartSec=10 - StandardOutput=journal - StandardError=journal - - # Prevent config from being read by other users - ReadWritePaths={{ datum_gateway_log_dir }} - ReadOnlyPaths={{ datum_gateway_config_dir }} - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: "0644" - notify: Restart datum-gateway - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start datum-gateway - systemd: - name: datum-gateway - enabled: yes - state: started - - # =========================================== - # Health Check Script + Systemd Timer - # =========================================== - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create DATUM Gateway health check script - when: uptime_kuma_enabled | default(false) - copy: - dest: /usr/local/bin/datum-gateway-healthcheck-push.sh - content: | - #!/bin/bash - UPTIME_KUMA_PUSH_URL="${UPTIME_KUMA_PUSH_URL}" - STRATUM_PORT={{ datum_gateway_stratum_port }} - - check_datum() { - # Service must be active and stratum port must be listening - systemctl is-active --quiet datum-gateway && \ - nc -z 127.0.0.1 "${STRATUM_PORT}" - } - - push_to_uptime_kuma() { - local status=$1 - local msg=$2 - if [ -z "$UPTIME_KUMA_PUSH_URL" ]; then - echo "ERROR: UPTIME_KUMA_PUSH_URL not set" - return 1 - fi - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${UPTIME_KUMA_PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true - } - - if check_datum; then - push_to_uptime_kuma "up" "OK" - exit 0 - else - push_to_uptime_kuma "down" "DATUM Gateway not responding" - exit 1 - fi - owner: root - group: root - mode: "0755" - - - name: Create datum-gateway health check systemd service - copy: - dest: /etc/systemd/system/datum-gateway-healthcheck.service - content: | - [Unit] - Description=DATUM Gateway Health Check - After=network.target datum-gateway.service - - [Service] - Type=oneshot - User=root - ExecStart=/usr/local/bin/datum-gateway-healthcheck-push.sh - Environment=UPTIME_KUMA_PUSH_URL= - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: "0644" - - - name: Create datum-gateway health check systemd timer - copy: - dest: /etc/systemd/system/datum-gateway-healthcheck.timer - content: | - [Unit] - Description=DATUM Gateway Health Check Timer - - [Timer] - OnBootSec=2min - OnUnitActiveSec=1min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: "0644" - - - name: Reload systemd daemon after health check units - systemd: - daemon_reload: yes - - - name: Enable and start datum-gateway health check timer - when: uptime_kuma_enabled | default(false) - systemd: - name: datum-gateway-healthcheck.timer - enabled: yes - state: started - - # =========================================== - # Uptime Kuma Push Monitor Setup - # =========================================== - - name: Create Uptime Kuma push monitor setup script for DATUM Gateway - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_datum_gateway_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import time - import traceback - import yaml - - try: - import socketio.exceptions - except ImportError: - pass - - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_datum_gateway_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - try: - api.add_monitor(type='group', name='services') - except Exception: - time.sleep(2) - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - # Check if monitor already exists - existing = next((m for m in monitors if m.get('name') == monitor_name), None) - - push_url = None - - if existing: - print(f"Monitor '{monitor_name}' already exists (ID: {existing['id']})") - push_token = existing.get('pushToken') or existing.get('push_token') - if push_token: - push_url = f"{url}/api/push/{push_token}" - else: - print(f"Creating push monitor '{monitor_name}'...") - try: - api.add_monitor( - type=MonitorType.PUSH, - name=monitor_name, - parent=group['id'], - interval=90, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - except Exception as e: - # socketio timeout: add_monitor may have succeeded server-side - print(f"add_monitor raised (possibly timeout): {e}", file=sys.stderr) - time.sleep(2) - - monitors = api.get_monitors() - new_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - if new_monitor: - push_token = new_monitor.get('pushToken') or new_monitor.get('push_token') - if push_token: - push_url = f"{url}/api/push/{push_token}" - - api.disconnect() - - if push_url: - print(f"PUSH_URL={push_url}") - with open('/tmp/datum_gateway_push_url.txt', 'w') as f: - f.write(push_url) - - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: "0755" - - - name: Create temporary config for push monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_datum_gateway_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_name: "DATUM Gateway" - mode: "0644" - - - name: Run Uptime Kuma push monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_datum_gateway_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Display monitor setup output - debug: - msg: "{{ monitor_setup.stdout_lines }}" - when: monitor_setup.stdout is defined - - - name: Read push URL from file - when: uptime_kuma_enabled | default(false) - slurp: - src: /tmp/datum_gateway_push_url.txt - delegate_to: localhost - become: no - register: push_url_file - ignore_errors: yes - - - name: Parse push URL - set_fact: - datum_push_url: "{{ push_url_file.content | b64decode | trim }}" - when: push_url_file.content is defined - - - name: Update health check service with push URL - lineinfile: - path: /etc/systemd/system/datum-gateway-healthcheck.service - regexp: "^Environment=UPTIME_KUMA_PUSH_URL=" - line: "Environment=UPTIME_KUMA_PUSH_URL={{ datum_push_url }}" - when: datum_push_url is defined - notify: Restart datum-gateway health check timer - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_datum_gateway_monitor.py - - /tmp/ansible_datum_gateway_config.yml - - /tmp/datum_gateway_push_url.txt - - handlers: - - name: Restart datum-gateway - systemd: - name: datum-gateway - state: restarted - daemon_reload: yes - - - name: Restart datum-gateway health check timer - when: uptime_kuma_enabled | default(false) - systemd: - name: datum-gateway-healthcheck.timer - state: restarted - daemon_reload: yes - - -# =========================================== -# Caddy Reverse Proxy for DATUM Dashboard (on vipy) -# =========================================== -- name: Configure Caddy reverse proxy for DATUM Gateway dashboard on the edge host +- name: Configure Caddy reverse proxy for the DATUM Gateway dashboard on the edge host hosts: edge become: yes vars_files: - ../../infra_vars.yml - ../../services_config.yml - ../../infra_secrets.yml - - ./datum_gateway_vars.yml - vars: - datum_gateway_subdomain: "{{ subdomains.datum_gateway }}" - datum_gateway_domain: "{{ datum_gateway_subdomain }}.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - tasks: - name: Publish the DATUM Gateway dashboard through Caddy ansible.builtin.include_role: name: caddy_site vars: caddy_site_name: datum-gateway - caddy_site_domain: "{{ datum_gateway_domain }}" - caddy_site_upstream: "knots-box:{{ datum_gateway_api_port }}" + caddy_site_domain: "{{ subdomains.datum_gateway }}.{{ root_domain }}" + caddy_site_upstream: "{{ service_settings.datum_gateway.tailscale_hostname }}:{{ service_settings.datum_gateway.api_port }}" caddy_site_resolvers: "100.100.100.100" caddy_site_basic_auth: - user: "{{ datum_dashboard_username }}" hash: "{{ datum_dashboard_password_hash }}" - # The role validates the site fragment on its own. This re-validates the - # whole assembled Caddyfile, which is the only thing that catches a - # conflict between this site and another. Kept from the hand-rolled - # version; the other nine services never had it. + # The role validates its own site fragment; this re-validates the whole + # assembled Caddyfile, which is the only thing that catches a conflict + # between two sites. - name: Validate the assembled Caddyfile ansible.builtin.command: caddy validate --config /etc/caddy/Caddyfile --adapter caddyfile changed_when: false - - name: Display DATUM Gateway dashboard URL - when: uptime_kuma_enabled | default(false) - debug: - msg: "DATUM Gateway dashboard: https://{{ datum_gateway_domain }}" - - # =========================================== - # Uptime Kuma HTTP Monitor for Public Dashboard - # =========================================== - - name: Create Uptime Kuma HTTP monitor setup script for DATUM dashboard - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_datum_http_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import time - import traceback - import yaml - - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_datum_http_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - monitors = api.get_monitors() - - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - try: - api.add_monitor(type='group', name='services') - except Exception: - time.sleep(2) - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - existing = next((m for m in monitors if m.get('name') == monitor_name), None) - - if existing: - print(f"Monitor '{monitor_name}' already exists (ID: {existing['id']})") - else: - print(f"Creating HTTP monitor '{monitor_name}'...") - try: - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - except Exception as e: - print(f"add_monitor raised (possibly timeout): {e}", file=sys.stderr) - time.sleep(2) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: "0755" - - - name: Create temporary config for HTTP monitor - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_datum_http_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ datum_gateway_domain }}" - monitor_name: "DATUM Gateway Dashboard" - mode: "0644" - - - name: Run Uptime Kuma HTTP monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_datum_http_monitor.py - delegate_to: localhost - become: no - register: http_monitor_setup - changed_when: "'SUCCESS' in http_monitor_setup.stdout" - ignore_errors: yes - - - name: Display HTTP monitor setup output - debug: - msg: "{{ http_monitor_setup.stdout_lines }}" - when: http_monitor_setup.stdout is defined - - - name: Clean up HTTP monitor temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_datum_http_monitor.py - - /tmp/ansible_datum_http_config.yml - - -# =========================================== -# Stratum Port Forwarding on vipy via systemd-socket-proxyd -# Miners connect to vipy:23334; traffic is forwarded to knots-box:23334 -# over the Tailscale network, matching the Bitcoin P2P proxy pattern. -# =========================================== - name: Setup public Stratum port forwarding on the edge host hosts: edge become: yes vars_files: - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - - ./datum_gateway_vars.yml - vars: - datum_tailscale_hostname: "knots-box" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - tasks: - name: Expose the DATUM Stratum port through a socket proxy ansible.builtin.include_role: @@ -669,134 +61,8 @@ vars: socket_proxy_name: datum-stratum socket_proxy_description: "DATUM Stratum" - socket_proxy_listen_port: "{{ datum_gateway_stratum_port }}" - socket_proxy_upstream_host: "{{ datum_tailscale_hostname }}" - # Matches the UFW comment already on vipy; the derived default would - # have said "DATUM Stratum" and rewritten the rule. + socket_proxy_listen_port: "{{ service_settings.datum_gateway.stratum_port }}" + socket_proxy_upstream_host: "{{ service_settings.datum_gateway.tailscale_hostname }}" + # Matches the UFW comment already on the edge host; the derived default + # would say "DATUM Stratum" and rewrite the rule. socket_proxy_ufw_comment: "DATUM Gateway Stratum public access" - - - name: Display public Stratum endpoint - when: uptime_kuma_enabled | default(false) - debug: - msg: "DATUM Stratum public endpoint: {{ ansible_host }}:{{ datum_gateway_stratum_port }}" - - # =========================================== - # Uptime Kuma TCP Monitor for Public Stratum - # =========================================== - - name: Create Uptime Kuma TCP monitor setup script for Stratum - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_datum_stratum_tcp_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import time - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_datum_stratum_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_host = config['monitor_host'] - monitor_port = config['monitor_port'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - monitors = api.get_monitors() - - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - try: - api.add_monitor(type='group', name='services') - except Exception: - time.sleep(2) - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - existing = next((m for m in monitors if m.get('name') == monitor_name), None) - - if existing: - print(f"Monitor '{monitor_name}' already exists (ID: {existing['id']})") - else: - print(f"Creating TCP monitor '{monitor_name}'...") - try: - api.add_monitor( - type=MonitorType.PORT, - name=monitor_name, - hostname=monitor_host, - port=monitor_port, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - except Exception as e: - print(f"add_monitor raised (possibly timeout): {e}", file=sys.stderr) - time.sleep(2) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: "0755" - - - name: Create temporary config for Stratum TCP monitor - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_datum_stratum_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_host: "{{ ansible_host }}" - monitor_port: {{ datum_gateway_stratum_port }} - monitor_name: "DATUM Stratum (public)" - mode: "0644" - - - name: Run Uptime Kuma TCP monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_datum_stratum_tcp_monitor.py - delegate_to: localhost - become: no - register: tcp_monitor_setup - changed_when: "'SUCCESS' in tcp_monitor_setup.stdout" - ignore_errors: yes - - - name: Display TCP monitor setup output - debug: - msg: "{{ tcp_monitor_setup.stdout_lines }}" - when: tcp_monitor_setup.stdout is defined - - - name: Clean up Stratum TCP monitor temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_datum_stratum_tcp_monitor.py - - /tmp/ansible_datum_stratum_config.yml - diff --git a/ansible/services_config.yml b/ansible/services_config.yml index e342a79..a6dc9c6 100644 --- a/ansible/services_config.yml +++ b/ansible/services_config.yml @@ -52,6 +52,13 @@ service_settings: # edge host. A role default cannot reach that second play. p2p_port: 8333 tailscale_hostname: knots-box + datum_gateway: + # Needed on two hosts: the datum_gateway role deploys on knots-box, while the + # Caddy play (dashboard) and the socket-proxy play (Stratum) both run on the + # edge host. A role default cannot reach either of those plays. + api_port: 7152 + stratum_port: 23334 + tailscale_hostname: knots-box fulcrum: # Same shape as mempool: the fulcrum role deploys on fulcrum-box, and the # socket-proxy play publishes the SSL port from the edge host. A role default From 0f03c503c884542d74a991dd807a2e51cf34cf4d Mon Sep 17 00:00:00 2001 From: counterweight Date: Sun, 13 Sep 2026 19:07:50 +0200 Subject: [PATCH 54/67] nodito: de-Uptime-Kuma the ZFS and NUT playbooks, extract templates These two are host-specific by design - nodito is a pet, not cattle - so they stay playbooks rather than becoming roles. But they had rotted. De-Kuma, following the pattern of the six service roles: Both plays opened with an assert on uptime_kuma_username/password, which were removed from the vault, so both failed before doing anything. Dropped that, the two embedded Python monitor-creation scripts, and their /tmp cleanup. Kept every check, threshold and systemd timer - those are the durable part. Reporting is now generic: `healthcheck_push_url` goes into the unit as Environment=HEALTHCHECK_PUSH_URL and the script reads ${HEALTHCHECK_PUSH_URL:-}, treating empty as normal rather than an error. The exit code is the real answer; systemd keeps it. Both scripts now also report status=down on failure instead of only going silent. Live push URLs harvested into the vault so nothing observable changes for ZFS. Three live bugs found while check-diffing: 1. 32_zfs would have DE-REGISTERED the Proxmox storage. `pvesm remove` was gated on the storage existing and `pvesm add` on it NOT existing - mutually exclusive - so a real run removed the proxmox-tank-1 entry backing every VM and never put it back. It would also have dropped `mountpoint /var/lib/vz`, which the live entry has and `pvesm add` does not set. Registration is now add-only. 2. zfs_disk_1 named ata-...WX11TN0Z, a disk no longer in the machine. The live mirror is WX120LHQ + WX11TN2P; a leg was replaced and the repo never caught up. Inert behind `when: zfs_pool_exists.rc != 0`, but wrong on any disaster-recovery run. This is the seventh instance of an identifier written down once whose hardware later moved. 3. 34_nut has NEVER been applied to nodito - no /etc/nut file carries the "Managed by Ansible" marker; they were written by hand in January 2026. The vault held the literal CHANGE_ME_TO_SECURE_PASSWORD, so applying it would have overwritten a working upsd/upsmon auth pair with a placeholder and restarted NUT, leaving the hypervisor's UPS unable to trigger a clean shutdown on mains loss. The Kuma assert was the only thing stopping that, so removing it without a replacement would have armed the gun: there is now an explicit assert that refuses to run on the placeholder. The real password is in the vault and `Configure upsd users` check-diffs clean. Templates reconciled with the live files first, so applying 34_nut is close to a no-op: added `maxretry = 3` to ups.conf and OFFDURATION / RBWARNTIME / NOCOMMWARNTIME / FINALDELAY plus quoted POWERDOWNFLAG to upsmon.conf. The only substantive additions left are the NOTIFYMSG/NOTIFYFLAG syslog lines. /usr/local/bin/ups-heartbeat.sh on the box is an orphan - mode 0644, not executable, referenced by no unit and no cron entry - but its push token belongs to a monitor that still exists and answers, so that monitor has had no heartbeat since January. Harvested as healthcheck_push_urls.ups; applying this play is what will finally feed it. Finish the host_vars migration: infra/nodito/nodito_vars.yml was byte-identical to host_vars/nodito/main.yml. Deleted it, moved nodito_secrets.yml to host_vars/nodito/vault.yml, and stripped the dead vars_files entries from all three playbooks. Extract the 13 inline `content: |` blocks to infra/nodito/templates/, pulled via a YAML load rather than retyped. 681+582 lines become 313+344 plus templates. Ownership parity against HEAD checked mechanically: no owner/group/mode drift on any surviving task. Neither playbook has been applied. ZFS play 2 check-runs failed=0 with two changes: the script rewrite (the deployed one is a hand-edited DEBUG VERSION with `set -x`) and the Environment line. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 383 +++++++-------- ansible/host_vars/nodito/main.yml | 7 +- ansible/host_vars/nodito/vault.yml | 11 + .../nodito/32_zfs_pool_setup_playbook.yml | 435 ++---------------- .../33_proxmox_debian_cloud_template.yml | 1 - .../nodito/34_nut_ups_setup_playbook.yml | 365 +++------------ ansible/infra/nodito/nodito_secrets.yml | 12 - ansible/infra/nodito/nodito_vars.yml | 28 -- ansible/infra/nodito/templates/nut.conf.j2 | 2 + .../nodito/templates/ups-heartbeat.service.j2 | 14 + .../nodito/templates/ups-heartbeat.timer.j2 | 11 + ansible/infra/nodito/templates/ups.conf.j2 | 9 + .../nodito/templates/ups_heartbeat.sh.j2 | 68 +++ ansible/infra/nodito/templates/upsd.conf.j2 | 2 + ansible/infra/nodito/templates/upsd.users.j2 | 4 + ansible/infra/nodito/templates/upsmon.conf.j2 | 34 ++ .../templates/zfs-health-monitor.service.j2 | 14 + .../templates/zfs-health-monitor.timer.j2 | 11 + .../templates/zfs-monthly-scrub.service.j2 | 13 + .../templates/zfs-monthly-scrub.timer.j2 | 10 + .../nodito/templates/zfs_health_monitor.sh.j2 | 181 ++++++++ ansible/infra_secrets.yml | 383 +++++++-------- 22 files changed, 876 insertions(+), 1122 deletions(-) create mode 100644 ansible/host_vars/nodito/vault.yml delete mode 100644 ansible/infra/nodito/nodito_secrets.yml delete mode 100644 ansible/infra/nodito/nodito_vars.yml create mode 100644 ansible/infra/nodito/templates/nut.conf.j2 create mode 100644 ansible/infra/nodito/templates/ups-heartbeat.service.j2 create mode 100644 ansible/infra/nodito/templates/ups-heartbeat.timer.j2 create mode 100644 ansible/infra/nodito/templates/ups.conf.j2 create mode 100644 ansible/infra/nodito/templates/ups_heartbeat.sh.j2 create mode 100644 ansible/infra/nodito/templates/upsd.conf.j2 create mode 100644 ansible/infra/nodito/templates/upsd.users.j2 create mode 100644 ansible/infra/nodito/templates/upsmon.conf.j2 create mode 100644 ansible/infra/nodito/templates/zfs-health-monitor.service.j2 create mode 100644 ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 create mode 100644 ansible/infra/nodito/templates/zfs-monthly-scrub.service.j2 create mode 100644 ansible/infra/nodito/templates/zfs-monthly-scrub.timer.j2 create mode 100644 ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index a2b0fcf..a619094 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,191 +1,194 @@ $ANSIBLE_VAULT;1.1;AES256 -32393162393333376133353565666661663262306566653830666366303432343431613035666166 -3638303965376639353538363166386236363033633165660a616638323233306330663432383233 -37316365353864356130653431323361303137383935356535326434326465336463633166376330 -3731336431306635640a633233353865303466393561396431366133643332653738663164643739 -36323936383031623134353531343930343964393438656238353836306538386633373464306563 -39323735646131373736616638336638343564376433386562326364353135373639316661653233 -66656161366438323966323435343130323265306262663035346563643965613230353563636561 -62653732633263363833343535626563663263356665383864356363393865396133346438663363 -30653332653935376264633165626463623033386633336435353636303238393332376163366532 -32663136383164636262376461303937326561383537383563366663313264316234633366303435 -62313965623731363262303837636239323466356362636236396232636666386365393064376538 -34333136656139653866346563376134316662343536656531363038343061373834313638393936 -64663165323635346133333066393662323163303462326561653735343331373138343565363932 -66396230356565346239346533616261633765316263653738393839343433336631363736643133 -34636533326362663365363933633664386266633539643733316163623933396262393235323962 -31623865343734643435343134373237346466356362316434376133306633653866343437656632 -34656261343861346463323932626135326666393565626537666365636537303131366434336231 -65363737376533616638356565373531643933653061303661373431306433333439346365356337 -61333635616137363137366331643965633261393538333631613830646436303432373062313434 -38373638613364313864663331386635633731663930373063633430303465366532656338356364 -65343631336338346438653239656134633239613037663261336134386266363562646134343630 -38346634613362326630613132623262333232373366343333376130333239323861353936666561 -31363638323461313236623238626366353636356339363633626162363539333139373366616630 -39653231623538663766666339666166333433646432663664336339666233363535346161636461 -37313066643964343664663336346332343463323937656635636239623265316431306330393137 -31336261343731616362303963376331316233303030613035653863383539376131333865333066 -65346639643233346661646563386364363334316661613531643938356439383766373132313030 -36363331383738333332643935613766393965333462383638653239316333383636613363303530 -32666536356332373737396131663531653938396161626266633062646664613963653630623032 -37373862636531643033303335383361333435353964376361376337613662366536326561623233 -38616637613136323538356666613531656635633166383739653239393266316661326634386563 -31326562313062343730643531663463383437626637313439333336616530363031636661366635 -38653637616235613136353138666665313031323735373364663164616636313461616236646239 -36613563316133316466313933303032316330353338636231383739613530666337663363303439 -38333061353966356538376566646138343831623131366466346338666461363066643162616638 -33393231383734386235653233616232346666323962636339313037643636346561363266353464 -38333563306663373664373833636265653135343465333134393732383936353665313165323339 -38323061323035643166356634373736376231663632346263336230303564323766333632353131 -64646164646166336435353064386166313236306165343930656332633365333538383133393037 -61623665396632333462393337653436316634656530303432316238353461643231393238396161 -33313736383463356538393836306132383738393338353036303164363938646532396635313034 -36303862313531353165653034313135393065306662326130353732353461333533383439663866 -30343539353539643036323363303638646338366164303239363766316562633936653330393466 -36393133303162613531643832303239303038323034383133313531633765643234666366643365 -38653831376334656466363564626636303931633434633337343430616230353938663539366434 -64633737623066653632313739373766333463363838366562333835326464653635653635663064 -37376466643566643737313265306237323663363761653538636462613636623666396531333063 -66663037306365383831316363633666666162356461623363613163373632383736353233393363 -39343962623835343335636262306663633266326363313735393766623236343133653032373265 -34636665633939613837666665383735346135373634643266393531386666666137626231396638 -38333166353863333938383163353431636630633037623139356437626464626330346138393832 -34326339373363313763343032343431643165616235376165323264383437666266376261616135 -35373730643837363331626138666565353033316132613631623931353264393837613436343334 -34393137376661343731653934386132326435343931383937366164356336313630663665643263 -64316265643139316433313663363130373766653337636237636638623231386161386238353664 -66663832383137333463396230316431353338353431323133643238363335653734323562306639 -31363162643630616336343362346132313134663236653664653461616566653433323763656333 -64333865626533323834323130343434626134303136643136326366316134366363306433303666 -32663064326235366330303030373163633836306435323630316563376161643331636161303365 -64313433326239386433656139313030323235333538666632663432313439393837666133313566 -62633666333964353031366234636134333737613939663463333632386361316239636339366565 -64393930313665343366623738313561306234316666316532373839653465336339393838326636 -61343638666263643434363862306364663162613637323931623661303932346262363037363963 -34303739646236643830336139656230346436396339373936393130323364633436363039393132 -32643365396134383666353935643530346130613563373731633133396365373364393466396339 -33366663663963376565313630663937333630376661356365356362643639656538336162363831 -62653764306333333137383435626639613035396232303434666163343430313933303839343631 -36373861363333613932653334326238633431663537313863333830323737373463383662323239 -65316161623933663134306430636435653661386333396262613737306537356266616338643637 -37636662373762393538333261623666356532633436383835393763386438393433336339613266 -34613832353363303132326664393064363961623130653439383031363163316132656665333337 -64613736353130376635646432343663653033376539616262383361343866386261333262653333 -33333136363366303564643633306634366361646261396230663231373231323163343139646262 -37333436393332316138643732303862646631643864613434373038393535333036366136363066 -32616331373761613766393933646536363137623238633865336466633639366562643236316163 -65383837303164323036353137393231396234313630623837663537346335343430343064373364 -36333365653936656539303634663066303362383737363130316663656162393733646338393432 -39643733666335376266396465623438656131373737613939656238323837653865316331353739 -33613334313037616533666230376536396230383334653338626663663261646234356237633665 -38383731663939353735646637646637623161353330653134363765376663356164646230353632 -63393634313633663365626634663137396265346265613163636538323031383064613165366435 -31396262336362346130303133373234336263653832633936316664666430393935393064356464 -38323663336434663963393637636234646636656636386263333438626337356230383138613439 -30343838633463663631326462646233656233313735383136343633633832393434366436656531 -34336132356464353232636664666432656261313537373635323430393934656633643961633465 -34353034663032613735343135323361343234663363306432393439636331346337656135316639 -61333235366465343466366139356536616334376362643464323230663037383837303064353833 -30653038383832393237316239633535366665363631363231373232353261663362336361616664 -33656131643263373465353730646561656464363535333361666631663433323638376364633831 -31306666343430366330623338616461626135363862656630336566306432626564616234346433 -64353639353162306464313433323439356130653065316134626435363938656331623831623333 -31656130316565663066363839343466303265396561636665623739356562383366303835636561 -39356433623230366234393332613364393530653731333133666165666538623662623666623061 -38356364333939666139626463633133386336633330336165386237616537353862373666373130 -32333135396634313066386637316330326633326465653463653839363361393833386264346530 -30363962323338333932623130643162386633386532393031626630386364366362306564356537 -38333461643836613036643239393636623832363966636131323131373332336434343062613764 -66313236613439616338373263336535363134343562356231613166623432393536343435346565 -37383966333665633666356462663538343930366163653964626266363938613436373261353339 -31366636626338333432663730633035353935623437353231636331633064303430303765623236 -36386434633437393066333336393534356438343562326665326563353732376566626630636165 -61356139366232333737376136663266666139336664623034323536663261366636396631393162 -64663834366465383562386632343537303263353565373534336430343266356630333665363439 -32663334336565633533623533336365313035643234363930323431393533623933626637393866 -31653432336635363938663039613630623239656566623839653062643034393032316235373363 -37643863373636666639386337666332633432366539306463346138323865306465643834343138 -31363264343038346161636563653962623130373836366666366135353430666266386361346564 -64343630623038626437623532353066336665636332316238323364633162653230393862663933 -65643466313139326566613862363833393534343536656132653730623333643732313438653261 -37633031303633666531663636616336313638306261663532366236386536616465666162623665 -36316231666166613264303938616262396234343464383433653331633466303530623334666566 -32623663653132616334646634326435363832346366366139336661366336353931326232383332 -35666566663763646164393962396535333630393366373737653562373937303965306265323562 -32623730663861323363663138343138613634313839323232313130353138643134616265636437 -62613062616661383262643632303839356535373232383265363565663830363736383730636430 -35666332333961303165653662336261343837366430626530306236656332393465363934366162 -32383439363934396336633430626366653963383134326636613333373536393234353938383631 -64616632393434666135656339656266613064313237346634353831336664666463373864633231 -38663039363632373733646534636262613132633739306531663439313836313761656139353930 -32306232626361616339383931646338313532366161356634373965656139356137616238623463 -37323336323131653335636565393133313162616130343039346536353333343635383234646338 -33363838646265386438353266623732306633643362333866613537313064356631393135653637 -63306432643164333365343461353166646462373131383935653533613433306539633865373363 -30666239373563646132663734616637383965343539643330356531376661633933396233616462 -66383739376236386362623531303531343735653438303031343530363330386434666137616239 -37626333323766626130623039633462616330383934393830623430393466656164373437346166 -64633234633661663736626335323061313431313638363439653736326563393362373736613036 -64636536363865626361313465326365316564353761353834623235343939376231393831333362 -36303532356266306434656538336663383163306631663836313532343335363966663137373033 -32303036393739653530653162653038373663613536366162353863326663636431623033306232 -63353437653936383933303733663235383835623364653637376566353134666239633836613365 -36313362653237373934633831636161323161643762623761643433373466383132613034663538 -31373561336232326463373761623537663833366534353937336565316133623566633138373230 -63643838653330626565363966373735656334313834363034633066393934653331366161396136 -34646536343230306534663661646132306537663132393265663636356638636332363730383961 -30396633333962353835636134343539346332376566646335353533303331363434623635306463 -33636535343938396533623961343465353131623338626233656632333733313732363633646130 -66613230306233383435326531363632313239306566373261373031663136393334626133303761 -66363263306139343837626433656237376631646533653963333834633138353234396631333239 -31323161373435626233643763643965656465633034373061356263346635376633343836333330 -38343139663938363036336161613364323161633163643061336136316663313666336161303733 -39646139663131313537303735653665333265323531353839393432656361633761373433613064 -38653233346433353133306266623534386538393466323334643464306632663336373166633661 -35313831616663353831636663616339643631303163343637656132313236616563373861396135 -65633234396439373431663931343134393739633733393631663766653530373031306363623462 -36383063336666326232373638333430316535363630623630303433393737393431633832316232 -61343363366634663237316666636165353838343261613231373363626633323235306162653165 -64383934636634653134366438383138346233383366386463626663313033313537383031333631 -35633834633639323038666637313431326564346566376333353931336232313739313239373566 -33613862616634663630336532616530643165366534613230623133653931383064363632623538 -62616130653135366533613337343030633663313030646630373633636339386661663762393539 -30396337323932663032346530303034313636326630643966363566643965633362633631333839 -63346461646230663337303537366233613532313032386264303736366262343937393261303264 -64323666306532346530313438323338643830353832373633653334316432353666383134313465 -64313962343464396131636466613130643530383738663564323734323534386435353231306463 -37633734306134386539666165366531656234333863653535323730323537613537363834363536 -30633065376437323364373538653765646433373733356163346531646665616134626533323466 -39343030616264646233633838343233396636626461616561646664326534653762363531396463 -31366163363136316338356135653565316363666537376438666136646635373366633338356363 -66663336643366366436643265633034626239323437396165653263653434313833393032393436 -37643334633639633265393865363731346561366564383430303332343734623130383563376661 -38386135653163343163653930396261376539646238636139616236303130363561326364663962 -33623739363235333633653865623436636437616435376439326662613037313730346266623432 -33363361666138346365313234623631383264356532643433353535386562366237373734623935 -32326666323363353132383230303633383264616138656363636261393734613133386462633436 -30303830336365333664376565313734666132373066396137663431333435373738366464313863 -30393132313937353237636437343935613861313635643637383237643531333638376533316364 -62646631306231346437663234363832393162383738356235663561376333656333646365323433 -66386561373935316662663130626632323930623161383964613830303538383836396135383266 -39613536316132653666396161363164333864343963366634356632383836656662663366613461 -31343964386333653631323733633866333531366336303062353837383366613434333136633830 -65653065646633613864643266323037373231373936646465356663383838373536623461393765 -36386565393038626661623264333036623561663362633161363064383263633931373265366539 -30373137383364326466666532313266343361633436316439623564653065353534373561313265 -39383630383666633138616666633031656663333734636131363338616135363136393366333033 -33396664613130383161333862633766656637383063643230393661323233306364626230393937 -38616438313664383262353735636138316538653930626639653232393531333738313564303731 -32306238336166333934336437303636643135346238386165616539306432336639343130353935 -66383336663738336562366630653330646330363063643632363837666233656236323237383732 -36316536613866623532633233393931333339656365343064393862623662306634336466383764 -65303737373430316238336139663637383133646438633833346339643239623836343138393933 -33636337363130383064366463313439386164316133666238346237313761333437663730373061 -65363733353863396163346534306233633764326462666234373732626232623361626438376336 -65653437353832616537653338356234666238663533373831633065356562373832666238363464 -31616139386533616132313534373534656335666230656264636531306163376164333265333634 -39366539326633613966366464633535363432303238343533653131396335306639623035393434 -65366134316636663238663862366165306132323339393334306435306465373636306337623664 -36666139613361306234366237356634366563666438346537646564653464376330303165316464 -65333030656639313263316637383233373630666536636335313337656266656164326465623030 -3264653937336633613533613866386139613665366337373262 +66646566313430366435336438316665363536356431633739373931326363363137653964333765 +6438363638393236306464663136663061386361363432390a623836393662636238626337373766 +39313636386137363334313465373637613363313634363230643032386132366533323934393933 +3033616238356165630a613765643762336161663863393163616636636163376537396638666139 +62376334383839373131646334383039396661396262633537333863613133346363303133353765 +61323462646434313539633833616162376637613436626365336265333863356534626339313661 +35376634386233653964353437373838343066323035366263663738643032646631633236663066 +34383535333961313439386462326264616535366434343032633639646631373930353038356330 +37626465653362366666316363333663343734663939393637373234303935363136366636306534 +65356362313662356332626530363563333337613661626531326138633936316137626334616162 +30336664323436653431393061343935386461383130356437646361333431313938393236363633 +39303063653936353736363332386366623565313861613838303735356530663138323935323066 +66613134653235353364366263343138666639396430306439396633393433636236363363393164 +66326330313530393064656331386233363466386336313430333139643830626535653538353331 +62643631366536653137393733666238386230383634356532663563343765306334383331333161 +35313336643266653231333034626139636133653236323235346238323335663934626133616664 +65326333323334663265303933353035353739323062323962396661666161333236666133343235 +65616562346131613136363932353632343935353139386234656138303261636230333264363230 +33333933623965343862393065313238306635636361373237626233353634373737336532626431 +30636163363835366330326163616539303031393561313538393338323166613739663039396566 +61613265653633386338663837396135356563353439336534633064363330303863333166393966 +35366337353636366336346436633734363833363133366561333835306433333138666134306462 +38626333663534386237303739646664663363303866656433313733393832316533383136316233 +66363234656664346465636562643832323266346631653262373166343036363532646533313636 +63303538656662383430623365383133313665393163333832626361386663346132313163313061 +39613732366435343930346639643161656165656538653531653036393764663962363832663861 +33333031353234646239386534623536376131306261653165393634383762383265313136326636 +33333430323030636164346433316337663137643133626639303536323735623637333031356633 +36613232313330636636633266636461616463376463366234303630613966363738633133636263 +64646532333764343861633935633737323639333237643831353339636235316533613266653865 +36333535333365333436653638663364386437613136313239376538323439613965343339656461 +39616239346435343665333831363134643762646532393366366533623461353964383263356630 +31373239666266613133323333323336363739346236656261323365363664303630663934653863 +34636534376633333331356162393339653762303837363765313633333535386164386439386266 +39303336646465366336316263343461613336303930653963613866663738333433326565643736 +61303631656132653064303439646637373734313364303135643963353064363061663835663632 +31623264383238323734373732393733343336323630366439303466316238336566623666303433 +34396633653762303730323334356166303134376535316131333266303438313562306135343233 +30303136333730626561656639623866666663646131383666613633663966636462663031626639 +65383732313561376261333534643937383165616137656232623664386232353337343563333262 +34633364323839353136616636366462666637313438396661303663316263353636363163633236 +64326265623262323735396637663538626139366165643437333233643830313962663735373063 +37363664363432326637386335386432616135666162363431356561333635616134626136316365 +38323539663966643262656635356334386236393538333034303565393439353931663466373966 +31376466393363396535396263336338306239613961356130613335316132313335636634353461 +35343830633036653337356131386538396139323265653230396664653533636239613239356639 +66316332323538656166393037636563626338386235633230623962373731386230383733366663 +64383064623362626639383030396339623138373935373539643335366531393635316137323266 +61383065353766626463636338383563366532323439616333623730386630656163343536646562 +63343538616139363933316266363531656661633532353765366534383735366664616565623030 +33373561386537613737653764326439366632626537346464363132303133393963313966373163 +32623239326631333338303736613465616435346535336332633163313439656232626239653836 +33343638643562626338353137346365386431663334336231656435373238323535323733623461 +37356533613738663764656466396632366430316433623334666665363463386131303832326564 +62613163366661346435343031373433343236643839396536383966363863393836623265393135 +62303838393539366166383137343737383432333235633230363036656466373837396462393564 +61353535623738616239663535333666323839666133346336646533633735623639376332656464 +35633839633665313033353834383530336164346233643033373336656265363266363534316465 +36346536383535333138313761356561323661643163393831343962633036373265643938336533 +63623135386630356262386563353634623439383631633866366661336336613038623439326264 +39353933643233323863343839656634373732363661373966313839363130646533323737646433 +33626634643964336236646233376235376230616435323930323935373730633663393435646331 +64613839366663663933626537323339613363643839313632313034633135643166323235356632 +66643437353130333831306530653834386137386632626363646563343631333235653065353730 +38323363373665623961306530316562353165656138393134653933303239663636643464306136 +66303234353537633062663735653232366639373030336330336635323065643935323936396333 +64316661323263386664323031333030326534306566333934306361353233373536356435326338 +33303561326233656233366431653237333763326434313162663165633637636262343130636133 +61366635663730653539386366353237633438323966633961303032366136336630623436393035 +62633463313338353432323939343239623630383565636162616139613664613863396165643736 +35376462333938656432373133383838623035653131646366373239306262323539313439343166 +66323830616362363263636663333566386239353765343830343465353265333262313331343434 +30313339303064313263633864623261616130343530336134333661333931663564323461353034 +64663737393430383233633330306565363230383761346362663530646238363631383062646230 +62633432623563353239353737623938663437303061623866363431343436313837343539343436 +30393737346435363563346235633438306339366639613161383763613730353437636537366565 +66396534353230613137326139333435393034346266643364646138666130656532323166623664 +36656631336636393131316361376339383333333837663263343065623937663062346666323464 +31636662386362633337306338633138623434323563643663663835656430333762366631343033 +63343861613365336434653035333132636363656139393065356330376161383039633533393366 +38326630396535383830303437666333623735393131336630663936373465326466616133343735 +34393535326335646265383333646233386434636530613464643266326134396462313762363066 +64303637386261653866623334666532323130393838313034633038356361363239386435656239 +63386163653362386565646561353238653962343033386632336364313630353063356336626338 +63356137616530643434313862643235333063363533666635346661363861383366326533646665 +66316531633163313365653337663633393661313466643331346230613539653663336637353162 +63323730386433626364333632383761323362363664306264336334613766373532386533336662 +65396336313332633539633966313733386335613236666234313664303631326361376232383036 +33663563613864643532363038303832363434386265366637376561373434643239396231643634 +61633935363234383935393333666566383766646461623061613361636531386363363238393465 +66643636633465343033306437626632373734323762666164356231333137376535313831373236 +66373435323339316533383432393766633566306635336166336336643532313430393539336332 +37613666333763396439396466313535383066636237343964306661636337383931636631633463 +64383130613964376462663837383762353634623064313431623266383839333434653537626663 +34363736643937326134616136306236656261313136366635303036646435353863306662353034 +39363266356636623135356232383630373162363330396262323734643564366232333632373739 +36656366633335626131643930366534326164646265643339623539323863393632353838633861 +36363763313039633530623736313239353036666162346432643732343438366265306164383235 +66376536653761383665316665396435646634363436376632646563633163623766366635623731 +66623831393230326566313932653536343838626566623038653534623434313239343763313433 +30376630656638623036336166346133633738303638366238666639653030323832323365353766 +33623261343664323631393166313539623734393438346562613735623636396137333733303235 +34636630643932383536633836343765373234623639643437306532353764386435663665643738 +65313065393534383935303639656534633363303235393761343330386630633734313561376433 +63623464653034336661386435386431306634613239393039386439316538343261306234393534 +30323164363734303037663134663734363636653935336539396330383931353461353966356533 +63316462376262626261623031353063353030363230653966393832626662623963306266393635 +34643565386137663435373639393537626632373535333865366165353839343732353439616265 +37373563326438656134363931623565313565666334323533653264356633353931323661663832 +35396637353163383466653765376132353561343031623039383161623034396134626562666632 +64643737306634613966663362373038396231616632636539663966313364383137326164396533 +38643266316464366161343761633332633036383537653565653139336562666461666465376230 +61656361623264366566656461373863656161393333353933326665333730356333343531653766 +36303964353134396533306630343832303961336239643462353439363962353738663731663033 +38343833323365366662343665633130306537383861613762613763646631626330613937393432 +36393338626661306130353461653565313766623762613564356466393663383339356335306436 +35373636396239653361633764366133363334643232643734383335346635386139383130363938 +35643665316139613639666436633531396537353439633839623063383134376230376235366534 +64393634376633343063646433323439303438303432323335393265366633343332616461386665 +62626132346364373931303936623963663037333665333639333966646431663635303365343630 +36316464346266313166356262346331656432363865643135613964356662366632323736613437 +37336561376334303534653237303964656665643561376237653531653764363463363464313737 +66313335383838333233663162336337373764363131643836363365353039323836346433313139 +33633636306131623335623639666130393535646536346438303435366436623132613562643066 +65623336643463333932383865353232333339343231613964373566653063643662376338316437 +65343466396530386230396236656463396566336230343838366264656161323063336264643364 +31643066366632653962323636346565313833393966343761623066336464333733646238386332 +37393534356333396661353130363462633366343535616338653330393262663066326662366530 +30383536653838333066623063306233613735363665623937373362623337323339373663303933 +64306538636533643337373766346266326238653937613366616161336333636662386238346237 +35623237333131623363376230663132373939366235376631626263393764366661346438383537 +33306233663630323863343839636166303461666135313639356637303435643062303364646132 +64656636396264303962663363376166323834393863393264323464336662393930636533363834 +61363937383766623930623537353037623031363736633332393766323238383464663531653438 +34653933623166326530343632386165666266303438623739356635346163346239643065336537 +33316631613431623030323433306133376237316535306666343433646339313965663565653130 +61303065303530616639383066623534353566366632313465316136613437383765643039313137 +64393333353834353033353532373834343831326535393837323037373136303462373530646430 +62633964626164626139373562616430613662646265396139356561666462376435643031643964 +62316339303561666435663438393838356437646336623664613863303134343261373830303730 +36303732393262363530366563343165386436313833633533623637646464616663353035363235 +31353764353135343033363562313465383563376164323766323735646336353834613264316464 +65376134346437333439376636313531666638313333313963636363353965363239323830663434 +30373234373235326635633063366232333235653935616365656135396465366564313530363161 +33363864303063363363363535323333616661323338323735323135613562366163393366393963 +31363331373661636134333063363731343831333061326431633061663064356466306438646565 +62646431613861393237313138353330323734636666333934666437343665653965316436636464 +64333432313561656339366533363765343365613139626331396636376535326165623735346633 +61666364373465613438643032316362303335333431306166393766663032666665666564313837 +38323232373439376563633939623835656464313361333666623336666461383330663764666166 +34343638646637646338313737336566623937616666626637386639346530616439316635396661 +61656439346231663436386665653031643334393734346163353963633333646239623536333730 +39363731316361633139326637303839633633633738633737326465343664336261646337343831 +33623261363136613935316635663434633962613635616333383238613965646637363735646264 +39626635323532396339653861383137336136346539303931323262663635366432376365623732 +35356639376161323937306136376236643431643863393630633662346531343230643636666633 +33636139323663353835313663656162656264313739663066616663346639326434353762666330 +62303235363964323964663664353265316461373637346230646364393065653763646636303664 +62623735633234303063306531393662326363333234333763613335666233373131303334366665 +30633037323934633232316366663563326562353834643132623031613138323835373933376536 +37323638396561643363663235333632323234303734643637663261393736613738346433333539 +33623865363862653731366466333530343561653661653638353738346438633236343730326239 +35363837393736343632663765626635613865656165396336646538376434653537633037616566 +37393766636661383538386164363630626438616365666532333039663464303835373134333432 +30386439333732666266666331633566326463316438333639336665613933343963363562636335 +31356465303535663233656466616462646161666538346334323335396561656634396631356262 +62323561316261653134363436636333353665613165323935353439666164363163303530353964 +66633536306432323335306530323733623536383366653566653339383161383434613330343934 +63666131333963356336333763386338343437613035313036393766636535386234613665636632 +34383032633931663633386636343233646131393066393766366635343935356538306231336232 +32353836353335363731353234303337343737363064636463613132393563316136626539326264 +30343230343166393466353664613438383462623734363038363962306437663938306235316235 +64373433626366636133373964633131636534333839623033383163613561363430343534313966 +63366561613939623334653434616238616136353665386665346366666236326363376135373363 +36343965343131356662373834366535313963333366633536343930633561626230373736303030 +65373063386232316332343963343631333162376334636665633531313936303232323933313263 +31373639613365396533303431376532643434626538653538356630626532343530663163366232 +39376566353361333639343237653937303161646164613831643235613139373961663566333630 +62663430313832663432306634356464623838303835373962373533313565663735326535326461 +66653939623836666461626631656534616162323433356431303366306635323938373138613031 +66366461353133386436633765386262373736633539643637373237326235633764353132636561 +63343438666361303364333732303738323463353832613438636663363332333132373361633439 +64373334386661626364393664346262653739613134346365636132613538666339393036333163 +64373363643031393062316135336561303665623465336464356165313466633462666331303030 +62626535396634623366313662643066313462636330343237316433646434313737626562346237 +65393161333336386430353938336536336461326132393933343335656131393664356463376631 +38373862363862343839666161663136336534633862303636653931363136303337353664393737 +35316138353361306133636463366463333137616562366135653230343039643365646665636364 +65366564643738356539663331613439656536376263306362366165306230373336386166613865 +33626631623364663062396466626531666238616436326163643533386366633164633262356431 +32313762336561396531386535383434396135633066316334626632633431356363373034663135 +33323166373562326237636265353634623236666364643837353232306263643132343761373230 +63626331393461653662383666646332303632386634313034313664346432656265 diff --git a/ansible/host_vars/nodito/main.yml b/ansible/host_vars/nodito/main.yml index c0002f3..a6f6a58 100644 --- a/ansible/host_vars/nodito/main.yml +++ b/ansible/host_vars/nodito/main.yml @@ -14,7 +14,12 @@ systemd_service_name: nodito-cpu-temp-monitor # ZFS Pool Configuration zfs_pool_name: "proxmox-tank-1" -zfs_disk_1: "/dev/disk/by-id/ata-ST4000NT001-3M2101_WX11TN0Z" # First disk for RAID 1 mirror +# Corrected 2026-09-13: this said WX11TN0Z, a disk that is no longer in the +# machine. The live mirror is WX120LHQ + WX11TN2P - a leg was evidently +# replaced and the repo never caught up. Pool creation is guarded by +# `when: zfs_pool_exists.rc != 0` so it was inert, but it would have been +# wrong on any disaster-recovery run. +zfs_disk_1: "/dev/disk/by-id/ata-ST4000NT001-3M2101_WX120LHQ" # First disk for RAID 1 mirror zfs_disk_2: "/dev/disk/by-id/ata-ST4000NT001-3M2101_WX11TN2P" # Second disk for RAID 1 mirror zfs_pool_mountpoint: "/var/lib/vz" diff --git a/ansible/host_vars/nodito/vault.yml b/ansible/host_vars/nodito/vault.yml new file mode 100644 index 0000000..e64085e --- /dev/null +++ b/ansible/host_vars/nodito/vault.yml @@ -0,0 +1,11 @@ +$ANSIBLE_VAULT;1.1;AES256 +30333035323663393939343061323234336164396465623665346165393534646366333332376463 +3364373463333664363334373964323838336531353364310a636636373539623464336630666164 +61376532616339376562373238383436306664313564663266303534346461666466383965323538 +3163313239626663310a613033336332653165333537313366636361663036383031376561613761 +31313563373062333033323037653939663762343161656264633436343361663737626366663732 +39366666626338323436383134646263643538333564313566346336323563663534653161396136 +39333565393538366238643563323630346166643461643063393631643665363566623631373762 +36643866646637306231653837363838656163613766636265383139333838396535626335343163 +32613935353330636263616333666230323436663935326133636362343836323535623237646235 +6266316363366335323162663039366137633865396237373632 diff --git a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml index 8cdde6a..97e0cab 100644 --- a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml +++ b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml @@ -3,7 +3,6 @@ become: true vars_files: - ../infra_vars.yml - - nodito_vars.yml tasks: - name: Verify Proxmox VE is running @@ -139,17 +138,18 @@ Config file exists: {{ storage_cfg_file.stat.exists }} Storage check result: {{ storage_exists_check.rc }} Pool exists: {{ zfs_pool_exists.rc == 0 }} - Will remove storage: {{ zfs_pool_exists.rc == 0 and storage_exists_check.rc == 0 }} Will add storage: {{ zfs_pool_exists.rc == 0 and storage_exists_check.rc != 0 }} - - name: Remove existing storage if it exists - command: pvesm remove {{ zfs_pool_name }} - register: pvesm_remove_result - failed_when: false - when: - - zfs_pool_exists.rc == 0 - - storage_exists_check.rc == 0 - + # Registration is add-only on purpose. There used to be a "Remove existing + # storage if it exists" task here that ran `pvesm remove` whenever the + # storage WAS present, paired with an add that only ran when it was ABSENT. + # The two conditions are mutually exclusive, so a real run against a + # correctly-configured hypervisor removed the storage entry backing every VM + # and never put it back. It would also have dropped `mountpoint /var/lib/vz`, + # which the live entry has and which `pvesm add` below does not set. + # + # If the storage entry ever needs its options changed, edit + # /etc/pve/storage.cfg or use `pvesm set` - do not re-register it from here. - name: Add ZFS pool storage to Proxmox using pvesm command: > pvesm add zfspool {{ zfs_pool_name }} @@ -171,27 +171,24 @@ msg: "ZFS pool {{ zfs_pool_name }} is not in a healthy state" when: "'ONLINE' not in final_zfs_status.stdout" -# ═════════════════════════════════════════════════════════════════════════════ -# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. +# ───────────────────────────────────────────────────────────────────────────── +# ZFS health monitoring and monthly scrub. # -# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username -# and uptime_kuma_password were removed from the vault, so the "Validate Uptime -# Kuma configuration" assert stops it before anything is installed or changed. +# The check script decides healthy/unhealthy and says so in its exit code, which +# systemd keeps: systemctl is-failed zfs-health-monitor.service # -# It is kept because the CHECK LOGIC is the durable part — what gets measured, -# the thresholds, and the systemd timer plumbing. When something replaces Uptime -# Kuma, only the push transport needs rewriting; the rest still applies. -# -# What was being monitored: archive/uptime_kuma/MONITORS.md -# ═════════════════════════════════════════════════════════════════════════════ +# Reporting anywhere else is optional. Set `healthcheck_push_url` and the script +# will also GET it with ?status=up|down; leave it empty and the exit code is +# still the whole answer. Today that URL points at Uptime Kuma, which is still +# running on watchtower but is no longer deployed by Ansible - the credentials +# were retired, the service was not. If it is ever replaced, `healthcheck_push_url` +# is the only thing that needs to change here. +# ───────────────────────────────────────────────────────────────────────────── - name: Setup ZFS Pool Health Monitoring and Monthly Scrubs hosts: hypervisor become: true vars_files: - ../../infra_vars.yml - - ../../services_config.yml - - ../../infra_secrets.yml - - nodito_vars.yml vars: zfs_check_interval_seconds: 86400 # 24 hours @@ -202,143 +199,10 @@ zfs_log_file: "{{ zfs_monitoring_script_dir }}/zfs_health_monitor.log" zfs_systemd_health_service_name: zfs-health-monitor zfs_systemd_scrub_service_name: zfs-monthly-scrub - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - ntfy_topic: "{{ service_settings.ntfy.topic }}" + # Optional. Empty is fine and is not an error - see the banner above. + healthcheck_push_url: "{{ healthcheck_push_urls.zfs_health | default('') }}" tasks: - - name: Validate Uptime Kuma configuration - assert: - that: - - uptime_kuma_api_url is defined - - uptime_kuma_api_url != "" - - uptime_kuma_username is defined - - uptime_kuma_username != "" - - uptime_kuma_password is defined - - uptime_kuma_password != "" - fail_msg: "uptime_kuma_api_url, uptime_kuma_username and uptime_kuma_password must be set" - - - name: Get hostname for monitor identification - command: hostname - register: host_name - changed_when: false - - - name: Set monitor name and group based on hostname - set_fact: - monitor_name: "zfs-health-{{ host_name.stdout }}" - monitor_friendly_name: "ZFS Pool Health: {{ host_name.stdout }}" - uptime_kuma_monitor_group: "{{ host_name.stdout }} - infra" - - - name: Create Uptime Kuma ZFS health monitor setup script - copy: - dest: /tmp/setup_uptime_kuma_zfs_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - retries = int(sys.argv[8]) - ntfy_topic = sys.argv[9] if len(sys.argv) > 9 else "alerts" - - api = UptimeKumaApi(api_url, timeout=120, wait_events=2.0) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Get all notifications and find ntfy notification - notifications = api.get_notifications() - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - # Find or create group - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name=group_name) - # Refresh to get the full group object with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - # Find or create/update push monitor - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': False, # Normal heartbeat mode: receiving pings = healthy - 'maxretries': retries, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - monitor = api.edit_monitor(existing_monitor['id'], **monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - else: - monitor_result = api.add_monitor(**monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - # Output result as JSON - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma ZFS monitor setup script - command: > - {{ ansible_playbook_python }} - /tmp/setup_uptime_kuma_zfs_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "{{ uptime_kuma_monitor_group }}" - "{{ monitor_name }}" - "{{ monitor_friendly_name }} - Daily health check for pool {{ zfs_pool_name }}" - "{{ zfs_check_timeout_seconds }}" - "{{ zfs_check_retries }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL and monitor ID as facts - set_fact: - uptime_kuma_zfs_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - uptime_kuma_monitor_id: "{{ monitor_info_parsed.monitor_id }}" - - name: Install required packages for ZFS monitoring package: name: @@ -355,264 +219,41 @@ mode: '0755' - name: Create ZFS health monitoring script - copy: + template: dest: "{{ zfs_monitoring_script_path }}" - content: | - #!/bin/bash - - # ZFS Pool Health Monitoring Script - # Checks ZFS pool health using JSON output and sends heartbeat to Uptime Kuma if healthy - # If any issues detected, does NOT send heartbeat (triggers timeout alert) - - LOG_FILE="{{ zfs_log_file }}" - UPTIME_KUMA_URL="{{ uptime_kuma_zfs_push_url }}" - POOL_NAME="{{ zfs_pool_name }}" - HOSTNAME=$(hostname) - - # Function to log messages - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - # Function to check pool health using JSON output - check_pool_health() { - local pool="$1" - local issues_found=0 - - # Get pool status as JSON - local pool_json - pool_json=$(zpool status -j "$pool" 2>&1) - - if [ $? -ne 0 ]; then - log_message "ERROR: Failed to get pool status for $pool" - log_message " -> $pool_json" - return 1 - fi - - # Check 1: Pool state must be ONLINE - local pool_state - pool_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].state') - - if [ "$pool_state" != "ONLINE" ]; then - log_message "ISSUE: Pool state is $pool_state (expected ONLINE)" - issues_found=1 - else - log_message "OK: Pool state is ONLINE" - fi - - # Check 2: Check all vdevs and devices for non-ONLINE states - local bad_states - bad_states=$(echo "$pool_json" | jq -r --arg pool "$pool" ' - .pools[$pool].vdevs[] | - .. | objects | - select(.state? and .state != "ONLINE") | - "\(.name // "unknown"): \(.state)" - ' 2>/dev/null) - - if [ -n "$bad_states" ]; then - log_message "ISSUE: Found devices not in ONLINE state:" - echo "$bad_states" | while read -r line; do - log_message " -> $line" - done - issues_found=1 - else - log_message "OK: All devices are ONLINE" - fi - - # Check 3: Check for resilvering in progress - local scan_function scan_state - scan_function=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.function // "NONE"') - scan_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.state // "NONE"') - - if [ "$scan_function" = "RESILVER" ] && [ "$scan_state" = "SCANNING" ]; then - local resilver_progress - resilver_progress=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.issued // "unknown"') - log_message "ISSUE: Pool is currently resilvering (disk reconstruction in progress) - ${resilver_progress} processed" - issues_found=1 - fi - - # Check 4: Check for read/write/checksum errors on all devices - # Note: ZFS JSON output has error counts as strings, so convert to numbers for comparison - local devices_with_errors - devices_with_errors=$(echo "$pool_json" | jq -r --arg pool "$pool" ' - .pools[$pool].vdevs[] | - .. | objects | - select(.name? and ((.read_errors // "0" | tonumber) > 0 or (.write_errors // "0" | tonumber) > 0 or (.checksum_errors // "0" | tonumber) > 0)) | - "\(.name): read=\(.read_errors // 0) write=\(.write_errors // 0) cksum=\(.checksum_errors // 0)" - ' 2>/dev/null) - - if [ -n "$devices_with_errors" ]; then - log_message "ISSUE: Found devices with I/O errors:" - echo "$devices_with_errors" | while read -r line; do - log_message " -> $line" - done - issues_found=1 - else - log_message "OK: No read/write/checksum errors detected" - fi - - # Check 5: Check for scan errors (from last scrub/resilver) - local scan_errors - scan_errors=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.errors // "0"') - - if [ "$scan_errors" != "0" ] && [ "$scan_errors" != "null" ] && [ -n "$scan_errors" ]; then - log_message "ISSUE: Last scan reported $scan_errors errors" - issues_found=1 - else - log_message "OK: No scan errors" - fi - - return $issues_found - } - - # Function to get last scrub info for status message - get_scrub_info() { - local pool="$1" - local pool_json - pool_json=$(zpool status -j "$pool" 2>/dev/null) - - local scan_func scan_state scan_start - scan_func=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.function // "NONE"') - scan_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.state // "NONE"') - scan_start=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.start_time // ""') - - if [ "$scan_func" = "SCRUB" ] && [ "$scan_state" = "SCANNING" ]; then - echo "scrub in progress (started $scan_start)" - elif [ "$scan_func" = "SCRUB" ] && [ -n "$scan_start" ]; then - echo "last scrub: $scan_start" - else - echo "no scrub history" - fi - } - - # Function to send heartbeat to Uptime Kuma - send_heartbeat() { - local message="$1" - - log_message "Sending heartbeat to Uptime Kuma: $message" - - # URL encode the message - local encoded_message - encoded_message=$(printf '%s\n' "$message" | sed 's/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') - - local response http_code - response=$(curl -s -w "\n%{http_code}" "$UPTIME_KUMA_URL?status=up&msg=$encoded_message" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Heartbeat sent successfully (HTTP $http_code)" - return 0 - else - log_message "ERROR: Failed to send heartbeat (HTTP $http_code)" - return 1 - fi - } - - # Main health check logic - main() { - log_message "==========================================" - log_message "Starting ZFS health check for pool: $POOL_NAME on $HOSTNAME" - - # Run all health checks - if check_pool_health "$POOL_NAME"; then - # All checks passed - send heartbeat - local scrub_info - scrub_info=$(get_scrub_info "$POOL_NAME") - - local message="Pool $POOL_NAME healthy ($scrub_info)" - send_heartbeat "$message" - - log_message "Health check completed: ALL OK" - exit 0 - else - # Issues found - do NOT send heartbeat (will trigger timeout alert) - log_message "Health check completed: ISSUES DETECTED - NOT sending heartbeat" - log_message "Uptime Kuma will alert after timeout due to missing heartbeat" - exit 1 - fi - } - - # Run main function - main + src: templates/zfs_health_monitor.sh.j2 owner: root group: root mode: '0755' - name: Create systemd service for ZFS health monitoring - copy: + template: dest: "/etc/systemd/system/{{ zfs_systemd_health_service_name }}.service" - content: | - [Unit] - Description=ZFS Pool Health Monitor - After=zfs.target network.target - - [Service] - Type=oneshot - ExecStart={{ zfs_monitoring_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target + src: templates/zfs-health-monitor.service.j2 owner: root group: root mode: '0644' - name: Create systemd timer for daily ZFS health monitoring - copy: + template: dest: "/etc/systemd/system/{{ zfs_systemd_health_service_name }}.timer" - content: | - [Unit] - Description=Run ZFS Pool Health Monitor daily - Requires={{ zfs_systemd_health_service_name }}.service - - [Timer] - OnBootSec=5min - OnUnitActiveSec={{ zfs_check_interval_seconds }}sec - Persistent=true - - [Install] - WantedBy=timers.target + src: templates/zfs-health-monitor.timer.j2 owner: root group: root mode: '0644' - name: Create systemd service for ZFS monthly scrub - copy: + template: dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.service" - content: | - [Unit] - Description=ZFS Monthly Scrub for {{ zfs_pool_name }} - After=zfs.target - - [Service] - Type=oneshot - ExecStart=/sbin/zpool scrub {{ zfs_pool_name }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target + src: templates/zfs-monthly-scrub.service.j2 owner: root group: root mode: '0644' - name: Create systemd timer for monthly ZFS scrub - copy: + template: dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.timer" - content: | - [Unit] - Description=Run ZFS Scrub on last day of every month at 4:00 AM - Requires={{ zfs_systemd_scrub_service_name }}.service - - [Timer] - OnCalendar=*-*~01 04:00:00 - Persistent=true - - [Install] - WantedBy=timers.target + src: templates/zfs-monthly-scrub.timer.j2 owner: root group: root mode: '0644' @@ -649,9 +290,8 @@ msg: | ✓ ZFS Pool Health Monitoring deployed successfully! - Monitor Name: {{ monitor_friendly_name }} - Monitor Group: {{ uptime_kuma_monitor_group }} Pool Name: {{ zfs_pool_name }} + Push URL: {{ healthcheck_push_url | default('', true) | ternary('set', 'not set - exit code only') }} Health Check: - Frequency: Every {{ zfs_check_interval_seconds }} seconds (24 hours) @@ -672,10 +312,3 @@ - Resilver status (alerts if resilvering) - Read/Write/Checksum errors - Scrub errors - - - name: Clean up temporary Uptime Kuma setup script - file: - path: /tmp/setup_uptime_kuma_zfs_monitor.py - state: absent - delegate_to: localhost - become: no diff --git a/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml b/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml index 3b687ee..e2b1f5d 100644 --- a/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml +++ b/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml @@ -3,7 +3,6 @@ become: true vars_files: - ../../infra_vars.yml - - nodito_vars.yml vars: # Defaults (override via vars_files or --extra-vars as needed) diff --git a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml index f39fb56..da1136f 100644 --- a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml +++ b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml @@ -3,10 +3,33 @@ become: true vars_files: - ../../infra_vars.yml - - nodito_vars.yml - - nodito_secrets.yml tasks: + # ------------------------------------------------------------------ + # Safety catch + # + # /etc/nut/upsd.users and /etc/nut/upsmon.conf on nodito were written by + # hand in January 2026 and carry a working password. host_vars/nodito/vault.yml + # (formerly infra/nodito/nodito_secrets.yml) still holds the literal string + # CHANGE_ME_TO_SECURE_PASSWORD, so running this play would overwrite that + # working pair with a placeholder and restart NUT - leaving the hypervisor's + # UPS unmonitored and unable to trigger a clean shutdown on mains loss. + # + # Until the real password is put in the vault, stop here. + # ansible-vault edit host_vars/nodito/vault.yml + # ------------------------------------------------------------------ + - name: Refuse to run with a placeholder UPS password + assert: + that: + - ups_password is defined + - ups_password | length > 0 + - ups_password != "CHANGE_ME_TO_SECURE_PASSWORD" + fail_msg: >- + ups_password is unset or still the placeholder. Applying this play would + overwrite the working /etc/nut/upsd.users and /etc/nut/upsmon.conf on + nodito and restart NUT. Put the real password in the vault first: + ansible-vault edit host_vars/nodito/vault.yml + # ------------------------------------------------------------------ # Installation # ------------------------------------------------------------------ @@ -75,90 +98,45 @@ # Configuration files # ------------------------------------------------------------------ - name: Configure NUT mode (standalone) - copy: + template: dest: /etc/nut/nut.conf - content: | - # Managed by Ansible - MODE=standalone + src: templates/nut.conf.j2 owner: root group: nut mode: "0640" notify: Restart NUT services - name: Configure UPS device - copy: + template: dest: /etc/nut/ups.conf - content: | - # Managed by Ansible - [{{ ups_name }}] - driver = {{ ups_driver }} - port = {{ ups_port }} - desc = "{{ ups_desc }}" - offdelay = {{ ups_offdelay }} - ondelay = {{ ups_ondelay }} + src: templates/ups.conf.j2 owner: root group: nut mode: "0640" notify: Restart NUT services - name: Configure upsd to listen on localhost - copy: + template: dest: /etc/nut/upsd.conf - content: | - # Managed by Ansible - LISTEN 127.0.0.1 3493 + src: templates/upsd.conf.j2 owner: root group: nut mode: "0640" notify: Restart NUT services - name: Configure upsd users - copy: + template: dest: /etc/nut/upsd.users - content: | - # Managed by Ansible - [{{ ups_user }}] - password = {{ ups_password }} - upsmon master + src: templates/upsd.users.j2 owner: root group: nut mode: "0640" notify: Restart NUT services - name: Configure upsmon - copy: + template: dest: /etc/nut/upsmon.conf - content: | - # Managed by Ansible - MONITOR {{ ups_name }}@localhost 1 {{ ups_user }} {{ ups_password }} master - - MINSUPPLIES 1 - SHUTDOWNCMD "/sbin/shutdown -h +0" - POLLFREQ 5 - POLLFREQALERT 5 - HOSTSYNC 15 - DEADTIME 15 - POWERDOWNFLAG /etc/killpower - - # Notifications - NOTIFYMSG ONLINE "UPS %s on line power" - NOTIFYMSG ONBATT "UPS %s on battery" - NOTIFYMSG LOWBATT "UPS %s battery is low" - NOTIFYMSG FSD "UPS %s: forced shutdown in progress" - NOTIFYMSG COMMOK "Communications with UPS %s established" - NOTIFYMSG COMMBAD "Communications with UPS %s lost" - NOTIFYMSG SHUTDOWN "Auto logout and shutdown proceeding" - NOTIFYMSG REPLBATT "UPS %s battery needs replacing" - - # Log all events to syslog - NOTIFYFLAG ONLINE SYSLOG - NOTIFYFLAG ONBATT SYSLOG - NOTIFYFLAG LOWBATT SYSLOG - NOTIFYFLAG FSD SYSLOG - NOTIFYFLAG COMMOK SYSLOG - NOTIFYFLAG COMMBAD SYSLOG - NOTIFYFLAG SHUTDOWN SYSLOG - NOTIFYFLAG REPLBATT SYSLOG + src: templates/upsmon.conf.j2 owner: root group: nut mode: "0640" @@ -250,28 +228,33 @@ - nut-monitor -# ═════════════════════════════════════════════════════════════════════════════ -# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. +# ───────────────────────────────────────────────────────────────────────────── +# UPS heartbeat monitoring. # -# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username -# and uptime_kuma_password were removed from the vault, so the "Validate Uptime -# Kuma configuration" assert stops it before anything is installed or changed. +# The script decides on-mains/on-battery and says so in its exit code, which +# systemd keeps: systemctl is-failed ups-heartbeat.service # -# It is kept because the CHECK LOGIC is the durable part — what gets measured, -# the thresholds, and the systemd timer plumbing. When something replaces Uptime -# Kuma, only the push transport needs rewriting; the rest still applies. +# Reporting anywhere else is optional. Set `healthcheck_push_url` and the script +# will also GET it with ?status=up|down; leave it empty and the exit code is +# still the whole answer. Today that URL points at Uptime Kuma, which is still +# running on watchtower but is no longer deployed by Ansible - the credentials +# were retired, the service was not. If it is ever replaced, `healthcheck_push_url` +# is the only thing that needs to change here. # -# What was being monitored: archive/uptime_kuma/MONITORS.md -# ═════════════════════════════════════════════════════════════════════════════ -- name: Setup UPS Heartbeat Monitoring with Uptime Kuma +# NOTE: this play has never actually been applied to nodito. /opt/ups-monitoring +# does not exist and there is no ups-heartbeat timer. What is on the box is a +# hand-written /usr/local/bin/ups-heartbeat.sh - mode 0644, not executable, and +# referenced by no unit and no cron entry, so nothing has ever run it. Its push +# token (uLmCPkLLO4) belongs to a monitor that DOES still exist and answer, so +# that monitor has had no heartbeat since January 2026. It is now +# healthcheck_push_urls.ups in the vault, and running this play is what will +# finally start feeding it. +# ───────────────────────────────────────────────────────────────────────────── +- name: Setup UPS Heartbeat Monitoring hosts: hypervisor become: true vars_files: - ../../infra_vars.yml - - ../../services_config.yml - - ../../infra_secrets.yml - - nodito_vars.yml - - nodito_secrets.yml vars: ups_heartbeat_interval_seconds: 60 @@ -281,134 +264,10 @@ ups_monitoring_script_path: "{{ ups_monitoring_script_dir }}/ups_heartbeat.sh" ups_log_file: "{{ ups_monitoring_script_dir }}/ups_heartbeat.log" ups_systemd_service_name: ups-heartbeat - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - ntfy_topic: "{{ service_settings.ntfy.topic }}" + # Optional. Empty is fine and is not an error - see the banner above. + healthcheck_push_url: "{{ healthcheck_push_urls.ups | default('') }}" tasks: - - name: Validate Uptime Kuma configuration - assert: - that: - - uptime_kuma_api_url is defined - - uptime_kuma_api_url != "" - - uptime_kuma_username is defined - - uptime_kuma_username != "" - - uptime_kuma_password is defined - - uptime_kuma_password != "" - fail_msg: "uptime_kuma_api_url, uptime_kuma_username and uptime_kuma_password must be set" - - - name: Get hostname for monitor identification - command: hostname - register: host_name - changed_when: false - - - name: Set monitor name and group based on hostname - set_fact: - monitor_name: "ups-{{ host_name.stdout }}" - monitor_friendly_name: "UPS Status: {{ host_name.stdout }}" - uptime_kuma_monitor_group: "{{ host_name.stdout }} - infra" - - - name: Create Uptime Kuma UPS monitor setup script - copy: - dest: /tmp/setup_uptime_kuma_ups_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - retries = int(sys.argv[8]) - ntfy_topic = sys.argv[9] if len(sys.argv) > 9 else "alerts" - - api = UptimeKumaApi(api_url, timeout=120, wait_events=2.0) - api.login(username, password) - - monitors = api.get_monitors() - notifications = api.get_notifications() - - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - api.add_monitor(type='group', name=group_name) - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': False, # Normal heartbeat mode: receiving pings = healthy - 'maxretries': retries, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - api.edit_monitor(existing_monitor['id'], **monitor_data) - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - else: - api.add_monitor(**monitor_data) - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma UPS monitor setup script - command: > - {{ ansible_playbook_python }} - /tmp/setup_uptime_kuma_ups_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "{{ uptime_kuma_monitor_group }}" - "{{ monitor_name }}" - "{{ monitor_friendly_name }} - Alerts when UPS goes on battery or loses communication" - "{{ ups_heartbeat_timeout_seconds }}" - "{{ ups_heartbeat_retries }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL as fact - set_fact: - uptime_kuma_ups_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - - name: Install required packages for UPS monitoring package: name: @@ -424,107 +283,25 @@ mode: '0755' - name: Create UPS heartbeat monitoring script - copy: + template: dest: "{{ ups_monitoring_script_path }}" - content: | - #!/bin/bash - - # UPS Heartbeat Monitoring Script - # Sends heartbeat to Uptime Kuma only when UPS is on mains power - # When on battery or communication lost, no heartbeat is sent (triggers timeout alert) - - LOG_FILE="{{ ups_log_file }}" - UPTIME_KUMA_URL="{{ uptime_kuma_ups_push_url }}" - UPS_NAME="{{ ups_name }}" - - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - send_heartbeat() { - local message="$1" - - local encoded_message - encoded_message=$(printf '%s\n' "$message" | sed 's/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g; s/%/%25/g') - - local response http_code - response=$(curl -s -w "\n%{http_code}" "$UPTIME_KUMA_URL?status=up&msg=$encoded_message" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Heartbeat sent: $message (HTTP $http_code)" - return 0 - else - log_message "ERROR: Failed to send heartbeat (HTTP $http_code)" - return 1 - fi - } - - main() { - local status charge runtime load - - status=$(upsc ${UPS_NAME}@localhost ups.status 2>/dev/null) - - if [ -z "$status" ]; then - log_message "ERROR: Cannot communicate with UPS - NOT sending heartbeat" - exit 1 - fi - - charge=$(upsc ${UPS_NAME}@localhost battery.charge 2>/dev/null) - runtime=$(upsc ${UPS_NAME}@localhost battery.runtime 2>/dev/null) - load=$(upsc ${UPS_NAME}@localhost ups.load 2>/dev/null) - - if [[ "$status" == *"OL"* ]]; then - local message="UPS on mains (charge=${charge}% runtime=${runtime}s load=${load}%)" - send_heartbeat "$message" - exit 0 - else - log_message "UPS not on mains power (status=$status) - NOT sending heartbeat" - exit 1 - fi - } - - main + src: templates/ups_heartbeat.sh.j2 owner: root group: root mode: '0755' - name: Create systemd service for UPS heartbeat - copy: + template: dest: "/etc/systemd/system/{{ ups_systemd_service_name }}.service" - content: | - [Unit] - Description=UPS Heartbeat Monitor - After=network.target nut-monitor.service - - [Service] - Type=oneshot - ExecStart={{ ups_monitoring_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target + src: templates/ups-heartbeat.service.j2 owner: root group: root mode: '0644' - name: Create systemd timer for UPS heartbeat - copy: + template: dest: "/etc/systemd/system/{{ ups_systemd_service_name }}.timer" - content: | - [Unit] - Description=Run UPS Heartbeat Monitor every {{ ups_heartbeat_interval_seconds }} seconds - Requires={{ ups_systemd_service_name }}.service - - [Timer] - OnBootSec=1min - OnUnitActiveSec={{ ups_heartbeat_interval_seconds }}sec - Persistent=true - - [Install] - WantedBy=timers.target + src: templates/ups-heartbeat.timer.j2 owner: root group: root mode: '0644' @@ -561,22 +338,12 @@ - " Off Delay: {{ ups_offdelay }}s (time after shutdown before UPS cuts power)" - " On Delay: {{ ups_ondelay }}s (time after mains returns before UPS restores power)" - "" - - "Uptime Kuma Monitoring:" - - " Monitor Name: {{ monitor_friendly_name }}" - - " Monitor Group: {{ uptime_kuma_monitor_group }}" - - " Push URL: {{ uptime_kuma_ups_push_url }}" - - " Heartbeat Interval: {{ ups_heartbeat_interval_seconds }}s" - - " Timeout: {{ ups_heartbeat_timeout_seconds }}s" + - "Health reporting:" + - " Check interval: {{ ups_heartbeat_interval_seconds }}s" + - " Push URL: {{ healthcheck_push_url | default('', true) | ternary('set', 'not set - exit code only') }}" - "" - "Scripts and Services:" - " Script: {{ ups_monitoring_script_path }}" - " Log: {{ ups_log_file }}" - " Service: {{ ups_systemd_service_name }}.service" - " Timer: {{ ups_systemd_service_name }}.timer" - - - name: Clean up temporary Uptime Kuma setup script - file: - path: /tmp/setup_uptime_kuma_ups_monitor.py - state: absent - delegate_to: localhost - become: no diff --git a/ansible/infra/nodito/nodito_secrets.yml b/ansible/infra/nodito/nodito_secrets.yml deleted file mode 100644 index c34b1e8..0000000 --- a/ansible/infra/nodito/nodito_secrets.yml +++ /dev/null @@ -1,12 +0,0 @@ -$ANSIBLE_VAULT;1.1;AES256 -39386134336137333139343861353965353665626161303634323563386563356364373265616163 -3237623438636161323366313363356335613566303338630a663732316663333136653737663935 -34373466316538663962636134303530323837316563336136653636643430363566376363323666 -3531613636393331310a306633613030343138656237663061653438336562323135633732356563 -33396136656564353465636361613161326465303166373131343562363338303966666362643532 -33383561366539393930633766363533313363653733393263376634316362643863376635366638 -36396333633231616662323465323932656565396138313264613832346261616231333265636439 -61666334613662373839613833613663333436373365376534643662656335316536303739616437 -65333635613330353536353162646234323266316338643435653864666364313734386665303830 -34383961633734646134373866623330663038366130306265656466653562643764346162313333 -353664363335303433393330306332666336 diff --git a/ansible/infra/nodito/nodito_vars.yml b/ansible/infra/nodito/nodito_vars.yml deleted file mode 100644 index c0002f3..0000000 --- a/ansible/infra/nodito/nodito_vars.yml +++ /dev/null @@ -1,28 +0,0 @@ -# Nodito CPU Temperature Monitoring Configuration - -# Temperature Monitoring Configuration -temp_threshold_celsius: 80 -temp_check_interval_minutes: 1 - -# Script Configuration -monitoring_script_dir: /opt/nodito-monitoring -monitoring_script_path: "{{ monitoring_script_dir }}/cpu_temp_monitor.sh" -log_file: "{{ monitoring_script_dir }}/cpu_temp_monitor.log" - -# System Configuration -systemd_service_name: nodito-cpu-temp-monitor - -# ZFS Pool Configuration -zfs_pool_name: "proxmox-tank-1" -zfs_disk_1: "/dev/disk/by-id/ata-ST4000NT001-3M2101_WX11TN0Z" # First disk for RAID 1 mirror -zfs_disk_2: "/dev/disk/by-id/ata-ST4000NT001-3M2101_WX11TN2P" # Second disk for RAID 1 mirror -zfs_pool_mountpoint: "/var/lib/vz" - -# UPS Configuration (CyberPower CP900EPFCLCD via USB) -ups_name: cyberpower -ups_desc: "CyberPower CP900EPFCLCD" -ups_driver: usbhid-ups -ups_port: auto -ups_user: counterweight -ups_offdelay: 120 # Seconds after shutdown before UPS cuts outlet power -ups_ondelay: 30 # Seconds after mains returns before UPS restores outlet power diff --git a/ansible/infra/nodito/templates/nut.conf.j2 b/ansible/infra/nodito/templates/nut.conf.j2 new file mode 100644 index 0000000..1f8a72f --- /dev/null +++ b/ansible/infra/nodito/templates/nut.conf.j2 @@ -0,0 +1,2 @@ +# Managed by Ansible +MODE=standalone diff --git a/ansible/infra/nodito/templates/ups-heartbeat.service.j2 b/ansible/infra/nodito/templates/ups-heartbeat.service.j2 new file mode 100644 index 0000000..71484c1 --- /dev/null +++ b/ansible/infra/nodito/templates/ups-heartbeat.service.j2 @@ -0,0 +1,14 @@ +[Unit] +Description=UPS Heartbeat Monitor +After=network.target nut-monitor.service + +[Service] +Type=oneshot +ExecStart={{ ups_monitoring_script_path }} +User=root +Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 b/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 new file mode 100644 index 0000000..925998e --- /dev/null +++ b/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 @@ -0,0 +1,11 @@ +[Unit] +Description=Run UPS Heartbeat Monitor every {{ ups_heartbeat_interval_seconds }} seconds +Requires={{ ups_systemd_service_name }}.service + +[Timer] +OnBootSec=1min +OnUnitActiveSec={{ ups_heartbeat_interval_seconds }}sec +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/infra/nodito/templates/ups.conf.j2 b/ansible/infra/nodito/templates/ups.conf.j2 new file mode 100644 index 0000000..0fa0a60 --- /dev/null +++ b/ansible/infra/nodito/templates/ups.conf.j2 @@ -0,0 +1,9 @@ +# Managed by Ansible +maxretry = 3 + +[{{ ups_name }}] + driver = {{ ups_driver }} + port = {{ ups_port }} + desc = "{{ ups_desc }}" + offdelay = {{ ups_offdelay }} + ondelay = {{ ups_ondelay }} diff --git a/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 b/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 new file mode 100644 index 0000000..c8d5d0d --- /dev/null +++ b/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 @@ -0,0 +1,68 @@ +#!/bin/bash + +# UPS heartbeat check - managed by Ansible (infra/nodito/34_nut_ups_setup_playbook.yml) +# +# The exit code is the answer and systemd keeps it: +# systemctl is-failed {{ ups_systemd_service_name }}.service +# Reporting anywhere else is optional and generic. + +LOG_FILE="{{ ups_log_file }}" +UPS_NAME="{{ ups_name }}" +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" + +log_message() { + echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" +} + +report() { + local status="$1" + local message="$2" + + # No push URL is normal, not an error: the exit code below is still a + # complete answer for anything reading unit state. + [ -n "$PUSH_URL" ] || return 0 + + local encoded_message + encoded_message=$(printf '%s\n' "$message" | sed 's/%/%25/g; s/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') + + local response http_code + response=$(curl -s --max-time 10 --retry 2 -w "\n%{http_code}" "${PUSH_URL}?status=${status}&msg=${encoded_message}&ping=" 2>&1) + http_code=$(echo "$response" | tail -n1) + + if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then + log_message "Reported ${status}: $message (HTTP $http_code)" + return 0 + else + log_message "ERROR: Failed to report ${status} (HTTP $http_code)" + return 1 + fi +} + +main() { + local status charge runtime load + + status=$(upsc ${UPS_NAME}@localhost ups.status 2>/dev/null) + + if [ -z "$status" ]; then + log_message "ERROR: Cannot communicate with UPS" + report "down" "cannot communicate with UPS ${UPS_NAME}" + exit 1 + fi + + charge=$(upsc ${UPS_NAME}@localhost battery.charge 2>/dev/null) + runtime=$(upsc ${UPS_NAME}@localhost battery.runtime 2>/dev/null) + load=$(upsc ${UPS_NAME}@localhost ups.load 2>/dev/null) + + if [[ "$status" == *"OL"* ]]; then + local message="UPS on mains (charge=${charge}% runtime=${runtime}s load=${load}%)" + log_message "$message" + report "up" "$message" + exit 0 + else + log_message "UPS not on mains power (status=$status)" + report "down" "UPS not on mains (status=${status} charge=${charge}%)" + exit 1 + fi +} + +main diff --git a/ansible/infra/nodito/templates/upsd.conf.j2 b/ansible/infra/nodito/templates/upsd.conf.j2 new file mode 100644 index 0000000..baddf21 --- /dev/null +++ b/ansible/infra/nodito/templates/upsd.conf.j2 @@ -0,0 +1,2 @@ +# Managed by Ansible +LISTEN 127.0.0.1 3493 diff --git a/ansible/infra/nodito/templates/upsd.users.j2 b/ansible/infra/nodito/templates/upsd.users.j2 new file mode 100644 index 0000000..8d43784 --- /dev/null +++ b/ansible/infra/nodito/templates/upsd.users.j2 @@ -0,0 +1,4 @@ +# Managed by Ansible +[{{ ups_user }}] + password = {{ ups_password }} + upsmon master diff --git a/ansible/infra/nodito/templates/upsmon.conf.j2 b/ansible/infra/nodito/templates/upsmon.conf.j2 new file mode 100644 index 0000000..dc301e1 --- /dev/null +++ b/ansible/infra/nodito/templates/upsmon.conf.j2 @@ -0,0 +1,34 @@ +# Managed by Ansible +MONITOR {{ ups_name }}@localhost 1 {{ ups_user }} {{ ups_password }} master + +MINSUPPLIES 1 +SHUTDOWNCMD "/sbin/shutdown -h +0" +POLLFREQ 5 +POLLFREQALERT 5 +HOSTSYNC 15 +DEADTIME 15 +POWERDOWNFLAG "/etc/killpower" +OFFDURATION 30 +RBWARNTIME 43200 +NOCOMMWARNTIME 300 +FINALDELAY 5 + +# Notifications +NOTIFYMSG ONLINE "UPS %s on line power" +NOTIFYMSG ONBATT "UPS %s on battery" +NOTIFYMSG LOWBATT "UPS %s battery is low" +NOTIFYMSG FSD "UPS %s: forced shutdown in progress" +NOTIFYMSG COMMOK "Communications with UPS %s established" +NOTIFYMSG COMMBAD "Communications with UPS %s lost" +NOTIFYMSG SHUTDOWN "Auto logout and shutdown proceeding" +NOTIFYMSG REPLBATT "UPS %s battery needs replacing" + +# Log all events to syslog +NOTIFYFLAG ONLINE SYSLOG +NOTIFYFLAG ONBATT SYSLOG +NOTIFYFLAG LOWBATT SYSLOG +NOTIFYFLAG FSD SYSLOG +NOTIFYFLAG COMMOK SYSLOG +NOTIFYFLAG COMMBAD SYSLOG +NOTIFYFLAG SHUTDOWN SYSLOG +NOTIFYFLAG REPLBATT SYSLOG diff --git a/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 b/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 new file mode 100644 index 0000000..104d974 --- /dev/null +++ b/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 @@ -0,0 +1,14 @@ +[Unit] +Description=ZFS Pool Health Monitor +After=zfs.target network.target + +[Service] +Type=oneshot +ExecStart={{ zfs_monitoring_script_path }} +User=root +Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 b/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 new file mode 100644 index 0000000..7878ee3 --- /dev/null +++ b/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 @@ -0,0 +1,11 @@ +[Unit] +Description=Run ZFS Pool Health Monitor daily +Requires={{ zfs_systemd_health_service_name }}.service + +[Timer] +OnBootSec=5min +OnUnitActiveSec={{ zfs_check_interval_seconds }}sec +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/infra/nodito/templates/zfs-monthly-scrub.service.j2 b/ansible/infra/nodito/templates/zfs-monthly-scrub.service.j2 new file mode 100644 index 0000000..04b3514 --- /dev/null +++ b/ansible/infra/nodito/templates/zfs-monthly-scrub.service.j2 @@ -0,0 +1,13 @@ +[Unit] +Description=ZFS Monthly Scrub for {{ zfs_pool_name }} +After=zfs.target + +[Service] +Type=oneshot +ExecStart=/sbin/zpool scrub {{ zfs_pool_name }} +User=root +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/infra/nodito/templates/zfs-monthly-scrub.timer.j2 b/ansible/infra/nodito/templates/zfs-monthly-scrub.timer.j2 new file mode 100644 index 0000000..36ad3d7 --- /dev/null +++ b/ansible/infra/nodito/templates/zfs-monthly-scrub.timer.j2 @@ -0,0 +1,10 @@ +[Unit] +Description=Run ZFS Scrub on last day of every month at 4:00 AM +Requires={{ zfs_systemd_scrub_service_name }}.service + +[Timer] +OnCalendar=*-*~01 04:00:00 +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 b/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 new file mode 100644 index 0000000..3ff4c41 --- /dev/null +++ b/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 @@ -0,0 +1,181 @@ +#!/bin/bash + +# ZFS pool health check - managed by Ansible (infra/nodito/32_zfs_pool_setup_playbook.yml) +# +# The exit code is the answer and systemd keeps it: +# systemctl is-failed {{ zfs_systemd_health_service_name }}.service +# Reporting anywhere else is optional and generic. + +LOG_FILE="{{ zfs_log_file }}" +POOL_NAME="{{ zfs_pool_name }}" +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +HOSTNAME=$(hostname) + +# Function to log messages +log_message() { + echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" +} + +# Function to check pool health using JSON output +check_pool_health() { + local pool="$1" + local issues_found=0 + + # Get pool status as JSON + local pool_json + pool_json=$(zpool status -j "$pool" 2>&1) + + if [ $? -ne 0 ]; then + log_message "ERROR: Failed to get pool status for $pool" + log_message " -> $pool_json" + return 1 + fi + + # Check 1: Pool state must be ONLINE + local pool_state + pool_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].state') + + if [ "$pool_state" != "ONLINE" ]; then + log_message "ISSUE: Pool state is $pool_state (expected ONLINE)" + issues_found=1 + else + log_message "OK: Pool state is ONLINE" + fi + + # Check 2: Check all vdevs and devices for non-ONLINE states + local bad_states + bad_states=$(echo "$pool_json" | jq -r --arg pool "$pool" ' + .pools[$pool].vdevs[] | + .. | objects | + select(.state? and .state != "ONLINE") | + "\(.name // "unknown"): \(.state)" + ' 2>/dev/null) + + if [ -n "$bad_states" ]; then + log_message "ISSUE: Found devices not in ONLINE state:" + echo "$bad_states" | while read -r line; do + log_message " -> $line" + done + issues_found=1 + else + log_message "OK: All devices are ONLINE" + fi + + # Check 3: Check for resilvering in progress + local scan_function scan_state + scan_function=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.function // "NONE"') + scan_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.state // "NONE"') + + if [ "$scan_function" = "RESILVER" ] && [ "$scan_state" = "SCANNING" ]; then + local resilver_progress + resilver_progress=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.issued // "unknown"') + log_message "ISSUE: Pool is currently resilvering (disk reconstruction in progress) - ${resilver_progress} processed" + issues_found=1 + fi + + # Check 4: Check for read/write/checksum errors on all devices + # Note: ZFS JSON output has error counts as strings, so convert to numbers for comparison + local devices_with_errors + devices_with_errors=$(echo "$pool_json" | jq -r --arg pool "$pool" ' + .pools[$pool].vdevs[] | + .. | objects | + select(.name? and ((.read_errors // "0" | tonumber) > 0 or (.write_errors // "0" | tonumber) > 0 or (.checksum_errors // "0" | tonumber) > 0)) | + "\(.name): read=\(.read_errors // 0) write=\(.write_errors // 0) cksum=\(.checksum_errors // 0)" + ' 2>/dev/null) + + if [ -n "$devices_with_errors" ]; then + log_message "ISSUE: Found devices with I/O errors:" + echo "$devices_with_errors" | while read -r line; do + log_message " -> $line" + done + issues_found=1 + else + log_message "OK: No read/write/checksum errors detected" + fi + + # Check 5: Check for scan errors (from last scrub/resilver) + local scan_errors + scan_errors=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.errors // "0"') + + if [ "$scan_errors" != "0" ] && [ "$scan_errors" != "null" ] && [ -n "$scan_errors" ]; then + log_message "ISSUE: Last scan reported $scan_errors errors" + issues_found=1 + else + log_message "OK: No scan errors" + fi + + return $issues_found +} + +# Function to get last scrub info for status message +get_scrub_info() { + local pool="$1" + local pool_json + pool_json=$(zpool status -j "$pool" 2>/dev/null) + + local scan_func scan_state scan_start + scan_func=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.function // "NONE"') + scan_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.state // "NONE"') + scan_start=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.start_time // ""') + + if [ "$scan_func" = "SCRUB" ] && [ "$scan_state" = "SCANNING" ]; then + echo "scrub in progress (started $scan_start)" + elif [ "$scan_func" = "SCRUB" ] && [ -n "$scan_start" ]; then + echo "last scrub: $scan_start" + else + echo "no scrub history" + fi +} + +# Optional reporting to whatever is watching. No push URL is normal, not an +# error: the script's exit code is still a complete answer for anything reading +# unit state. +report() { + local status="$1" + local message="$2" + + [ -n "$PUSH_URL" ] || return 0 + + log_message "Reporting ${status}: $message" + + # URL encode the message + local encoded_message + encoded_message=$(printf '%s\n' "$message" | sed 's/%/%25/g; s/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') + + local response http_code + response=$(curl -s --max-time 10 --retry 2 -w "\n%{http_code}" "${PUSH_URL}?status=${status}&msg=${encoded_message}&ping=" 2>&1) + http_code=$(echo "$response" | tail -n1) + + if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then + log_message "Report sent successfully (HTTP $http_code)" + return 0 + else + log_message "ERROR: Failed to report (HTTP $http_code)" + return 1 + fi +} + +# Main health check logic +main() { + log_message "==========================================" + log_message "Starting ZFS health check for pool: $POOL_NAME on $HOSTNAME" + + # Run all health checks + if check_pool_health "$POOL_NAME"; then + local scrub_info + scrub_info=$(get_scrub_info "$POOL_NAME") + + local message="Pool $POOL_NAME healthy ($scrub_info)" + report "up" "$message" + + log_message "Health check completed: ALL OK" + exit 0 + else + log_message "Health check completed: ISSUES DETECTED" + report "down" "Pool $POOL_NAME unhealthy - see $LOG_FILE" + exit 1 + fi +} + +# Run main function +main diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml index f7fa9d5..567c768 100644 --- a/ansible/infra_secrets.yml +++ b/ansible/infra_secrets.yml @@ -1,191 +1,194 @@ $ANSIBLE_VAULT;1.1;AES256 -35396230336432656135363562623433303661376462666131373963633839386238636162613065 -6563613965386631663235303662636261323635393031620a633032313564306337666631646130 -32623664343966653866643438613965303936616234313036303061336139633535653333636439 -3432666530356639640a353631666335653863326232323061663136646332316535656437336235 -31623433376561633931626537333162323331336338326332633630343530656364323536303336 -62383763643033383138333365336130356264626661663732313561323465366437323164393733 -38333532376239373861613566333466316234623564316434383434393064323330663438393139 -33323263323038376135623439376132663637653634613465653865623163366365633861336161 -31323337656463396464303963353236383638646438646531393934656137303063343238363331 -61656262313037306531343437306361646637373864356662386333346134626637346363383931 -39386565643335633863396662363264633432646336383235303466616262663533303965623831 -39333563353431636438646639376663666439343161396163656463366333373362313466626465 -36376665653561636135383531336236643461303261393534373331633934346438323035663637 -63343634376532386532633435616661656336313966363566323937653533613665363461306638 -65393366346662356264353433386534326537363436313836373735616663303131643063313064 -61396634623262616661333365313466636332653261666539653135656166383933373436333032 -64393634396531656536323436373061626531663436313630373865323838376639386430376331 -64306138653931383335366130386334386236363062363631333465303635303531386261383662 -65343236663334356631306439613263653162626536303163343463353431363766643636653166 -33613439643634323331323165386337316463383232646232616439306565353866646635353163 -30343965316234666164633863333632376561383638323639353361313263616561613335316135 -62326466623134313236316430303861356335346633333337373639343766393639353431653061 -39363266663732393434326539353032666161623265633464626237613364396239626163343466 -36303364326638346230326632333636643866663937373864326161663765363535333761373138 -36333034623532616261613336326134626334313063353165303035383166303733393939303231 -30643731353963303839623265336663663436643839666366656661313232623036643232646465 -33666539343565363464373530323033353330343139666239623663646166373733363736633335 -61316134646234653638373639356665396663376366366366383831643938633933656261316431 -65346334653864663863333239383131326665316234363536356234373266653064353562623934 -61656632343235376235323466316562336337326637386534393436643664396634326261643736 -65376264376430333461353336656539636133343933323966323431376638653161316535323862 -65383661373763336638333461303433373365633161623333326535343932366537633362663431 -38663630336139313831656634653736633162353038333438643533633962633166633330333939 -38636266316162303039303737666265633230373364316362393562383532393539376133633137 -31366562666337313866343662363732333932643866646164633964323435393139633766376165 -39616262383366376164383334353633313431393830663962356165633734306337666330386235 -61303664303231346133353130666132623533343938613962363334663466313833633430393337 -38393631326534316363333966366138313836303830393763376633356234353335353366323962 -66366362386338353830346232303032666265343138313836633539333763626132653065656136 -35653634616638346138613464393464636239633762346236613030663766363839373134323730 -34373936323838633035373764373461636533333033646439386665653539653066616135333664 -38643964613966623833653233343464333464633134353431396536656432353738333761363932 -35646436323565656463613433353563373936316130373565636638333237373038613435633765 -63393538613531323063383434643739383134643863313739343863396437333461343936653666 -30656161656638316534633264626338666431333035633432346666636235343438303936363738 -61633838303238346336316365623062356464393463386237366664656134626231633831313964 -39393663626337366264663462663532636365623834386337333538633833633337646138373331 -30633639313633643233386664643033386136306265393233313236343166373838393165316234 -36366339373730343933343661353462656165643361626335663661343666313762613534313934 -37333764363565316163656334613631633938646564323535336164396438626261346365306233 -39383338303063343839633161623631343165336466383838633935323466613831313137313732 -34303961306638363233346436363432363366666162326438623962653138643661336233303464 -38626464326331623735313335356637376633373065366235303834343661633866303965646236 -34373066323464396363663562363239343261643930306632356566656364373231323039316339 -63656630356566343131363935396630313632383137323263363936383862623534623532636563 -61616236346237373031343034646231663833613132356362303532633062356638383266303438 -38303937376166663333653766633165306631366333323831333534613438363063323839633263 -32363132626335656434303130386330626530346463616136393137333863393965346639366261 -35313166356437393163663866626437383633323031373931643439623866353162646566303739 -61373530353264663436643937323431393762623933383233383833663235656434356639353664 -31336230666464396533633735313632663064333436306235303334653064303237303737636132 -62663461383761303863663538626262313038343739656262306564653034353031383234666534 -39643765393634323937656263626463373865643561376235663131306431306564363463346266 -61313835623933336436636131316130346337623835366366343735613562616137643466616631 -65386234656235383639353432333663626538303564303065383064636638333466366663346535 -62373034353430393733656163303933393332373030646630646638373861393463663434303264 -31626439663239373732336330356131613862343836363232326663643436643264323531366530 -62346532633565373961666464626431386431623038313464396361643934646332373735383339 -31343333623739623437356431623439333134333932363734663734636239623930663234656562 -63633561646362643738306632343732636166303636383939626135356536356134656262356636 -36346338643433393462303162343737643330656230643339386536393738353161303831663338 -65346262346136363565333261656164356664323533626537363562313935653766623535313361 -64343637376530333833343935633338303738376664323363643131383938333636636365333134 -33306363396233613164363461616238616364633461326666343366343534663166646439316437 -30396236333236393466323639663930636333663261373563346364613366616336643335646139 -39613632663131323432663938373537663335643934616333656662346432353766326631643034 -35316436653765386235366464373565386463333137343039653732386235633162373834383730 -66323139363137333835613132383364333437643332613930313463343539393736333336363039 -37333136373933353932353637663238363336613133626237393436353762316463633437626130 -33373762643736643465353030653538653262363531643935363836396631653431663231373636 -33656637336535646538383364393165313461616565316330653362643037333635386433626566 -64346166393130636661313731343436383139663831323031346663613761363637353537373632 -37316135333930386632613039356465353761383433333465333033646433373235323932343333 -39643532306133383665643838303366613631386439303862613963656434616461316236383233 -39356665383463356435646630313932626263643037623038613662383536663831316537613038 -63336262346361326664333432386465376634383534373562643238323462633334626366343761 -65303264316332656663616161373335313863396363643462656661613766306632636138623234 -35623666313130633836366161623862333161376464326166616539633835613861343038363736 -62646661393837353266663962313836326162306665323230353764633965336634373934366230 -31633364373164653163626162353934333933303332623564653362396665303961393039343565 -31386632363536373561643838613465633462346539343266613562393236626166383332653335 -38633432333535633532363239653332386565626232653963653930623735633632356264653463 -38376238393961653834616134376462363862633462333265663932653338656362363461613533 -33666137333639616237306263366264623236343732393134643266353331323634343462613438 -64616464316535363766666331373638313566643132316131336337393131346334393632633366 -32633964616461366435643234326234366434363131643437346131396139323331626237396365 -61666437643335396432373932653066663862623035646365343561616630663835663161303537 -63653134303134633861393036653464613265646135353061343663623964333436663834396336 -30313063336131626534396438393731616233333365393965383863323461303561356332363763 -66313863373164663734663462393431653864623764656536366266643231313765363430393134 -33306565646532306639346635323461313736663533303266643433306634326565383232643939 -65623738643739353363653266353534323332363037373834366135643537313739353263636638 -32653234353538386461323632613536366434613262653538333661313333363539393664363363 -64656461356661613161383562653465353866623961653232386265613864343562356639336339 -30653131623130633332343931343537326636356134303336306139333736303763303533323735 -37636131323530336264356264383263343038316264646634356338353761383033646631396364 -62303330386362623436343161353731623435356331373666613335653761653366356433373564 -34623062613439376432303765333763346662393866326337613061616138396430326133376139 -34663337393938343563363266386536616334623964633132303361393137623936623434393562 -37383038666166643537316439323333373431383661643863653034323330366565306630643161 -34643931373966306539623361323261633061363837643538373830343730646534366435636335 -35303862396563333835636431636433393933363533643636343265613638343933373438323563 -64613937666330343038343437616131303630323961306135386264363361383163393836353963 -35303064333065373533366535366361393436393535343965333865646137366166376530346436 -62386432316264663532333465623265313734333337323038613066663533356638666662323334 -35353165633564666130353962386439373863353535373236653338646633356531373230373262 -38666637663661396538333031653966616365353961653530613465623362393932643064323035 -39636534663464363064323862336633663038393834663333353235353562363934373739656664 -65653362346632613862393133316362366665646130326232313138323237646664383565366466 -34666231346433656235363362313434313438396361383231656135306330303035393530393530 -61643331626464646266326239336565613562353462323834393731326339383461313164633930 -31653633373633333539373962383866663739653135336463623837643165323831366562326536 -63613233383533636338646466363962313933393035356339316162616436306138383262633430 -37356366636662363231366532393763633564366461383162613539633734626563303864316233 -61653339393465626631373664306165353964373966623733386532636233346535326132366261 -34663036343665316234303439313064643939663833386633366264653163353234356237343837 -39613134623138616264613732383065623434303336393866633836306432663562393730343833 -35633139343130343236653931346266613665616230633032626665663031313262306530396364 -30663863326632633261346363633838653339393239376462386631303865636364616238646663 -37663033363433333531353332383866383065363337643265636264353562663635393766303265 -39323832663261653831616561333639373838316138356333646439356535333531626530633238 -33623039666433633436363365666431623039353065386564653139336138346136333238663866 -63356237373130343163303936643562373037613361663834386562353337663061313934643430 -32646137636139646263336439663730393737613061663933386165343034386161343333623834 -33323163626266363136376536616362636631353264363736663366636266653238656362633934 -30373664643933613637346363343964346463363036313961633961303832323834376337666630 -38363862613136396663626436303237396435333464303938303562333034613532373134353233 -38653436336461386139303864393338656661653137646137336332616435656638393231323838 -62663430663330653963313939663362656430303761656633343837306265366332393030363930 -30346439393837363333636136353235353463333563656332633236663033633835363531633261 -61643632396666663934346139373235346261626339386431333064333632333034333031666439 -66336561303536656462343363653136616562306362333662643938616437663938393933633463 -38636339356565663532663235326266386332353764656639363165336634653639336463616362 -66393131656631646133376461343438386162306430613561323630613865626235303438346233 -33373839323732666437363333346336663963323066353561396338383633626138323337383032 -35663132623364666537343036303136336333633864323834633265343936366631363865333962 -63306433333035363565303866333234363038663934366361656539333963373932306535313638 -61626661336330393265353064616263396533316361616534343566613833363061346230333036 -62626130383630306361646531666436613461333865343131626465323361373163336539343163 -39313036373236646162333563393561646663326631613333646238626463396364343061383233 -31373365396535326131616237323636623364633263386564613763623636623138386430376364 -39326637386237616332363531393534636262393336366335306665636466346134633365386665 -33646266386635636361383937616630376530643237396366666231313930373939313566383938 -64646436613039633835363966343838656461313834643438356166356566663861663833666137 -32666437626637323766303962303164393565626362656430353439383839313166303966366433 -64383730336566633336303966376366343332633366373662353337663965616333363033393765 -36306536316135303164306164313239333938646630633864336237346161303963376531373464 -34373465366231336365323030376235306662313433616433633962363564326262363631666136 -65613538656534366136303030303339383136323137633934633662646537353235646534313332 -31366365613533653339646131653239343663346239613934393666373536386635656534376165 -64613638643338613738353263633664636164353265616536626132346663636133393137616433 -33663635343063373465303936313731333765396137393235643566656332666634636462386461 -37323065373733333433316463663635653630656131313235386663306134373364393765356239 -38393964346361356135626331653161633536333463666463653936656531313463666632393061 -36383166633935343063663763663536326466646663383763333161613165393665383030396262 -34666631316262616131346233613035343236613838333036663661386464346136383737323036 -66366632313130646430616563653731346362343339363235616335393964633335326633376135 -34386666353864303161346263663265393731326332303231323934383838343934376339373139 -34663666363335663165636237303330306338393136653339633733663364623461303134303839 -62366337396363613737623138666565613530643861653830616332656435373233656266353461 -66616636393864643338616164316131623464613466396562386238663364623630303338643837 -64633433666139333864313862663832306230663339643933663530383939396230393032323861 -30376464336566303833306664613633333335326562396462326437343437323032363065363237 -61646238363436383134343434353534326537666431333430343065633035353437616366383039 -36616362353536396436623835656462353231653838336666636335666333653130633330353232 -64643733393034343139373733623365333961343139353532313230613134346365303061616663 -36636465353234316361376261303834333532623336646335306636353965613361366430643436 -33306135633538653034643361323235353566636661363436326133326238633964646263376166 -32333862623239333661393930346638663461633762653332386631616666366333333863336239 -63356265663437343037396464396365323336313038336636636562343262396466353333313765 -33396636396562323338383561616366323832616566323464653035366432653833633831633234 -30636237616332383434623338616639633732643134383637343162353234316639333733613033 -31306362623163373631633965623631643537393662626538663966373765383530343064313165 -66303337656537343539323565663636333139383033393462326331643234616138323236663630 -62653163393733613838633739393439343464653738616161663330633236626234313034346462 -31633539303139306361613839366234643535383961346130363433633033373434633565613063 -35323839386463383438663131303665386130653332366463363330363530653533313661363361 -63313539653239303861356638303236396138326131626564666337626332383462633732373465 -36353966383036663733333336643633666633346233363732633837656334316331313032323064 -3739636230343038353664353163363939363465326433613561 +36353131353730316436313965366263666236396164323163376431623232313961323863633035 +6338333965643036666538303864376463623434326564660a396239633264353034653963366466 +66353130376366663730626331346235373935306434303462663763613763663131613335633430 +6335633535333166330a366232306230303033663931333734353636656132383034616336653235 +35636234306134376666666538386561663561626262396135313632393831363665646130643261 +65663137613530393162333838313962393934396135613064333564303666353634396266316135 +65346363663232626261643933353239663635343137646431623561653963363131303033623764 +34616339343835353765363962396665636335623066303062626463343733643137626265643366 +34653139616163666131333330376132343363306132653163343934353034353932383066373462 +31316664383938633361373839353762393838386636393564633430346334323734383430386663 +39313938383830373062336335303135396334323330316537303965653331323130323866663866 +38323462643633396531386632613961343835313931613933373963326465333836323935613436 +63346562313239623162333038663535656633323037363532373536616161303833626539366634 +35363462663539353366623732626237326465663436356630626335623334356465646439376461 +63653230343333613932646431383238373039653461306365326234303365373234613038343338 +30353937353734633631313765346662616138383462353864656264653534653239383936616437 +34313531383438613465626437313130326562316430366234326230303133356631363534396366 +64313732303837666537383161373865653162303034376533636264656564316530613738353435 +38353031616538346438336563336134643261366331623562663534643966306164306132646332 +31336539663131646232353633326637306637353239616331356337353761313763643933303838 +61636333663137623233323561306665666666363731383438373664363538613762393130646261 +66346335373437326536356330613962386662383031366536373635353438316664613362633966 +35393331613634653462323739343466313735376463643035326336396435656364356134626535 +32623036383737623631623332666264653964383939316465353765373132383230336334353463 +38303136356630653561333363306136363230383238666663373130663763336532323631663637 +32633263633835326662366237613639313431366433363730666666333961323036363533613236 +39343963613239333731336461393937633463303530643333626662316563313165613564386663 +34373636383231636461653862356332353161366331633036383837353434373865363139386235 +30323130343534386531666333343765373265626530373335633766353562356562326239336564 +62313033376330383563633138623838633132343130343262663030316435393835313762313564 +30393366393161663265303439383937333739396661393862356132343537616333623035653566 +64316333323733336135383637333739393931333461653731383863616138613234316161623765 +30343964623939663339633139383866336163613933363231303837646565626262303330363338 +64353931623366326666386637363866383761666232663866666438313535393434373933653139 +64636236306632613930363935353736393065353634653235303534316463653438646162663866 +63663839346162396262366563613131356161333137623966653039623436373466663163313938 +64636362383532303136316163306236333134346164376362363730363030643436356631306639 +31376131653162343338613830376466356363356238343436366232393062336433623238323864 +32376133346264383736396166343539363361623864613364613030623936396634616431386263 +35323632353334643466383933356236383432323162386532613535343932353531373065633732 +36626361313436613036363662376337613237353532366163656236323937316437613633383535 +39306334656532383063306332636538666462353836633366353932623962396131663532343364 +64393035303265336638343733366236373466343832333038633138306535303034386134653032 +62633532353462613664636333373461356434666130376332373762653966386238346463333939 +32386233326264666234353139643837636663633362626163356633396465643030383639663538 +31626331393161653038613438613735643364326565633362386331646231333539626133633739 +31336236663863376166636632363965343765666335633464613937653738313230356438313436 +39373936663363626337626164383237626164396337613430373035373339616266666264346164 +66386238373664386138636536616266353466666363383131616532323430663761633139626164 +33323936356565303332383463656533633935373564353461626139343265336334363433333138 +64363534363836666338366339636435366334383339663463376164613866333839366264666237 +35306231656536613138353663626166303366373761323866666436356438653764353431383430 +33323766643431303337323265646631373434393436333335666366306138356337343163336362 +39323131653734366561613864383266386131626164363361656534386363613565336532666162 +38333031656330306635313032343861653037323037616564323537666261316438303631616530 +35303238366338376363383764633865326530643736306634363436643330376432396535363932 +39646666333234326261373835646336663138366237346562396238326530356533326363373664 +38303634383036323232363934626562663864646239653066326565623635653535623137663065 +65376633356136333938623763396138376630636636376434333463313833386435633365623036 +30373937353433396264643231323832373963396437363562623639353764623063363733366165 +38613833353231333836373637626133323965316665313837653335386534326434363034653630 +38323335666466306665663262343737636564393934373338623130636362383564376434626533 +35623761396465333730626563633734373131373132343262653333343439623433313964623463 +33613639616636393563663331326535346131663832313861353434313965393164643538643437 +36363130363934623333666330353236386530626536616333303131396265386263363462356364 +38393864623461643831333731656566636532643338306361666335316265396430376432393730 +34633638636466313232376363306333356564393038323130313634386162333639366162623437 +66306234346430346537343338343064623131366337356262653130616464646336353136336234 +34623063336332666138303961333332326632326137366664306263666538616361316363323163 +38613431353762653139636365353031343465366138396161366538393833666236656264623132 +63313630633231613438613865613333343066376261326333623864636430643866656132353538 +65653731626464323930383636663364313437313531646462623163646630646664356335303764 +61313462623234303961356363386464333061306239643263333135653538313338383134343236 +63396461346438373830643031336165376563636530353836626436613561313536623130613663 +63383939663535326236313736613630313061323837346339353834313434393437383237633534 +65363334643238373864636530386434636565623131366433313562623933366633396565646565 +33626433373834633638656365363261643866343961663566333761306138383865316536303664 +35653463633336616438373732303938393032663961653262333134353762333433373039636638 +33366466336133383061373866323231663038373831386139613535376539393165363034303537 +35346638613137616666623362396564346464376133316262646339333936363961316161663332 +38303239356334633137316333363439303935346230613032643564353734643863636264316136 +39663131376638653138623838323739343064623430386166643962633363373335316239663339 +37633934393664346433316633353831373534653437623065363134626661323732656664633639 +32616539323931626130353836303533363232353066343666366263663565623862333062323237 +31393935313133333837626632346461356134376339633239396438336333376534353063633937 +61653230653666623935346134666130373934643438323464346239336430373635396165323738 +36613432356239323039386364643535346261373739363463373839373063346565616466663239 +33353165353631343730636531363565383363623839346434383737313136646330323461633665 +32376631663637353534636262666532643636366237383539346235383832313933336432363734 +35653866643136383934613261633439343831363234616438333530336437326562366536383834 +33343330316337323064386464656436333430643061663665306534386235336437643965393165 +61373038323539316161363731306334633833376532626537643164373638303438326637376630 +35343333343364326538333862643265336166396466373362326162303430383434626535336163 +31666662653261343631366165656131376462333063626134613466636637383936303933346362 +32386162323230666563353838613033393936386535646130613861356563633566393430313661 +37666561643361653630353262653234636232303534393661343834366562353864343638313134 +36656463653166393432646263626530316361643761646265613534643563306161333539633361 +33613830356238623962623739623437646635356334386637633035373262313539386662353732 +35666534663965333836396664316432633137363761623838356430316566383131363530316462 +62636235396136363538653836633133663465386139636339353664666634343238663861613437 +62656435363832623461363662643262323265353066376538306163376438303838373961303239 +63623066656532353933396534623637356362356231336361393534393465633565653132656130 +37363863666136643461383033653936383935333131343565643664386664326463663466633361 +36313638616533623431316361386539383434653866376363383630323632636237373666393561 +31313135303339633763303762653939386663393439646135656537303331366236396230303063 +32336336366533323936633166623434623365633163643461653966383563653533363335666638 +61313764656437613138663038386336336664396539373930373933326234636662653833656435 +61343962373935653238326265663164363561616435363134313634666331636431633133356563 +37326133363762383266353730343537373934653634366336316635653063313461303237633164 +61333231353135343666343461306161626432636331653962613334376439613265653437663333 +34623638313561356464356534343635653265633531323736386335656635303263636663323930 +39373462663438326366363433343735343234383735643363326339626135313564646432656330 +38653836653938383932643032616534363035366530336663653235366230353332636531653138 +63623537353165623037383361643937623466356464313931666430666632633866363232623230 +32353766383432613233313331356432326430383132616263656361613334383936643066326133 +30643935653662376433353039323836613239336265376663616336663264343331373236373237 +64326462626566386563623235633664653665356461316362663462343665303233313035366435 +32623465656138303161396233623661333238306334646663353263383437653461383366366466 +39303061363530373339623435613765313637616634313731666631653161623439323734383735 +65383666366330623338653561353032373231396431353466313335303935386136613566336331 +62326539383631636263353234393866333031643737663064383130313066396461663466623431 +38616364303439393432663035623138353264656635393633646363366538633262306466336564 +35623733613564323233303034636264336464373566303065383438366338643538666638643466 +63343130613461316465626237343235616536613838623930636136333131616437656631376564 +34366535383961656237333030376139353237343636306165646161323732646663376231393832 +61396263393561356438326237633634333533646638383865336164656465383166373236643063 +63643263303132656336333339623539663765333039396364643462396161366562626362343162 +34363936323230373839373734343332663664373634313364383062353130363362623937656139 +32653237643061646531623030646335383463333136366264363133303666663261343631663762 +36346432386465393765353763353265633837623165303634646137646564356237653336666331 +38376463376130353863636334353633313361313239373264333134333232353765346666326665 +65623333396165353334636532303537366566666130343366383964396365333461646566333431 +37303032643731346361393963303061363837373562393866353962333231623763623236323161 +39383538663837326663636166643864393733653764353536303433653563316339663438353838 +37616135373336326536333932636538336465323130626266613930393266633164636439663532 +37633264356661353836346462626538323631613539636631396436646139666538383838323437 +62663561356430336235343439633135326661653031653063363030326132336462336164373232 +31363131326334613361356461353934656364346666663762306161356463386332366562336630 +39326236656335363630363564316134623435343538386462663161396332366639363033383235 +35306434646330636137323939623565663939643161616336386633633133363963383739633434 +32336433663132363239356266653461633033623232346135353032353265343861316638643265 +66303932643235333764323239653430346132666136663133383935613962353235313336306136 +38373938376238313034346237373035383861613930323936663831346538313937343538663737 +62616139663634333636333635636364643966303565656265653634653437633233396138323831 +64313465643233663663303130653133626638313162356133316139333030663530313232626437 +35613439326639393032393962653933376636663934333764393064346465633637613836333933 +33646336653533666631656132623035323963623134653432363133646339346363336263323135 +34356233373965333362326132383762626436333365363731396134376664363465623362316366 +31623361376161393836623732613034643831363838663733386561313961373632643639666635 +37633061303235306437666337306236316331616330396632356266636137313833346366653031 +64366236306139363861323237396232336665633331326139373461353237336432373366333266 +32326565376464383937356562633730383961306535666464383364656137633662333662626366 +30653736646333333963656566326431376361666465363532393765393764633562626232643836 +32626261373835356634633461303664653362346231343030343433376364643664383464636666 +35323738366332306337616563646231663963353135623133613636666363396530643239626566 +35383330663138333631386363643032383161626439633437343934366635656433386163613663 +30623738656535396134363830626338353864356330613131623832363330633064376531303965 +35313036393031383035633035636165313363333564633938306166666639636436353733303662 +66653264356432383166333630646533313736366130666331306537393262363538313030646134 +38306363663932313230346664303531636639656339323062333739303239333861616533666132 +30353332303233323837346234336633646163643137636166633330633464663935653838313161 +37393533646139313236393234313763353533663638363031393764633862363938643838353937 +36613564353366663434633839303036343665313933326531353831396139613330316632613637 +37666432363330326232366664656462313336323866316633396533313638373462386365663332 +63626136396661353739633263363038326630623037353831323930346431666263333431643562 +39373133636465373064613664323335353236326562343966616439646565383934613462623363 +38613962386236356666643038303435376266656165336263653365353537666362616638636639 +34326262303930303237636339666563613663373863666339663135326661663866346264613734 +34353565613832323132343730396535656264376233356162353265623739613332333261393331 +33616230333033613766643264396633343535376461633330633064613662336532613163373962 +61343065373630383838306631633031656566343765333365373932373234313733396165623639 +65313731346632333235303463393039656232653163336161633265376434623466373361333865 +66323664343766306566663335656138323537656563383835653263323039656166613330613733 +38323064306365656561313163306439653937623536616435616431616133643466336362666330 +64303832343739383430626339663532653363626263393237343234666266323933316566343937 +32373335373237653265303638663331363838313961343936616639646438343633326461626363 +30653964646639663836636437653861343064393332306638363864326439626335373663653335 +33653230626337633863356534393466633162323734623135393931363339383163353466633031 +39343435616137373166376538633330386232333666316566336234373933613534666430353336 +33626637346337623132313833343430613832303932323739326537666531356630356137653963 +30623262373032353934353938346263396338336265393937336539386530343062623761646632 +39653035393535356630626164393464663738616239613166613364396539343039306538613230 +38636261633932643965323966346632616264343262643933346163656436326330336564623265 +30646561303563323937393533663066393733353638323332663736353766336664343265613733 +66643631633864386262343232636465613962386538366163613734386635303837636164383465 +35373331363864393934633563623662326632386534323436663634613030646563643035386539 +37323262646365633739626539346638383039643433343837363561303530376434336131353965 +32336133393139626232313933636136646264613338643238646431316566333662353837636232 +32623639636432333134643962313430623533333237343464383135643361336236396533653761 +66383165363539363938653130633630643865636663313665353839646462613334383266663563 +62333035626539383739656565633861363863333563656339353735336330333633633861643130 +35363062306165623861316639336237373434353437626461383931656531306134313264383664 +37353966616136326264376135373532653630393335336665343463323466353162 From 954b683c716bb03c0e0225bdaad57285519fff4f Mon Sep 17 00:00:00 2001 From: counterweight Date: Sun, 13 Sep 2026 20:58:46 +0200 Subject: [PATCH 55/67] ansible: delete the duplicated vars files, move globals to group_vars/all Three files existed only as second copies of things group_vars/all already auto-loads, and 34 playbooks named them in vars_files: - which outranks group_vars, so the copies won. The day someone edited one and not the other, those plays would silently keep the stale value. infra_vars.yml was already drifting: group_vars/all/main.yml had grown age_backup_recipient and backup_pull_public_key that it lacked. infra_vars.yml - a strict subset of group_vars/all/main.yml infra_secrets.yml - decrypts byte-identical to group_vars/all/vault.yml infra_secrets.yml.example - documented Uptime Kuma credentials as the reason the file exists, which stopped being true Deleted, along with 62 vars_files entries across 34 playbooks (12 of which named ../../group_vars/all/main.yml directly - same defect, a vars_files entry duplicating an auto-loaded file at higher precedence than the file itself). Checked before touching anything: infra_secrets.yml was listed LAST in 10 plays, after services_config.yml, so removal would flip precedence if the two shared a key. They share none, and neither does services_config.yml with group_vars/all/main.yml, so the removal is provably inert. services_config.yml was the last one standing. It held four unrelated things: caddy_sites_dir - an identical copy of roles/caddy_site/defaults/. Deleted; the role default is now the only one. *.tailscale_hostname (x3) - a THIRD copy of each box's identity, which inventory.ini already holds as ansible_host. Deleted. Edge plays now read hostvars[''].ansible_host - verified an edge play resolves that with nothing loaded and the other host in no play. Three copies of one name is how bitcoin_rpc_host ended up labelled "knots_box" while pointing at fulcrum-box. subdomains, ntfy topic, - genuinely global: their readers span managed, headscale namespace monitoring, vpn_control and edge, so no single group covers them. Moved to group_vars/all/main.yml where they auto-load. The ntfy_topic and headscale_namespace indirection through service_settings collapses to the global name. the four cross-host ports - the only entries with a real justification. Left in place; they move in the next commit. Also dead, all Uptime Kuma residue or duplication: phoenixd_monitor_name, forgejo_runner healthcheck_timeout_seconds/retries, fulcrum_tailscale_hostname, and bitcoin_knots_version - the last being a v-prefixed copy of bitcoin_knots_version_short that nothing read, two hand-maintained copies of one version string. Corrected a false comment: services_config.yml claimed the uptime_kuma subdomain "no longer resolves to anything". It resolves to 164.92.239.72 and answers HTTP 302, and 11 playbooks still template it. Same wrong premise as PLAN_3. Verification: all 37 playbooks' --list-tasks output is byte-identical before and after. A probe resolving all 22 values services_config.yml used to supply returns 21 identical and one intended deletion (caddy_sites_dir, now role-only - confirmed the role still resolves it: "Ensure Caddy sites-enabled directory exists" comes back ok against the real path). memos check-diff identical before and after. Syntax passes on every playbook. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/main.yml | 43 ++++ .../01_user_and_access_setup_playbook.yml | 2 - .../02_firewall_and_fail2ban_playbook.yml | 2 - ansible/infra/410_disk_usage_alerts.yml | 5 +- ansible/infra/420_system_healthcheck.yml | 5 +- ansible/infra/430_cpu_temp_alerts.yml | 3 - ansible/infra/900_install_rsync.yml | 2 - ansible/infra/920_join_headscale_mesh.yml | 2 - .../nodito/30_proxmox_bootstrap_playbook.yml | 2 - .../31_proxmox_community_repos_playbook.yml | 2 - .../nodito/32_zfs_pool_setup_playbook.yml | 4 - .../33_proxmox_debian_cloud_template.yml | 2 - .../nodito/34_nut_ups_setup_playbook.yml | 4 - ansible/infra_secrets.yml | 194 ------------------ ansible/infra_secrets.yml.example | 38 ---- ansible/infra_vars.yml | 9 - ansible/roles/bitcoin_knots/defaults/main.yml | 6 +- ansible/roles/caddy_site/defaults/main.yml | 3 +- ansible/roles/datum_gateway/defaults/main.yml | 2 +- .../roles/forgejo_runner/defaults/main.yml | 2 - ansible/roles/fulcrum/defaults/main.yml | 6 +- ansible/roles/mempool/defaults/main.yml | 4 +- ansible/roles/phoenixd/defaults/main.yml | 1 - .../deploy_bitcoin_knots_playbook.yml | 5 +- .../deploy_datum_gateway_playbook.yml | 9 +- .../deploy_forgejo_runner_playbook.yml | 2 - .../forgejo/deploy_forgejo_playbook.yml | 2 - ansible/services/forgejo/forgejo_vars.yml | 2 +- .../services/forgejo/setup_backup_forgejo.yml | 1 - .../fulcrum/deploy_fulcrum_playbook.yml | 5 +- .../headscale/deploy_headscale_playbook.yml | 3 - ansible/services/headscale/headscale_vars.yml | 4 +- .../headscale/setup_backup_headscale.yml | 1 - .../lnbits/deploy_lnbits_playbook.yml | 2 - ansible/services/lnbits/lnbits_vars.yml | 2 +- .../services/lnbits/setup_backup_lnbits.yml | 1 - .../services/memos/deploy_memos_playbook.yml | 4 - ansible/services/memos/memos_vars.yml | 2 +- ansible/services/memos/setup_backup_memos.yml | 1 - .../mempool/deploy_mempool_playbook.yml | 3 - .../deploy_ntfy_emergency_app_playbook.yml | 2 - .../ntfy_emergency_app_vars.yml | 2 +- .../services/ntfy/deploy_ntfy_playbook.yml | 2 - ansible/services/ntfy/ntfy_vars.yml | 2 +- .../setup_ntfy_uptime_kuma_notification.yml | 3 - .../deploy_personal_blog_playbook.yml | 2 - .../personal-blog/setup_deploy_alias_lapy.yml | 1 - .../phoenixd/deploy_phoenixd_playbook.yml | 2 - .../deploy_vaultwarden_playbook.yml | 2 - .../disable_vaultwarden_sign_ups_playbook.yml | 1 - .../vaultwarden/setup_backup_vaultwarden.yml | 1 - .../services/vaultwarden/vaultwarden_vars.yml | 2 +- ansible/services_config.yml | 71 ++----- 53 files changed, 82 insertions(+), 403 deletions(-) delete mode 100644 ansible/infra_secrets.yml delete mode 100644 ansible/infra_secrets.yml.example delete mode 100644 ansible/infra_vars.yml diff --git a/ansible/group_vars/all/main.yml b/ansible/group_vars/all/main.yml index 0be0162..1dc3541 100644 --- a/ansible/group_vars/all/main.yml +++ b/ansible/group_vars/all/main.yml @@ -14,3 +14,46 @@ age_backup_recipient: "age192wwdaseqej2ggwyp884gtm05c396anp7chr0vr8m47g50fahpyqr # Public key small-backups-box pulls with # Authorised on each source host for an unprivileged, dedicated user only backup_pull_public_key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIOfIixKMhA9z+Nvyx6ToZIniC8aEgyiInRiboaTTemgX offsite-backup-pull" + + +# ───────────────────────────────────────────────────────────────────────────── +# Subdomains. Global because the edge host proxies for services that live on +# other machines, so no single inventory group covers the readers. Combine with +# root_domain above to build an FQDN. +# +# Moved here from services_config.yml, which 30 plays had to remember to name in +# vars_files: - a file everyone must opt into is a file someone will forget. +# ───────────────────────────────────────────────────────────────────────────── +subdomains: + # Monitoring (watchtower) + ntfy: ntfy + # Uptime Kuma IS still running and this subdomain DOES resolve + # (164.92.239.72, HTTP 302). Only the Ansible code and the vault credentials + # were retired. A comment here previously claimed the opposite. + uptime_kuma: uptime + + # VPN infrastructure (spacey) + headscale: headscale + + # Core services (vipy) + vaultwarden: vault + forgejo: forgejo + lnbits: wallet + + # Secondary services (vipy) + ntfy_emergency_app: avisame + personal_blog: pablohere + + # Memos (memos-box) + memos: memos + + # Mempool block explorer (mempool-box, proxied via vipy) + mempool: mempool + + # DATUM Gateway dashboard (knots-box, proxied via vipy) + datum_gateway: datum + +# Read by plays targeting managed, monitoring, vpn_control and edge - no one +# group covers them, so these are global rather than group_vars/. +ntfy_topic: alerts +headscale_namespace: counter-net diff --git a/ansible/infra/01_user_and_access_setup_playbook.yml b/ansible/infra/01_user_and_access_setup_playbook.yml index c6eed18..0e2c914 100644 --- a/ansible/infra/01_user_and_access_setup_playbook.yml +++ b/ansible/infra/01_user_and_access_setup_playbook.yml @@ -1,7 +1,5 @@ - name: Secure Debian hosts: managed - vars_files: - - ../infra_vars.yml become: true tasks: diff --git a/ansible/infra/02_firewall_and_fail2ban_playbook.yml b/ansible/infra/02_firewall_and_fail2ban_playbook.yml index 9f37c70..309db56 100644 --- a/ansible/infra/02_firewall_and_fail2ban_playbook.yml +++ b/ansible/infra/02_firewall_and_fail2ban_playbook.yml @@ -1,7 +1,5 @@ - name: Secure Debian hosts: managed - vars_files: - - ../infra_vars.yml become: true tasks: diff --git a/ansible/infra/410_disk_usage_alerts.yml b/ansible/infra/410_disk_usage_alerts.yml index dcd9bdc..85947d9 100644 --- a/ansible/infra/410_disk_usage_alerts.yml +++ b/ansible/infra/410_disk_usage_alerts.yml @@ -15,9 +15,7 @@ hosts: managed become: yes vars_files: - - ../infra_vars.yml - ../services_config.yml - - ../infra_secrets.yml vars: disk_usage_threshold_percent: 80 @@ -27,9 +25,8 @@ monitoring_script_path: "{{ monitoring_script_dir }}/disk_usage_monitor.sh" log_file: "{{ monitoring_script_dir }}/disk_usage_monitor.log" systemd_service_name: disk-usage-monitor - # Uptime Kuma configuration (auto-configured from services_config.yml and infra_secrets.yml) + # Uptime Kuma configuration (auto-configured from group_vars/all/) uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - ntfy_topic: "{{ service_settings.ntfy.topic }}" tasks: - name: Validate Uptime Kuma configuration diff --git a/ansible/infra/420_system_healthcheck.yml b/ansible/infra/420_system_healthcheck.yml index 05532a7..a69b456 100644 --- a/ansible/infra/420_system_healthcheck.yml +++ b/ansible/infra/420_system_healthcheck.yml @@ -15,9 +15,7 @@ hosts: managed become: yes vars_files: - - ../infra_vars.yml - ../services_config.yml - - ../infra_secrets.yml vars: healthcheck_interval_seconds: 60 # Send healthcheck every 60 seconds (1 minute) @@ -27,9 +25,8 @@ monitoring_script_path: "{{ monitoring_script_dir }}/system_healthcheck.sh" log_file: "{{ monitoring_script_dir }}/system_healthcheck.log" systemd_service_name: system-healthcheck - # Uptime Kuma configuration (auto-configured from services_config.yml and infra_secrets.yml) + # Uptime Kuma configuration (auto-configured from group_vars/all/) uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - ntfy_topic: "{{ service_settings.ntfy.topic }}" tasks: - name: Validate Uptime Kuma configuration diff --git a/ansible/infra/430_cpu_temp_alerts.yml b/ansible/infra/430_cpu_temp_alerts.yml index 5c9e855..048f216 100644 --- a/ansible/infra/430_cpu_temp_alerts.yml +++ b/ansible/infra/430_cpu_temp_alerts.yml @@ -15,9 +15,7 @@ hosts: hypervisor become: yes vars_files: - - ../infra_vars.yml - ../services_config.yml - - ../infra_secrets.yml vars: temp_threshold_celsius: 80 @@ -27,7 +25,6 @@ log_file: "{{ monitoring_script_dir }}/cpu_temp_monitor.log" systemd_service_name: nodito-cpu-temp-monitor uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - ntfy_topic: "{{ service_settings.ntfy.topic }}" tasks: - name: Validate Uptime Kuma configuration diff --git a/ansible/infra/900_install_rsync.yml b/ansible/infra/900_install_rsync.yml index 6c6b10e..b690009 100644 --- a/ansible/infra/900_install_rsync.yml +++ b/ansible/infra/900_install_rsync.yml @@ -1,7 +1,5 @@ - name: Install rsync hosts: managed - vars_files: - - ../infra_vars.yml become: true tasks: diff --git a/ansible/infra/920_join_headscale_mesh.yml b/ansible/infra/920_join_headscale_mesh.yml index 5b5e4ce..3a8641c 100644 --- a/ansible/infra/920_join_headscale_mesh.yml +++ b/ansible/infra/920_join_headscale_mesh.yml @@ -2,13 +2,11 @@ hosts: managed become: yes vars_files: - - ../infra_vars.yml - ../services_config.yml vars: headscale_host_name: "spacey" headscale_subdomain: "{{ subdomains.headscale }}" headscale_domain: "https://{{ headscale_subdomain }}.{{ root_domain }}" - headscale_namespace: "{{ service_settings.headscale.namespace }}" tasks: - name: Set facts for headscale server connection diff --git a/ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml b/ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml index 86f693f..4edab06 100644 --- a/ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml +++ b/ansible/infra/nodito/30_proxmox_bootstrap_playbook.yml @@ -1,8 +1,6 @@ - name: Bootstrap Nodito SSH Key Access hosts: hypervisor become: true - vars_files: - - ../infra_vars.yml tasks: - name: Install sudo package diff --git a/ansible/infra/nodito/31_proxmox_community_repos_playbook.yml b/ansible/infra/nodito/31_proxmox_community_repos_playbook.yml index 378a674..0fab184 100644 --- a/ansible/infra/nodito/31_proxmox_community_repos_playbook.yml +++ b/ansible/infra/nodito/31_proxmox_community_repos_playbook.yml @@ -1,8 +1,6 @@ - name: Switch Proxmox VE from Enterprise to Community Repositories hosts: hypervisor become: true - vars_files: - - ../infra_vars.yml tasks: - name: Check for deb822 sources format diff --git a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml index 97e0cab..1aad2e7 100644 --- a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml +++ b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml @@ -1,8 +1,6 @@ - name: Setup ZFS RAID 1 Pool for Proxmox Storage hosts: hypervisor become: true - vars_files: - - ../infra_vars.yml tasks: - name: Verify Proxmox VE is running @@ -187,8 +185,6 @@ - name: Setup ZFS Pool Health Monitoring and Monthly Scrubs hosts: hypervisor become: true - vars_files: - - ../../infra_vars.yml vars: zfs_check_interval_seconds: 86400 # 24 hours diff --git a/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml b/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml index e2b1f5d..faf93a5 100644 --- a/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml +++ b/ansible/infra/nodito/33_proxmox_debian_cloud_template.yml @@ -1,8 +1,6 @@ - name: Create Proxmox template from Debian cloud image (no VM clone) hosts: hypervisor become: true - vars_files: - - ../../infra_vars.yml vars: # Defaults (override via vars_files or --extra-vars as needed) diff --git a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml index da1136f..8e188b6 100644 --- a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml +++ b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml @@ -1,8 +1,6 @@ - name: Setup NUT (Network UPS Tools) for CyberPower UPS hosts: hypervisor become: true - vars_files: - - ../../infra_vars.yml tasks: # ------------------------------------------------------------------ @@ -253,8 +251,6 @@ - name: Setup UPS Heartbeat Monitoring hosts: hypervisor become: true - vars_files: - - ../../infra_vars.yml vars: ups_heartbeat_interval_seconds: 60 diff --git a/ansible/infra_secrets.yml b/ansible/infra_secrets.yml deleted file mode 100644 index 567c768..0000000 --- a/ansible/infra_secrets.yml +++ /dev/null @@ -1,194 +0,0 @@ -$ANSIBLE_VAULT;1.1;AES256 -36353131353730316436313965366263666236396164323163376431623232313961323863633035 -6338333965643036666538303864376463623434326564660a396239633264353034653963366466 -66353130376366663730626331346235373935306434303462663763613763663131613335633430 -6335633535333166330a366232306230303033663931333734353636656132383034616336653235 -35636234306134376666666538386561663561626262396135313632393831363665646130643261 -65663137613530393162333838313962393934396135613064333564303666353634396266316135 -65346363663232626261643933353239663635343137646431623561653963363131303033623764 -34616339343835353765363962396665636335623066303062626463343733643137626265643366 -34653139616163666131333330376132343363306132653163343934353034353932383066373462 -31316664383938633361373839353762393838386636393564633430346334323734383430386663 -39313938383830373062336335303135396334323330316537303965653331323130323866663866 -38323462643633396531386632613961343835313931613933373963326465333836323935613436 -63346562313239623162333038663535656633323037363532373536616161303833626539366634 -35363462663539353366623732626237326465663436356630626335623334356465646439376461 -63653230343333613932646431383238373039653461306365326234303365373234613038343338 -30353937353734633631313765346662616138383462353864656264653534653239383936616437 -34313531383438613465626437313130326562316430366234326230303133356631363534396366 -64313732303837666537383161373865653162303034376533636264656564316530613738353435 -38353031616538346438336563336134643261366331623562663534643966306164306132646332 -31336539663131646232353633326637306637353239616331356337353761313763643933303838 -61636333663137623233323561306665666666363731383438373664363538613762393130646261 -66346335373437326536356330613962386662383031366536373635353438316664613362633966 -35393331613634653462323739343466313735376463643035326336396435656364356134626535 -32623036383737623631623332666264653964383939316465353765373132383230336334353463 -38303136356630653561333363306136363230383238666663373130663763336532323631663637 -32633263633835326662366237613639313431366433363730666666333961323036363533613236 -39343963613239333731336461393937633463303530643333626662316563313165613564386663 -34373636383231636461653862356332353161366331633036383837353434373865363139386235 -30323130343534386531666333343765373265626530373335633766353562356562326239336564 -62313033376330383563633138623838633132343130343262663030316435393835313762313564 -30393366393161663265303439383937333739396661393862356132343537616333623035653566 -64316333323733336135383637333739393931333461653731383863616138613234316161623765 -30343964623939663339633139383866336163613933363231303837646565626262303330363338 -64353931623366326666386637363866383761666232663866666438313535393434373933653139 -64636236306632613930363935353736393065353634653235303534316463653438646162663866 -63663839346162396262366563613131356161333137623966653039623436373466663163313938 -64636362383532303136316163306236333134346164376362363730363030643436356631306639 -31376131653162343338613830376466356363356238343436366232393062336433623238323864 -32376133346264383736396166343539363361623864613364613030623936396634616431386263 -35323632353334643466383933356236383432323162386532613535343932353531373065633732 -36626361313436613036363662376337613237353532366163656236323937316437613633383535 -39306334656532383063306332636538666462353836633366353932623962396131663532343364 -64393035303265336638343733366236373466343832333038633138306535303034386134653032 -62633532353462613664636333373461356434666130376332373762653966386238346463333939 -32386233326264666234353139643837636663633362626163356633396465643030383639663538 -31626331393161653038613438613735643364326565633362386331646231333539626133633739 -31336236663863376166636632363965343765666335633464613937653738313230356438313436 -39373936663363626337626164383237626164396337613430373035373339616266666264346164 -66386238373664386138636536616266353466666363383131616532323430663761633139626164 -33323936356565303332383463656533633935373564353461626139343265336334363433333138 -64363534363836666338366339636435366334383339663463376164613866333839366264666237 -35306231656536613138353663626166303366373761323866666436356438653764353431383430 -33323766643431303337323265646631373434393436333335666366306138356337343163336362 -39323131653734366561613864383266386131626164363361656534386363613565336532666162 -38333031656330306635313032343861653037323037616564323537666261316438303631616530 -35303238366338376363383764633865326530643736306634363436643330376432396535363932 -39646666333234326261373835646336663138366237346562396238326530356533326363373664 -38303634383036323232363934626562663864646239653066326565623635653535623137663065 -65376633356136333938623763396138376630636636376434333463313833386435633365623036 -30373937353433396264643231323832373963396437363562623639353764623063363733366165 -38613833353231333836373637626133323965316665313837653335386534326434363034653630 -38323335666466306665663262343737636564393934373338623130636362383564376434626533 -35623761396465333730626563633734373131373132343262653333343439623433313964623463 -33613639616636393563663331326535346131663832313861353434313965393164643538643437 -36363130363934623333666330353236386530626536616333303131396265386263363462356364 -38393864623461643831333731656566636532643338306361666335316265396430376432393730 -34633638636466313232376363306333356564393038323130313634386162333639366162623437 -66306234346430346537343338343064623131366337356262653130616464646336353136336234 -34623063336332666138303961333332326632326137366664306263666538616361316363323163 -38613431353762653139636365353031343465366138396161366538393833666236656264623132 -63313630633231613438613865613333343066376261326333623864636430643866656132353538 -65653731626464323930383636663364313437313531646462623163646630646664356335303764 -61313462623234303961356363386464333061306239643263333135653538313338383134343236 -63396461346438373830643031336165376563636530353836626436613561313536623130613663 -63383939663535326236313736613630313061323837346339353834313434393437383237633534 -65363334643238373864636530386434636565623131366433313562623933366633396565646565 -33626433373834633638656365363261643866343961663566333761306138383865316536303664 -35653463633336616438373732303938393032663961653262333134353762333433373039636638 -33366466336133383061373866323231663038373831386139613535376539393165363034303537 -35346638613137616666623362396564346464376133316262646339333936363961316161663332 -38303239356334633137316333363439303935346230613032643564353734643863636264316136 -39663131376638653138623838323739343064623430386166643962633363373335316239663339 -37633934393664346433316633353831373534653437623065363134626661323732656664633639 -32616539323931626130353836303533363232353066343666366263663565623862333062323237 -31393935313133333837626632346461356134376339633239396438336333376534353063633937 -61653230653666623935346134666130373934643438323464346239336430373635396165323738 -36613432356239323039386364643535346261373739363463373839373063346565616466663239 -33353165353631343730636531363565383363623839346434383737313136646330323461633665 -32376631663637353534636262666532643636366237383539346235383832313933336432363734 -35653866643136383934613261633439343831363234616438333530336437326562366536383834 -33343330316337323064386464656436333430643061663665306534386235336437643965393165 -61373038323539316161363731306334633833376532626537643164373638303438326637376630 -35343333343364326538333862643265336166396466373362326162303430383434626535336163 -31666662653261343631366165656131376462333063626134613466636637383936303933346362 -32386162323230666563353838613033393936386535646130613861356563633566393430313661 -37666561643361653630353262653234636232303534393661343834366562353864343638313134 -36656463653166393432646263626530316361643761646265613534643563306161333539633361 -33613830356238623962623739623437646635356334386637633035373262313539386662353732 -35666534663965333836396664316432633137363761623838356430316566383131363530316462 -62636235396136363538653836633133663465386139636339353664666634343238663861613437 -62656435363832623461363662643262323265353066376538306163376438303838373961303239 -63623066656532353933396534623637356362356231336361393534393465633565653132656130 -37363863666136643461383033653936383935333131343565643664386664326463663466633361 -36313638616533623431316361386539383434653866376363383630323632636237373666393561 -31313135303339633763303762653939386663393439646135656537303331366236396230303063 -32336336366533323936633166623434623365633163643461653966383563653533363335666638 -61313764656437613138663038386336336664396539373930373933326234636662653833656435 -61343962373935653238326265663164363561616435363134313634666331636431633133356563 -37326133363762383266353730343537373934653634366336316635653063313461303237633164 -61333231353135343666343461306161626432636331653962613334376439613265653437663333 -34623638313561356464356534343635653265633531323736386335656635303263636663323930 -39373462663438326366363433343735343234383735643363326339626135313564646432656330 -38653836653938383932643032616534363035366530336663653235366230353332636531653138 -63623537353165623037383361643937623466356464313931666430666632633866363232623230 -32353766383432613233313331356432326430383132616263656361613334383936643066326133 -30643935653662376433353039323836613239336265376663616336663264343331373236373237 -64326462626566386563623235633664653665356461316362663462343665303233313035366435 -32623465656138303161396233623661333238306334646663353263383437653461383366366466 -39303061363530373339623435613765313637616634313731666631653161623439323734383735 -65383666366330623338653561353032373231396431353466313335303935386136613566336331 -62326539383631636263353234393866333031643737663064383130313066396461663466623431 -38616364303439393432663035623138353264656635393633646363366538633262306466336564 -35623733613564323233303034636264336464373566303065383438366338643538666638643466 -63343130613461316465626237343235616536613838623930636136333131616437656631376564 -34366535383961656237333030376139353237343636306165646161323732646663376231393832 -61396263393561356438326237633634333533646638383865336164656465383166373236643063 -63643263303132656336333339623539663765333039396364643462396161366562626362343162 -34363936323230373839373734343332663664373634313364383062353130363362623937656139 -32653237643061646531623030646335383463333136366264363133303666663261343631663762 -36346432386465393765353763353265633837623165303634646137646564356237653336666331 -38376463376130353863636334353633313361313239373264333134333232353765346666326665 -65623333396165353334636532303537366566666130343366383964396365333461646566333431 -37303032643731346361393963303061363837373562393866353962333231623763623236323161 -39383538663837326663636166643864393733653764353536303433653563316339663438353838 -37616135373336326536333932636538336465323130626266613930393266633164636439663532 -37633264356661353836346462626538323631613539636631396436646139666538383838323437 -62663561356430336235343439633135326661653031653063363030326132336462336164373232 -31363131326334613361356461353934656364346666663762306161356463386332366562336630 -39326236656335363630363564316134623435343538386462663161396332366639363033383235 -35306434646330636137323939623565663939643161616336386633633133363963383739633434 -32336433663132363239356266653461633033623232346135353032353265343861316638643265 -66303932643235333764323239653430346132666136663133383935613962353235313336306136 -38373938376238313034346237373035383861613930323936663831346538313937343538663737 -62616139663634333636333635636364643966303565656265653634653437633233396138323831 -64313465643233663663303130653133626638313162356133316139333030663530313232626437 -35613439326639393032393962653933376636663934333764393064346465633637613836333933 -33646336653533666631656132623035323963623134653432363133646339346363336263323135 -34356233373965333362326132383762626436333365363731396134376664363465623362316366 -31623361376161393836623732613034643831363838663733386561313961373632643639666635 -37633061303235306437666337306236316331616330396632356266636137313833346366653031 -64366236306139363861323237396232336665633331326139373461353237336432373366333266 -32326565376464383937356562633730383961306535666464383364656137633662333662626366 -30653736646333333963656566326431376361666465363532393765393764633562626232643836 -32626261373835356634633461303664653362346231343030343433376364643664383464636666 -35323738366332306337616563646231663963353135623133613636666363396530643239626566 -35383330663138333631386363643032383161626439633437343934366635656433386163613663 -30623738656535396134363830626338353864356330613131623832363330633064376531303965 -35313036393031383035633035636165313363333564633938306166666639636436353733303662 -66653264356432383166333630646533313736366130666331306537393262363538313030646134 -38306363663932313230346664303531636639656339323062333739303239333861616533666132 -30353332303233323837346234336633646163643137636166633330633464663935653838313161 -37393533646139313236393234313763353533663638363031393764633862363938643838353937 -36613564353366663434633839303036343665313933326531353831396139613330316632613637 -37666432363330326232366664656462313336323866316633396533313638373462386365663332 -63626136396661353739633263363038326630623037353831323930346431666263333431643562 -39373133636465373064613664323335353236326562343966616439646565383934613462623363 -38613962386236356666643038303435376266656165336263653365353537666362616638636639 -34326262303930303237636339666563613663373863666339663135326661663866346264613734 -34353565613832323132343730396535656264376233356162353265623739613332333261393331 -33616230333033613766643264396633343535376461633330633064613662336532613163373962 -61343065373630383838306631633031656566343765333365373932373234313733396165623639 -65313731346632333235303463393039656232653163336161633265376434623466373361333865 -66323664343766306566663335656138323537656563383835653263323039656166613330613733 -38323064306365656561313163306439653937623536616435616431616133643466336362666330 -64303832343739383430626339663532653363626263393237343234666266323933316566343937 -32373335373237653265303638663331363838313961343936616639646438343633326461626363 -30653964646639663836636437653861343064393332306638363864326439626335373663653335 -33653230626337633863356534393466633162323734623135393931363339383163353466633031 -39343435616137373166376538633330386232333666316566336234373933613534666430353336 -33626637346337623132313833343430613832303932323739326537666531356630356137653963 -30623262373032353934353938346263396338336265393937336539386530343062623761646632 -39653035393535356630626164393464663738616239613166613364396539343039306538613230 -38636261633932643965323966346632616264343262643933346163656436326330336564623265 -30646561303563323937393533663066393733353638323332663736353766336664343265613733 -66643631633864386262343232636465613962386538366163613734386635303837636164383465 -35373331363864393934633563623662326632386534323436663634613030646563643035386539 -37323262646365633739626539346638383039643433343837363561303530376434336131353965 -32336133393139626232313933636136646264613338643238646431316566333662353837636232 -32623639636432333134643962313430623533333237343464383135643361336236396533653761 -66383165363539363938653130633630643865636663313665353839646462613334383266663563 -62333035626539383739656565633861363863333563656339353735336330333633633861643130 -35363062306165623861316639336237373434353437626461383931656531306134313264383664 -37353966616136326264376135373532653630393335336665343463323466353162 diff --git a/ansible/infra_secrets.yml.example b/ansible/infra_secrets.yml.example deleted file mode 100644 index c95234d..0000000 --- a/ansible/infra_secrets.yml.example +++ /dev/null @@ -1,38 +0,0 @@ -# Uptime Kuma login credentials -# Used by the disk monitoring playbook to create monitors automatically - - -# ntfy credentials -# Used for notification channel setup in Uptime Kuma - -ntfy_username: "your_ntfy_username" -ntfy_password: "your_ntfy_password" - -# headscale-ui credentials -# Used for HTTP basic authentication via Caddy -# Provide either: -# - headscale_ui_password: plain text password (will be hashed automatically) -# - headscale_ui_password_hash: pre-hashed bcrypt password (more secure, use caddy hash-password to generate) - -headscale_ui_username: "admin" -headscale_ui_password: "your_secure_password_here" -# headscale_ui_password_hash: "$2a$14$..." # Optional: pre-hashed password - -bitcoin_rpc_user: "bitcoinrpc" -bitcoin_rpc_password: "CHANGE_ME_TO_SECURE_PASSWORD" - -# Mempool MariaDB credentials -# Used by: services/mempool/deploy_mempool_playbook.yml -mariadb_mempool_password: "CHANGE_ME_TO_SECURE_PASSWORD" - -# Forgejo Runner registration token -# Used by: services/forgejo-runner/deploy_forgejo_runner_playbook.yml -# See: services/forgejo-runner/SETUP.md for how to obtain this token -forgejo_runner_registration_token: "YOUR_RUNNER_TOKEN_HERE" - -# DATUM Gateway secrets -# Used by: services/datum-gateway/deploy_datum_gateway_playbook.yml -datum_mining_address: "YOUR_BITCOIN_ADDRESS_FOR_BLOCK_REWARDS" -datum_gateway_admin_password: "CHANGE_ME_TO_SECURE_PASSWORD" -datum_dashboard_username: "admin" -datum_dashboard_password_hash: "$2a$14$..." # Generate with: caddy hash-password diff --git a/ansible/infra_vars.yml b/ansible/infra_vars.yml deleted file mode 100644 index 36d35f8..0000000 --- a/ansible/infra_vars.yml +++ /dev/null @@ -1,9 +0,0 @@ -new_user: counterweight -ssh_port: 22 -allow_ssh_from: "any" -root_domain: contrapeso.xyz - -# Uptime Kuma was decommissioned on 2026-09-11. The monitoring blocks in the -# playbooks are kept deliberately — the check logic is meant to be rewired to -# whatever replaces it. This flag keeps them inert until then. See archive/uptime_kuma/. -uptime_kuma_enabled: false diff --git a/ansible/roles/bitcoin_knots/defaults/main.yml b/ansible/roles/bitcoin_knots/defaults/main.yml index 29d5aa0..cd9d9c1 100644 --- a/ansible/roles/bitcoin_knots/defaults/main.yml +++ b/ansible/roles/bitcoin_knots/defaults/main.yml @@ -1,8 +1,10 @@ # Bitcoin Knots Configuration Variables # Version - REQUIRED: Specify exact version/tag to build -bitcoin_knots_version: "v29.2.knots20251110" # Must specify exact version/tag -bitcoin_knots_version_short: "29.2.knots20251110" # Version without 'v' prefix (for tarball URLs) +# The only version string. There used to be a second, v-prefixed copy +# (bitcoin_knots_version) that nothing read - two hand-maintained copies of one +# fact, with nothing keeping them in step. +bitcoin_knots_version_short: "29.2.knots20251110" # Directories bitcoin_knots_dir: /opt/bitcoin-knots diff --git a/ansible/roles/caddy_site/defaults/main.yml b/ansible/roles/caddy_site/defaults/main.yml index cafa3dc..0a59906 100644 --- a/ansible/roles/caddy_site/defaults/main.yml +++ b/ansible/roles/caddy_site/defaults/main.yml @@ -15,7 +15,8 @@ caddy_site_headers_up: {} # {"X-Forwarded-Host": "wallet.example.com"} # expression for the username silently passes through as literal text. caddy_site_basic_auth: [] # [{user: "{{ x_user }}", hash: "{{ x_hash }}"}] -# Placement. caddy_sites_dir comes from services_config.yml; this is the fallback. +# Placement. This is now the only definition of caddy_sites_dir - services_config.yml +# used to carry an identical copy, which was removed as redundant. caddy_sites_dir: /etc/caddy/sites-enabled # Rendered site files can carry credentials (basic_auth hashes), so --diff is diff --git a/ansible/roles/datum_gateway/defaults/main.yml b/ansible/roles/datum_gateway/defaults/main.yml index d3647ec..3a8cac4 100644 --- a/ansible/roles/datum_gateway/defaults/main.yml +++ b/ansible/roles/datum_gateway/defaults/main.yml @@ -31,7 +31,7 @@ datum_gateway_build_jobs: 4 # The gateway runs on the same host as Bitcoin Knots so localhost RPC works. # datum_bitcoin_rpc_url should include http:// and port. datum_bitcoin_rpc_url: "http://127.0.0.1:8332" -# Note: bitcoin_rpc_user and bitcoin_rpc_password come from infra_secrets.yml +# Note: bitcoin_rpc_user and bitcoin_rpc_password come from group_vars/all/vault.yml # Mining config datum_coinbase_tag_primary: "DATUM" diff --git a/ansible/roles/forgejo_runner/defaults/main.yml b/ansible/roles/forgejo_runner/defaults/main.yml index e0bac24..4a80d73 100644 --- a/ansible/roles/forgejo_runner/defaults/main.yml +++ b/ansible/roles/forgejo_runner/defaults/main.yml @@ -21,8 +21,6 @@ forgejo_instance_url: "https://forgejo.contrapeso.xyz" # systemd stores it, so `systemctl is-failed forgejo-runner-healthcheck.service` # answers the question with no monitoring system involved at all. healthcheck_interval_seconds: 60 -healthcheck_timeout_seconds: 90 -healthcheck_retries: 1 healthcheck_script_dir: /opt/forgejo-runner-healthcheck healthcheck_script_path: "{{ healthcheck_script_dir }}/forgejo_runner_healthcheck.sh" healthcheck_log_file: "{{ healthcheck_script_dir }}/forgejo_runner_healthcheck.log" diff --git a/ansible/roles/fulcrum/defaults/main.yml b/ansible/roles/fulcrum/defaults/main.yml index df213be..0d5c190 100644 --- a/ansible/roles/fulcrum/defaults/main.yml +++ b/ansible/roles/fulcrum/defaults/main.yml @@ -11,7 +11,7 @@ fulcrum_binary_path: /usr/local/bin/Fulcrum # Network - Bitcoin RPC connection # Bitcoin Knots is on a different host (knots_box_local) -# Using RPC user/password authentication (credentials from infra_secrets.yml) +# Using RPC user/password authentication (credentials from group_vars/all/vault.yml) # Addressed by Tailscale name, never a LAN IP. This was # bitcoin_rpc_host: "192.168.1.140" # IP of knots_box_local # but .140 is fulcrum-box ITSELF - knots-box is .135. The DHCP leases had @@ -21,7 +21,7 @@ fulcrum_binary_path: /usr/local/bin/Fulcrum # agree with it. bitcoin_rpc_host: "knots-box" bitcoin_rpc_port: 8332 # Bitcoin Knots RPC port -# Note: bitcoin_rpc_user and bitcoin_rpc_password are loaded from infra_secrets.yml +# Note: bitcoin_rpc_user and bitcoin_rpc_password are loaded from group_vars/all/vault.yml # Network - Fulcrum server fulcrum_tcp_port: 50001 @@ -42,8 +42,6 @@ fulcrum_ssl_cert_path: "{{ fulcrum_config_dir }}/fulcrum.crt" fulcrum_ssl_key_path: "{{ fulcrum_config_dir }}/fulcrum.key" fulcrum_ssl_cert_days: 3650 # 10 years validity for self-signed cert -# Port forwarding configuration (for public access via VPS) -fulcrum_tailscale_hostname: "{{ service_settings.fulcrum.tailscale_hostname }}" # Performance # db_mem will be calculated as 75% of available RAM automatically in playbook diff --git a/ansible/roles/mempool/defaults/main.yml b/ansible/roles/mempool/defaults/main.yml index d0e36b2..586b004 100644 --- a/ansible/roles/mempool/defaults/main.yml +++ b/ansible/roles/mempool/defaults/main.yml @@ -11,7 +11,7 @@ mempool_mysql_dir: "{{ mempool_dir }}/mysql" # Network - Bitcoin Core/Knots connection (via Tailnet Magic DNS) bitcoin_host: "knots-box" bitcoin_rpc_port: 8332 -# Note: bitcoin_rpc_user and bitcoin_rpc_password are loaded from infra_secrets.yml +# Note: bitcoin_rpc_user and bitcoin_rpc_password are loaded from group_vars/all/vault.yml # Network - Fulcrum Electrum server (via Tailnet Magic DNS) fulcrum_host: "fulcrum-box" @@ -30,7 +30,7 @@ mempool_backend_port: 8999 # MariaDB settings mariadb_database: "mempool" mariadb_user: "mempool" -# Note: mariadb_mempool_password is loaded from infra_secrets.yml +# Note: mariadb_mempool_password is loaded from group_vars/all/vault.yml diff --git a/ansible/roles/phoenixd/defaults/main.yml b/ansible/roles/phoenixd/defaults/main.yml index ea9c889..9fea220 100644 --- a/ansible/roles/phoenixd/defaults/main.yml +++ b/ansible/roles/phoenixd/defaults/main.yml @@ -35,7 +35,6 @@ phoenixd_http_bind_port: 9740 # Optional webhook for payment events. Leave empty to disable. phoenixd_webhook_url: "" -phoenixd_monitor_name: "Phoenixd" diff --git a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml index bc073bd..709cb30 100644 --- a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +++ b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml @@ -10,9 +10,7 @@ hosts: bitcoin become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml vars: # Preserves the push URL this check has been reporting to. The role knows # nothing about Uptime Kuma — this is just "a URL that accepts a ping". @@ -24,7 +22,6 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml tasks: - name: Expose Bitcoin P2P through a socket proxy @@ -34,7 +31,7 @@ socket_proxy_name: bitcoin-p2p socket_proxy_description: "Bitcoin P2P" socket_proxy_listen_port: "{{ service_settings.bitcoin.p2p_port }}" - socket_proxy_upstream_host: "{{ service_settings.bitcoin.tailscale_hostname }}" + socket_proxy_upstream_host: "{{ hostvars['knots_box_local'].ansible_host }}" socket_proxy_documentation: "https://github.com/bitcoin/bitcoin" socket_proxy_free_bind: true socket_proxy_timeout_stop_sec: 5 diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index 371f3fa..ebb9633 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -11,9 +11,7 @@ hosts: bitcoin become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml vars: # Preserves the push URL this check reports to. The role knows nothing about # Uptime Kuma — this is just "a URL that accepts a ping". @@ -25,9 +23,7 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml tasks: - name: Publish the DATUM Gateway dashboard through Caddy ansible.builtin.include_role: @@ -35,7 +31,7 @@ vars: caddy_site_name: datum-gateway caddy_site_domain: "{{ subdomains.datum_gateway }}.{{ root_domain }}" - caddy_site_upstream: "{{ service_settings.datum_gateway.tailscale_hostname }}:{{ service_settings.datum_gateway.api_port }}" + caddy_site_upstream: "{{ hostvars['knots_box_local'].ansible_host }}:{{ service_settings.datum_gateway.api_port }}" caddy_site_resolvers: "100.100.100.100" caddy_site_basic_auth: - user: "{{ datum_dashboard_username }}" @@ -52,7 +48,6 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml tasks: - name: Expose the DATUM Stratum port through a socket proxy @@ -62,7 +57,7 @@ socket_proxy_name: datum-stratum socket_proxy_description: "DATUM Stratum" socket_proxy_listen_port: "{{ service_settings.datum_gateway.stratum_port }}" - socket_proxy_upstream_host: "{{ service_settings.datum_gateway.tailscale_hostname }}" + socket_proxy_upstream_host: "{{ hostvars['knots_box_local'].ansible_host }}" # Matches the UFW comment already on the edge host; the derived default # would say "DATUM Stratum" and rewrite the rule. socket_proxy_ufw_comment: "DATUM Gateway Stratum public access" diff --git a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml index 446cdff..9195087 100644 --- a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml +++ b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml @@ -3,9 +3,7 @@ hosts: ci_runner become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml vars: # Preserves the push URL this host has been reporting to all along, so the # move to a role changes no behaviour. The role itself knows nothing about diff --git a/ansible/services/forgejo/deploy_forgejo_playbook.yml b/ansible/services/forgejo/deploy_forgejo_playbook.yml index db78e95..a929d42 100644 --- a/ansible/services/forgejo/deploy_forgejo_playbook.yml +++ b/ansible/services/forgejo/deploy_forgejo_playbook.yml @@ -2,9 +2,7 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - ./forgejo_vars.yml vars: forgejo_subdomain: "{{ subdomains.forgejo }}" diff --git a/ansible/services/forgejo/forgejo_vars.yml b/ansible/services/forgejo/forgejo_vars.yml index 9fb4cc9..7bba0ed 100644 --- a/ansible/services/forgejo/forgejo_vars.yml +++ b/ansible/services/forgejo/forgejo_vars.yml @@ -9,7 +9,7 @@ forgejo_url: "https://codeberg.org/forgejo/forgejo/releases/download/v{{ forgejo forgejo_bin_path: "/usr/local/bin/forgejo" forgejo_user: "git" -# (caddy_sites_dir and subdomain now in services_config.yml) +# (subdomain in group_vars/all/main.yml, caddy_sites_dir in roles/caddy_site/defaults/) # Remote access remote_host_name: "{{ groups['edge'] | first }}" diff --git a/ansible/services/forgejo/setup_backup_forgejo.yml b/ansible/services/forgejo/setup_backup_forgejo.yml index d6aafef..ee1769b 100644 --- a/ansible/services/forgejo/setup_backup_forgejo.yml +++ b/ansible/services/forgejo/setup_backup_forgejo.yml @@ -9,7 +9,6 @@ hosts: edge become: yes vars_files: - - ../../group_vars/all/main.yml - ./forgejo_vars.yml tasks: diff --git a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml index 6f5c55f..6b8e207 100644 --- a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml +++ b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml @@ -6,9 +6,7 @@ hosts: electrum become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml vars: # Preserves the push URL this check has been configured with. The role knows # nothing about Uptime Kuma — this is just "a URL that accepts a ping". @@ -20,7 +18,6 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml tasks: - name: Expose Fulcrum SSL through a socket proxy @@ -30,4 +27,4 @@ socket_proxy_name: fulcrum-ssl socket_proxy_description: "Fulcrum SSL" socket_proxy_listen_port: "{{ service_settings.fulcrum.ssl_port }}" - socket_proxy_upstream_host: "{{ service_settings.fulcrum.tailscale_hostname }}" + socket_proxy_upstream_host: "{{ hostvars['fulcrum_box_local'].ansible_host }}" diff --git a/ansible/services/headscale/deploy_headscale_playbook.yml b/ansible/services/headscale/deploy_headscale_playbook.yml index 527e4c2..4308b1e 100644 --- a/ansible/services/headscale/deploy_headscale_playbook.yml +++ b/ansible/services/headscale/deploy_headscale_playbook.yml @@ -2,15 +2,12 @@ hosts: vpn_control become: no vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - ./headscale_vars.yml vars: headscale_subdomain: "{{ subdomains.headscale }}" headscale_domain: "{{ headscale_subdomain }}.{{ root_domain }}" headscale_base_domain: "tailnet.{{ root_domain }}" - headscale_namespace: "{{ service_settings.headscale.namespace }}" uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: diff --git a/ansible/services/headscale/headscale_vars.yml b/ansible/services/headscale/headscale_vars.yml index ab12a82..c3fb948 100644 --- a/ansible/services/headscale/headscale_vars.yml +++ b/ansible/services/headscale/headscale_vars.yml @@ -1,5 +1,5 @@ # Headscale service configuration -# (subdomain and caddy_sites_dir now in services_config.yml) +# (subdomain in group_vars/all/main.yml, caddy_sites_dir in roles/caddy_site/defaults/) headscale_port: 8080 headscale_grpc_port: 50443 @@ -10,7 +10,7 @@ headscale_version: "0.26.1" # Data directory headscale_data_dir: /var/lib/headscale -# Namespace now configured in services_config.yml under service_settings.headscale.namespace +# Namespace is headscale_namespace in group_vars/all/main.yml # Remote access remote_host_name: "{{ groups['vpn_control'] | first }}" diff --git a/ansible/services/headscale/setup_backup_headscale.yml b/ansible/services/headscale/setup_backup_headscale.yml index 9c72d1d..5ae3ad1 100644 --- a/ansible/services/headscale/setup_backup_headscale.yml +++ b/ansible/services/headscale/setup_backup_headscale.yml @@ -3,7 +3,6 @@ hosts: vpn_control become: yes vars_files: - - ../../group_vars/all/main.yml - ./headscale_vars.yml tasks: diff --git a/ansible/services/lnbits/deploy_lnbits_playbook.yml b/ansible/services/lnbits/deploy_lnbits_playbook.yml index 65bdfbf..3aa7a95 100644 --- a/ansible/services/lnbits/deploy_lnbits_playbook.yml +++ b/ansible/services/lnbits/deploy_lnbits_playbook.yml @@ -2,9 +2,7 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - ./lnbits_vars.yml vars: lnbits_subdomain: "{{ subdomains.lnbits }}" diff --git a/ansible/services/lnbits/lnbits_vars.yml b/ansible/services/lnbits/lnbits_vars.yml index e855d57..466ce78 100644 --- a/ansible/services/lnbits/lnbits_vars.yml +++ b/ansible/services/lnbits/lnbits_vars.yml @@ -3,7 +3,7 @@ lnbits_dir: /opt/lnbits lnbits_data_dir: "{{ lnbits_dir }}/data" lnbits_port: 8765 -# (caddy_sites_dir and subdomain now in services_config.yml) +# (subdomain in group_vars/all/main.yml, caddy_sites_dir in roles/caddy_site/defaults/) # Remote access remote_host_name: "{{ groups['edge'] | first }}" diff --git a/ansible/services/lnbits/setup_backup_lnbits.yml b/ansible/services/lnbits/setup_backup_lnbits.yml index 46abf31..0c45b29 100644 --- a/ansible/services/lnbits/setup_backup_lnbits.yml +++ b/ansible/services/lnbits/setup_backup_lnbits.yml @@ -9,7 +9,6 @@ hosts: edge become: yes vars_files: - - ../../group_vars/all/main.yml - ./lnbits_vars.yml tasks: diff --git a/ansible/services/memos/deploy_memos_playbook.yml b/ansible/services/memos/deploy_memos_playbook.yml index 04e99cf..e078d84 100644 --- a/ansible/services/memos/deploy_memos_playbook.yml +++ b/ansible/services/memos/deploy_memos_playbook.yml @@ -2,9 +2,7 @@ hosts: memos become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - ./memos_vars.yml vars: memos_subdomain: "{{ subdomains.memos }}" @@ -160,9 +158,7 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - ./memos_vars.yml vars: memos_subdomain: "{{ subdomains.memos }}" diff --git a/ansible/services/memos/memos_vars.yml b/ansible/services/memos/memos_vars.yml index e1c42a3..94c6de7 100644 --- a/ansible/services/memos/memos_vars.yml +++ b/ansible/services/memos/memos_vars.yml @@ -12,7 +12,7 @@ memos_url: "https://github.com/usememos/memos/releases/download/v{{ memos_versio memos_tailscale_hostname: "memos-box" memos_tailscale_ip: "100.64.0.4" -# (caddy_sites_dir and subdomain in services_config.yml) +# (subdomain in group_vars/all/main.yml, caddy_sites_dir in roles/caddy_site/defaults/) # Remote access (for backup from lapy via Tailscale) diff --git a/ansible/services/memos/setup_backup_memos.yml b/ansible/services/memos/setup_backup_memos.yml index 2131924..36b8cf6 100644 --- a/ansible/services/memos/setup_backup_memos.yml +++ b/ansible/services/memos/setup_backup_memos.yml @@ -7,7 +7,6 @@ hosts: memos become: yes vars_files: - - ../../group_vars/all/main.yml - ./memos_vars.yml tasks: diff --git a/ansible/services/mempool/deploy_mempool_playbook.yml b/ansible/services/mempool/deploy_mempool_playbook.yml index 0766da0..c2c8f4f 100644 --- a/ansible/services/mempool/deploy_mempool_playbook.yml +++ b/ansible/services/mempool/deploy_mempool_playbook.yml @@ -3,9 +3,7 @@ hosts: mempool become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml vars: # Preserves the three push URLs these checks have been reporting to all # along, so the move to a role changes no behaviour. The role knows nothing @@ -22,7 +20,6 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml vars: mempool_domain: "{{ subdomains.mempool }}.{{ root_domain }}" diff --git a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml index 7379d5f..1b53732 100644 --- a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml +++ b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml @@ -2,8 +2,6 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - - ../../infra_secrets.yml - ../../services_config.yml - ./ntfy_emergency_app_vars.yml vars: diff --git a/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml b/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml index a3bb480..59ae5d6 100644 --- a/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml +++ b/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml @@ -2,7 +2,7 @@ ntfy_emergency_app_dir: /opt/ntfy-emergency-app ntfy_emergency_app_port: 3000 -# (caddy_sites_dir and subdomain now in services_config.yml) +# (subdomain in group_vars/all/main.yml, caddy_sites_dir in roles/caddy_site/defaults/) # ntfy configuration ntfy_emergency_app_topic: "emergencia" diff --git a/ansible/services/ntfy/deploy_ntfy_playbook.yml b/ansible/services/ntfy/deploy_ntfy_playbook.yml index 61fafe1..d6253b5 100644 --- a/ansible/services/ntfy/deploy_ntfy_playbook.yml +++ b/ansible/services/ntfy/deploy_ntfy_playbook.yml @@ -2,8 +2,6 @@ hosts: monitoring become: yes vars_files: - - ../../infra_vars.yml - - ../../infra_secrets.yml - ../../services_config.yml - ./ntfy_vars.yml vars: diff --git a/ansible/services/ntfy/ntfy_vars.yml b/ansible/services/ntfy/ntfy_vars.yml index 5ebec37..ba51792 100644 --- a/ansible/services/ntfy/ntfy_vars.yml +++ b/ansible/services/ntfy/ntfy_vars.yml @@ -1,3 +1,3 @@ ntfy_port: 6674 -# ntfy_topic now lives in services_config.yml under service_settings.ntfy.topic \ No newline at end of file +# ntfy_topic lives in group_vars/all/main.yml \ No newline at end of file diff --git a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml index 2d3d22a..861bfa5 100644 --- a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml +++ b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml @@ -15,14 +15,11 @@ hosts: monitoring become: no vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - ./ntfy_vars.yml vars: ntfy_subdomain: "{{ subdomains.ntfy }}" - ntfy_topic: "{{ service_settings.ntfy.topic }}" uptime_kuma_subdomain: "{{ subdomains.uptime_kuma }}" ntfy_domain: "{{ ntfy_subdomain }}.{{ root_domain }}" ntfy_server_url: "https://{{ ntfy_domain }}" diff --git a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml index 96d030f..968f423 100644 --- a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml +++ b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml @@ -2,9 +2,7 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - ./personal_blog_vars.yml vars: personal_blog_subdomain: "{{ subdomains.personal_blog }}" diff --git a/ansible/services/personal-blog/setup_deploy_alias_lapy.yml b/ansible/services/personal-blog/setup_deploy_alias_lapy.yml index 99f9b34..bd9b715 100644 --- a/ansible/services/personal-blog/setup_deploy_alias_lapy.yml +++ b/ansible/services/personal-blog/setup_deploy_alias_lapy.yml @@ -2,7 +2,6 @@ hosts: control gather_facts: no vars_files: - - ../../infra_vars.yml - ./personal_blog_vars.yml vars: bashrc_path: "{{ lookup('env', 'HOME') }}/.bashrc" diff --git a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml index 1e53ff0..45bea3c 100644 --- a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml +++ b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml @@ -5,9 +5,7 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml vars: # phoenixd's health check has never reported anywhere since the Uptime Kuma # decommissioning — its systemd Environment= was left empty. Leaving it empty diff --git a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml index 74e87d8..987fe27 100644 --- a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml +++ b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml @@ -2,9 +2,7 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ../../services_config.yml - - ../../infra_secrets.yml - ./vaultwarden_vars.yml vars: vaultwarden_subdomain: "{{ subdomains.vaultwarden }}" diff --git a/ansible/services/vaultwarden/disable_vaultwarden_sign_ups_playbook.yml b/ansible/services/vaultwarden/disable_vaultwarden_sign_ups_playbook.yml index eebb214..bccc2cd 100644 --- a/ansible/services/vaultwarden/disable_vaultwarden_sign_ups_playbook.yml +++ b/ansible/services/vaultwarden/disable_vaultwarden_sign_ups_playbook.yml @@ -2,7 +2,6 @@ hosts: edge become: yes vars_files: - - ../../infra_vars.yml - ./vaultwarden_vars.yml tasks: diff --git a/ansible/services/vaultwarden/setup_backup_vaultwarden.yml b/ansible/services/vaultwarden/setup_backup_vaultwarden.yml index 8430475..54fe147 100644 --- a/ansible/services/vaultwarden/setup_backup_vaultwarden.yml +++ b/ansible/services/vaultwarden/setup_backup_vaultwarden.yml @@ -6,7 +6,6 @@ hosts: edge become: yes vars_files: - - ../../group_vars/all/main.yml - ./vaultwarden_vars.yml tasks: diff --git a/ansible/services/vaultwarden/vaultwarden_vars.yml b/ansible/services/vaultwarden/vaultwarden_vars.yml index 0605d27..9edc2cd 100644 --- a/ansible/services/vaultwarden/vaultwarden_vars.yml +++ b/ansible/services/vaultwarden/vaultwarden_vars.yml @@ -3,7 +3,7 @@ vaultwarden_dir: /opt/vaultwarden vaultwarden_data_dir: "{{ vaultwarden_dir }}/data" vaultwarden_port: 8222 -# (caddy_sites_dir and subdomain now in services_config.yml) +# (subdomain in group_vars/all/main.yml, caddy_sites_dir in roles/caddy_site/defaults/) # Remote access remote_host_name: "{{ groups['edge'] | first }}" diff --git a/ansible/services_config.yml b/ansible/services_config.yml index a6dc9c6..531b824 100644 --- a/ansible/services_config.yml +++ b/ansible/services_config.yml @@ -1,67 +1,24 @@ -# Centralized Services Configuration -# Subdomains and Caddy settings for all services - -# Edit these subdomains to match your preferences -subdomains: - # Monitoring Services (on watchtower) - ntfy: ntfy - # DEPRECATED 2026-09-11 — Uptime Kuma is decommissioned and this subdomain no - # longer resolves to anything. Kept only because the deprecated monitoring - # blocks still template it into uptime_kuma_api_url. See archive/uptime_kuma/. - uptime_kuma: uptime - - # VPN Infrastructure (on spacey) - headscale: headscale - - # Core Services (on vipy) - vaultwarden: vault - forgejo: forgejo - lnbits: wallet - - # Secondary Services (on vipy) - ntfy_emergency_app: avisame - personal_blog: pablohere - - # Memos (on memos-box) - memos: memos - - # Mempool Block Explorer (on mempool_box, proxied via vipy) - mempool: mempool - - # DATUM Gateway dashboard (on knots_box, proxied via vipy) - datum_gateway: datum - -# Caddy configuration -caddy_sites_dir: /etc/caddy/sites-enabled - -# Service-specific settings shared across playbooks +# Cross-host service settings. +# +# These exist because a value is needed by the service's own role on one host +# AND by a play that runs on the edge host. A role default is invisible to that +# second play, so it cannot live in roles//defaults/. +# +# Everything else that used to be here has moved: +# subdomains, ntfy topic, headscale namespace -> group_vars/all/main.yml +# caddy_sites_dir -> roles/caddy_site/defaults/ +# *.tailscale_hostname -> deleted; inventory already +# holds each box's identity as ansible_host, and an edge play reads it with +# hostvars[''].ansible_host. Three copies of one name is how +# bitcoin_rpc_host ended up labelled "knots_box" while pointing at +# fulcrum-box. service_settings: - ntfy: - topic: alerts - headscale: - namespace: counter-net mempool: - # The frontend port is needed in two places on two different hosts: the - # mempool role deploys it on mempool-box, and the Caddy play proxies to it - # from the edge host. A role default cannot serve the second play, so it - # lives here rather than in roles/mempool/defaults. frontend_port: 8080 bitcoin: - # The P2P port is needed on two hosts: the bitcoin_knots role deploys the - # node on knots-box, and the socket-proxy play publishes the port from the - # edge host. A role default cannot reach that second play. p2p_port: 8333 - tailscale_hostname: knots-box datum_gateway: - # Needed on two hosts: the datum_gateway role deploys on knots-box, while the - # Caddy play (dashboard) and the socket-proxy play (Stratum) both run on the - # edge host. A role default cannot reach either of those plays. api_port: 7152 stratum_port: 23334 - tailscale_hostname: knots-box fulcrum: - # Same shape as mempool: the fulcrum role deploys on fulcrum-box, and the - # socket-proxy play publishes the SSL port from the edge host. A role default - # is invisible to that second play. ssl_port: 50002 - tailscale_hostname: fulcrum-box From 3711421af55210f6d50889a6df5ec45e0c72410b Mon Sep 17 00:00:00 2001 From: counterweight Date: Sun, 13 Sep 2026 21:02:57 +0200 Subject: [PATCH 56/67] ansible: move the cross-host ports to host_vars, delete services_config.yml The four ports were the only entries in services_config.yml with a real justification: each is read twice, by the role that deploys the service on its own box AND by a socket-proxy or Caddy play that runs on the EDGE host and publishes it. A role default is invisible to that second play. But the shape was wrong in two ways. The file had to be named in vars_files: by 30 plays - opt-in configuration that someone will eventually forget - and five role defaults silently interpolated service_settings.*, so bitcoin_knots, fulcrum, datum_gateway and mempool were not self-contained: using any of them without that one vars_file entry broke it. Each port now lives in host_vars//main.yml: host_vars/knots_box_local/main.yml bitcoin_p2p_port, datum_gateway_api_port, datum_gateway_stratum_port host_vars/fulcrum_box_local/main.yml fulcrum_ssl_port host_vars/mempool_box_local/main.yml mempool_frontend_port host_vars auto-loads and outranks role defaults, so the owning role picks the value up with no vars_files at all, and the edge play reads the same single definition as hostvars[''].. The role defaults keep the protocol standard (8333, 50002, ...) so each role still works standalone, with the live deployment's value in host_vars winning. Also fixed a fourth copy of an inventory identity: the mempool Caddy play had "mempool-box:{{ ... }}" hardcoded in the upstream. It now derives the host from hostvars['mempool_box_local'].ansible_host, so inventory is the only place any box's name is written down. services_config.yml is deleted, with 25 more vars_files entries across 19 playbooks. Between this and the previous commit, 87 vars_files entries are gone and every variable in the repo now comes from group_vars/all, host_vars, inventory, a role default, or that service's own *_vars.yml. Verification: an edge-host probe resolves all eight ports and hostnames to byte-identical values to the ones services_config.yml used to supply. Each owning host resolves its own port through host_vars. All 37 playbooks' --list-tasks output is unchanged. The four edge plays that consume these values all check-diff changed=0 - the socket-proxy and Caddy units on vipy are byte-identical, which is the direct proof the rewiring landed on the same values. fulcrum and datum-gateway check-diff exactly as before (ok=28/changed=1 and ok=15/changed=1, both the known timer re-arm). Co-Authored-By: Claude Opus 5 (1M context) --- ansible/host_vars/fulcrum_box_local/main.yml | 6 +++++ ansible/host_vars/knots_box_local/main.yml | 14 +++++++++++ ansible/host_vars/mempool_box_local/main.yml | 6 +++++ ansible/infra/410_disk_usage_alerts.yml | 2 -- ansible/infra/420_system_healthcheck.yml | 2 -- ansible/infra/430_cpu_temp_alerts.yml | 2 -- ansible/infra/920_join_headscale_mesh.yml | 2 -- ansible/roles/bitcoin_knots/defaults/main.yml | 8 ++++--- ansible/roles/datum_gateway/defaults/main.yml | 8 +++++-- ansible/roles/fulcrum/defaults/main.yml | 8 ++++--- ansible/roles/mempool/defaults/main.yml | 8 ++++--- .../deploy_bitcoin_knots_playbook.yml | 6 +---- .../deploy_datum_gateway_playbook.yml | 10 ++------ .../deploy_forgejo_runner_playbook.yml | 2 -- .../forgejo/deploy_forgejo_playbook.yml | 1 - .../fulcrum/deploy_fulcrum_playbook.yml | 6 +---- .../headscale/deploy_headscale_playbook.yml | 1 - .../lnbits/deploy_lnbits_playbook.yml | 1 - .../services/memos/deploy_memos_playbook.yml | 2 -- .../mempool/deploy_mempool_playbook.yml | 6 +---- .../deploy_ntfy_emergency_app_playbook.yml | 1 - .../services/ntfy/deploy_ntfy_playbook.yml | 1 - .../setup_ntfy_uptime_kuma_notification.yml | 1 - .../deploy_personal_blog_playbook.yml | 1 - .../phoenixd/deploy_phoenixd_playbook.yml | 2 -- .../deploy_vaultwarden_playbook.yml | 1 - ansible/services_config.yml | 24 ------------------- 27 files changed, 52 insertions(+), 80 deletions(-) create mode 100644 ansible/host_vars/fulcrum_box_local/main.yml create mode 100644 ansible/host_vars/knots_box_local/main.yml create mode 100644 ansible/host_vars/mempool_box_local/main.yml delete mode 100644 ansible/services_config.yml diff --git a/ansible/host_vars/fulcrum_box_local/main.yml b/ansible/host_vars/fulcrum_box_local/main.yml new file mode 100644 index 0000000..2cf3c65 --- /dev/null +++ b/ansible/host_vars/fulcrum_box_local/main.yml @@ -0,0 +1,6 @@ +# fulcrum-box: the Electrum server. +# +# Read by the fulcrum role here and by the socket-proxy play on the edge host, +# which publishes the port. See host_vars/knots_box_local/main.yml for why this +# lives in host_vars rather than in the role's defaults. +fulcrum_ssl_port: 50002 diff --git a/ansible/host_vars/knots_box_local/main.yml b/ansible/host_vars/knots_box_local/main.yml new file mode 100644 index 0000000..fb15905 --- /dev/null +++ b/ansible/host_vars/knots_box_local/main.yml @@ -0,0 +1,14 @@ +# knots-box: Bitcoin Knots and the DATUM Gateway. +# +# These ports are read twice: by the role that deploys the service here, and by +# the socket-proxy / Caddy plays that run on the EDGE host and publish them. +# A role default is invisible to that second play, which is why these live in +# host_vars rather than roles//defaults/ - the edge play reads them as +# hostvars['knots_box_local']., and the role picks them up automatically +# because host_vars outranks role defaults. +# +# They used to live in services_config.yml, a file 30 plays had to remember to +# name in vars_files: and that four role defaults silently depended on. +bitcoin_p2p_port: 8333 +datum_gateway_api_port: 7152 +datum_gateway_stratum_port: 23334 diff --git a/ansible/host_vars/mempool_box_local/main.yml b/ansible/host_vars/mempool_box_local/main.yml new file mode 100644 index 0000000..f46a1a5 --- /dev/null +++ b/ansible/host_vars/mempool_box_local/main.yml @@ -0,0 +1,6 @@ +# mempool-box: the Mempool block explorer. +# +# Read by the mempool role here and by the Caddy play on the edge host, which +# proxies to it. See host_vars/knots_box_local/main.yml for why this lives in +# host_vars rather than in the role's defaults. +mempool_frontend_port: 8080 diff --git a/ansible/infra/410_disk_usage_alerts.yml b/ansible/infra/410_disk_usage_alerts.yml index 85947d9..b286add 100644 --- a/ansible/infra/410_disk_usage_alerts.yml +++ b/ansible/infra/410_disk_usage_alerts.yml @@ -14,8 +14,6 @@ - name: Deploy Disk Usage Monitoring hosts: managed become: yes - vars_files: - - ../services_config.yml vars: disk_usage_threshold_percent: 80 diff --git a/ansible/infra/420_system_healthcheck.yml b/ansible/infra/420_system_healthcheck.yml index a69b456..9813d0c 100644 --- a/ansible/infra/420_system_healthcheck.yml +++ b/ansible/infra/420_system_healthcheck.yml @@ -14,8 +14,6 @@ - name: Deploy System Healthcheck Monitoring hosts: managed become: yes - vars_files: - - ../services_config.yml vars: healthcheck_interval_seconds: 60 # Send healthcheck every 60 seconds (1 minute) diff --git a/ansible/infra/430_cpu_temp_alerts.yml b/ansible/infra/430_cpu_temp_alerts.yml index 048f216..2e00bdb 100644 --- a/ansible/infra/430_cpu_temp_alerts.yml +++ b/ansible/infra/430_cpu_temp_alerts.yml @@ -14,8 +14,6 @@ - name: Deploy CPU Temperature Monitoring hosts: hypervisor become: yes - vars_files: - - ../services_config.yml vars: temp_threshold_celsius: 80 diff --git a/ansible/infra/920_join_headscale_mesh.yml b/ansible/infra/920_join_headscale_mesh.yml index 3a8641c..4decb5d 100644 --- a/ansible/infra/920_join_headscale_mesh.yml +++ b/ansible/infra/920_join_headscale_mesh.yml @@ -1,8 +1,6 @@ - name: Join machine to headscale mesh network hosts: managed become: yes - vars_files: - - ../services_config.yml vars: headscale_host_name: "spacey" headscale_subdomain: "{{ subdomains.headscale }}" diff --git a/ansible/roles/bitcoin_knots/defaults/main.yml b/ansible/roles/bitcoin_knots/defaults/main.yml index cd9d9c1..3a12ad1 100644 --- a/ansible/roles/bitcoin_knots/defaults/main.yml +++ b/ansible/roles/bitcoin_knots/defaults/main.yml @@ -15,9 +15,11 @@ bitcoin_conf_dir: /etc/bitcoin # Network bitcoin_rpc_port: 8332 -# Shared with the socket-proxy play on the edge host, so it lives in -# services_config.yml rather than only here. -bitcoin_p2p_port: "{{ service_settings.bitcoin.p2p_port }}" +# The edge host's socket-proxy/Caddy play needs this too, and a role default is +# invisible outside this role. The authoritative value for the live deployment is +# in host_vars/knots_box_local/main.yml, which outranks this; the value here is the +# protocol standard, so the role still works standalone. +bitcoin_p2p_port: 8333 bitcoin_rpc_bind: "0.0.0.0" # Build options diff --git a/ansible/roles/datum_gateway/defaults/main.yml b/ansible/roles/datum_gateway/defaults/main.yml index 3a8cac4..b9829f9 100644 --- a/ansible/roles/datum_gateway/defaults/main.yml +++ b/ansible/roles/datum_gateway/defaults/main.yml @@ -14,8 +14,12 @@ datum_gateway_log_dir: /var/log/datum-gateway datum_gateway_bin_path: /usr/local/bin/datum_gateway # Ports -datum_gateway_stratum_port: "{{ service_settings.datum_gateway.stratum_port }}" -datum_gateway_api_port: "{{ service_settings.datum_gateway.api_port }}" +# The edge host's socket-proxy/Caddy play needs this too, and a role default is +# invisible outside this role. The authoritative value for the live deployment is +# in host_vars/knots_box_local/main.yml, which outranks this; the value here is the +# protocol standard, so the role still works standalone. +datum_gateway_stratum_port: 23334 +datum_gateway_api_port: 7152 # Stratum settings datum_vardiff_min: 524288 # Minimum share difficulty (must be power of 2; OCEAN floor overrides if higher) diff --git a/ansible/roles/fulcrum/defaults/main.yml b/ansible/roles/fulcrum/defaults/main.yml index 0d5c190..cc778d4 100644 --- a/ansible/roles/fulcrum/defaults/main.yml +++ b/ansible/roles/fulcrum/defaults/main.yml @@ -25,9 +25,11 @@ bitcoin_rpc_port: 8332 # Bitcoin Knots RPC port # Network - Fulcrum server fulcrum_tcp_port: 50001 -# Shared with the socket-proxy play on the edge host, so it lives in -# services_config.yml rather than only here. -fulcrum_ssl_port: "{{ service_settings.fulcrum.ssl_port }}" +# The edge host's socket-proxy/Caddy play needs this too, and a role default is +# invisible outside this role. The authoritative value for the live deployment is +# in host_vars/fulcrum_box_local/main.yml, which outranks this; the value here is the +# protocol standard, so the role still works standalone. +fulcrum_ssl_port: 50002 # Binding address for Fulcrum TCP/SSL server: # - "127.0.0.1" = localhost only (use when Caddy is on the same box) # - "0.0.0.0" = all interfaces (use when Caddy is on a different box) diff --git a/ansible/roles/mempool/defaults/main.yml b/ansible/roles/mempool/defaults/main.yml index 586b004..2f983c6 100644 --- a/ansible/roles/mempool/defaults/main.yml +++ b/ansible/roles/mempool/defaults/main.yml @@ -22,9 +22,11 @@ fulcrum_tls: "false" mempool_network: "mainnet" # Container ports (internal) -# Sourced from services_config.yml: the Caddy play on the edge host needs this -# too, and a role default is not visible outside this role. -mempool_frontend_port: "{{ service_settings.mempool.frontend_port }}" +# The edge host's socket-proxy/Caddy play needs this too, and a role default is +# invisible outside this role. The authoritative value for the live deployment is +# in host_vars/mempool_box_local/main.yml, which outranks this; the value here is the +# protocol standard, so the role still works standalone. +mempool_frontend_port: 8080 mempool_backend_port: 8999 # MariaDB settings diff --git a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml index 709cb30..0ca1e13 100644 --- a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +++ b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml @@ -9,8 +9,6 @@ - name: Build and Deploy Bitcoin Knots from Source hosts: bitcoin become: yes - vars_files: - - ../../services_config.yml vars: # Preserves the push URL this check has been reporting to. The role knows # nothing about Uptime Kuma — this is just "a URL that accepts a ping". @@ -21,8 +19,6 @@ - name: Setup public Bitcoin P2P forwarding on the edge host hosts: edge become: yes - vars_files: - - ../../services_config.yml tasks: - name: Expose Bitcoin P2P through a socket proxy ansible.builtin.include_role: @@ -30,7 +26,7 @@ vars: socket_proxy_name: bitcoin-p2p socket_proxy_description: "Bitcoin P2P" - socket_proxy_listen_port: "{{ service_settings.bitcoin.p2p_port }}" + socket_proxy_listen_port: "{{ hostvars['knots_box_local'].bitcoin_p2p_port }}" socket_proxy_upstream_host: "{{ hostvars['knots_box_local'].ansible_host }}" socket_proxy_documentation: "https://github.com/bitcoin/bitcoin" socket_proxy_free_bind: true diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index ebb9633..84de178 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -10,8 +10,6 @@ - name: Deploy DATUM Gateway on the bitcoin host hosts: bitcoin become: yes - vars_files: - - ../../services_config.yml vars: # Preserves the push URL this check reports to. The role knows nothing about # Uptime Kuma — this is just "a URL that accepts a ping". @@ -22,8 +20,6 @@ - name: Configure Caddy reverse proxy for the DATUM Gateway dashboard on the edge host hosts: edge become: yes - vars_files: - - ../../services_config.yml tasks: - name: Publish the DATUM Gateway dashboard through Caddy ansible.builtin.include_role: @@ -31,7 +27,7 @@ vars: caddy_site_name: datum-gateway caddy_site_domain: "{{ subdomains.datum_gateway }}.{{ root_domain }}" - caddy_site_upstream: "{{ hostvars['knots_box_local'].ansible_host }}:{{ service_settings.datum_gateway.api_port }}" + caddy_site_upstream: "{{ hostvars['knots_box_local'].ansible_host }}:{{ hostvars['knots_box_local'].datum_gateway_api_port }}" caddy_site_resolvers: "100.100.100.100" caddy_site_basic_auth: - user: "{{ datum_dashboard_username }}" @@ -47,8 +43,6 @@ - name: Setup public Stratum port forwarding on the edge host hosts: edge become: yes - vars_files: - - ../../services_config.yml tasks: - name: Expose the DATUM Stratum port through a socket proxy ansible.builtin.include_role: @@ -56,7 +50,7 @@ vars: socket_proxy_name: datum-stratum socket_proxy_description: "DATUM Stratum" - socket_proxy_listen_port: "{{ service_settings.datum_gateway.stratum_port }}" + socket_proxy_listen_port: "{{ hostvars['knots_box_local'].datum_gateway_stratum_port }}" socket_proxy_upstream_host: "{{ hostvars['knots_box_local'].ansible_host }}" # Matches the UFW comment already on the edge host; the derived default # would say "DATUM Stratum" and rewrite the rule. diff --git a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml index 9195087..031f081 100644 --- a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml +++ b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml @@ -2,8 +2,6 @@ - name: Install Forgejo Runner on Debian 13 hosts: ci_runner become: yes - vars_files: - - ../../services_config.yml vars: # Preserves the push URL this host has been reporting to all along, so the # move to a role changes no behaviour. The role itself knows nothing about diff --git a/ansible/services/forgejo/deploy_forgejo_playbook.yml b/ansible/services/forgejo/deploy_forgejo_playbook.yml index a929d42..04d7041 100644 --- a/ansible/services/forgejo/deploy_forgejo_playbook.yml +++ b/ansible/services/forgejo/deploy_forgejo_playbook.yml @@ -2,7 +2,6 @@ hosts: edge become: yes vars_files: - - ../../services_config.yml - ./forgejo_vars.yml vars: forgejo_subdomain: "{{ subdomains.forgejo }}" diff --git a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml index 6b8e207..8927aba 100644 --- a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml +++ b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml @@ -5,8 +5,6 @@ - name: Deploy Fulcrum Electrum Server hosts: electrum become: yes - vars_files: - - ../../services_config.yml vars: # Preserves the push URL this check has been configured with. The role knows # nothing about Uptime Kuma — this is just "a URL that accepts a ping". @@ -17,8 +15,6 @@ - name: Setup public Fulcrum SSL forwarding on the edge host hosts: edge become: yes - vars_files: - - ../../services_config.yml tasks: - name: Expose Fulcrum SSL through a socket proxy ansible.builtin.include_role: @@ -26,5 +22,5 @@ vars: socket_proxy_name: fulcrum-ssl socket_proxy_description: "Fulcrum SSL" - socket_proxy_listen_port: "{{ service_settings.fulcrum.ssl_port }}" + socket_proxy_listen_port: "{{ hostvars['fulcrum_box_local'].fulcrum_ssl_port }}" socket_proxy_upstream_host: "{{ hostvars['fulcrum_box_local'].ansible_host }}" diff --git a/ansible/services/headscale/deploy_headscale_playbook.yml b/ansible/services/headscale/deploy_headscale_playbook.yml index 4308b1e..11c967e 100644 --- a/ansible/services/headscale/deploy_headscale_playbook.yml +++ b/ansible/services/headscale/deploy_headscale_playbook.yml @@ -2,7 +2,6 @@ hosts: vpn_control become: no vars_files: - - ../../services_config.yml - ./headscale_vars.yml vars: headscale_subdomain: "{{ subdomains.headscale }}" diff --git a/ansible/services/lnbits/deploy_lnbits_playbook.yml b/ansible/services/lnbits/deploy_lnbits_playbook.yml index 3aa7a95..5d0c21d 100644 --- a/ansible/services/lnbits/deploy_lnbits_playbook.yml +++ b/ansible/services/lnbits/deploy_lnbits_playbook.yml @@ -2,7 +2,6 @@ hosts: edge become: yes vars_files: - - ../../services_config.yml - ./lnbits_vars.yml vars: lnbits_subdomain: "{{ subdomains.lnbits }}" diff --git a/ansible/services/memos/deploy_memos_playbook.yml b/ansible/services/memos/deploy_memos_playbook.yml index e078d84..756ab4f 100644 --- a/ansible/services/memos/deploy_memos_playbook.yml +++ b/ansible/services/memos/deploy_memos_playbook.yml @@ -2,7 +2,6 @@ hosts: memos become: yes vars_files: - - ../../services_config.yml - ./memos_vars.yml vars: memos_subdomain: "{{ subdomains.memos }}" @@ -158,7 +157,6 @@ hosts: edge become: yes vars_files: - - ../../services_config.yml - ./memos_vars.yml vars: memos_subdomain: "{{ subdomains.memos }}" diff --git a/ansible/services/mempool/deploy_mempool_playbook.yml b/ansible/services/mempool/deploy_mempool_playbook.yml index c2c8f4f..76a3eb0 100644 --- a/ansible/services/mempool/deploy_mempool_playbook.yml +++ b/ansible/services/mempool/deploy_mempool_playbook.yml @@ -2,8 +2,6 @@ - name: Deploy Mempool Block Explorer with Docker hosts: mempool become: yes - vars_files: - - ../../services_config.yml vars: # Preserves the three push URLs these checks have been reporting to all # along, so the move to a role changes no behaviour. The role knows nothing @@ -19,8 +17,6 @@ - name: Configure Caddy reverse proxy for Mempool on the edge host hosts: edge become: yes - vars_files: - - ../../services_config.yml vars: mempool_domain: "{{ subdomains.mempool }}.{{ root_domain }}" tasks: @@ -30,5 +26,5 @@ vars: caddy_site_name: mempool caddy_site_domain: "{{ mempool_domain }}" - caddy_site_upstream: "mempool-box:{{ service_settings.mempool.frontend_port }}" + caddy_site_upstream: "{{ hostvars['mempool_box_local'].ansible_host }}:{{ hostvars['mempool_box_local'].mempool_frontend_port }}" caddy_site_resolvers: "100.100.100.100" diff --git a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml index 1b53732..c759592 100644 --- a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml +++ b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml @@ -2,7 +2,6 @@ hosts: edge become: yes vars_files: - - ../../services_config.yml - ./ntfy_emergency_app_vars.yml vars: ntfy_emergency_app_subdomain: "{{ subdomains.ntfy_emergency_app }}" diff --git a/ansible/services/ntfy/deploy_ntfy_playbook.yml b/ansible/services/ntfy/deploy_ntfy_playbook.yml index d6253b5..afd8e69 100644 --- a/ansible/services/ntfy/deploy_ntfy_playbook.yml +++ b/ansible/services/ntfy/deploy_ntfy_playbook.yml @@ -2,7 +2,6 @@ hosts: monitoring become: yes vars_files: - - ../../services_config.yml - ./ntfy_vars.yml vars: ntfy_subdomain: "{{ subdomains.ntfy }}" diff --git a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml index 861bfa5..7588618 100644 --- a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml +++ b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml @@ -15,7 +15,6 @@ hosts: monitoring become: no vars_files: - - ../../services_config.yml - ./ntfy_vars.yml vars: diff --git a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml index 968f423..5e3780d 100644 --- a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml +++ b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml @@ -2,7 +2,6 @@ hosts: edge become: yes vars_files: - - ../../services_config.yml - ./personal_blog_vars.yml vars: personal_blog_subdomain: "{{ subdomains.personal_blog }}" diff --git a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml index 45bea3c..0df1223 100644 --- a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml +++ b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml @@ -4,8 +4,6 @@ - name: Deploy phoenixd on the edge host hosts: edge become: yes - vars_files: - - ../../services_config.yml vars: # phoenixd's health check has never reported anywhere since the Uptime Kuma # decommissioning — its systemd Environment= was left empty. Leaving it empty diff --git a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml index 987fe27..d03363c 100644 --- a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml +++ b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml @@ -2,7 +2,6 @@ hosts: edge become: yes vars_files: - - ../../services_config.yml - ./vaultwarden_vars.yml vars: vaultwarden_subdomain: "{{ subdomains.vaultwarden }}" diff --git a/ansible/services_config.yml b/ansible/services_config.yml deleted file mode 100644 index 531b824..0000000 --- a/ansible/services_config.yml +++ /dev/null @@ -1,24 +0,0 @@ -# Cross-host service settings. -# -# These exist because a value is needed by the service's own role on one host -# AND by a play that runs on the edge host. A role default is invisible to that -# second play, so it cannot live in roles//defaults/. -# -# Everything else that used to be here has moved: -# subdomains, ntfy topic, headscale namespace -> group_vars/all/main.yml -# caddy_sites_dir -> roles/caddy_site/defaults/ -# *.tailscale_hostname -> deleted; inventory already -# holds each box's identity as ansible_host, and an edge play reads it with -# hostvars[''].ansible_host. Three copies of one name is how -# bitcoin_rpc_host ended up labelled "knots_box" while pointing at -# fulcrum-box. -service_settings: - mempool: - frontend_port: 8080 - bitcoin: - p2p_port: 8333 - datum_gateway: - api_port: 7152 - stratum_port: 23334 - fulcrum: - ssl_port: 50002 From e8eae0c3c5a57cba3a81a8d65ace29c852c878cb Mon Sep 17 00:00:00 2001 From: counterweight Date: Sun, 13 Sep 2026 21:24:09 +0200 Subject: [PATCH 57/67] ansible: add site.yml, and rename the monitoring group off the host's name site.yml is a TABLE OF CONTENTS, not a second source of truth. It is 25 import_playbook: lines and comments - no `hosts:`, no `roles:`. Which hosts get what stays on the `hosts:` line inside each playbook, exactly where it already was; nothing moved. Every role is already wrapped in a thin playbook carrying its own `hosts:` line, so there is no roles-vs-playbooks split to reconcile: from here everything is a playbook. What it buys: What runs on a host? ansible-playbook site.yml --limit --list-hosts Who gets thing Y? the `hosts:` line in Y's own playbook What is a host? ansible-inventory --graph Note --list-hosts, not --list-tasks: the latter prints every play regardless of --limit, so it will happily show you the bitcoin play under memos-box. Nine playbooks are deliberately excluded and the file names every one with a reason, so it accounts for all of them: the three infra/4xx monitoring plays (still assert on the removed Uptime Kuma credentials and fail immediately), 910_docker (says `hosts: managed`, but Docker is on 5 of 11 managed hosts and those 5 are exactly the ones that need it - running it installs Docker on the Bitcoin node and the hypervisor), two nodito one-shots, the Kuma notification setup, and two deliberate manual actions. Writing it surfaced an inventory collision. There is a HOST named `monitoring` in [vps] AND a group [monitoring], so Ansible warned and resolved `hosts: monitoring` to the host: [WARNING]: Found both group and host with same name: monitoring The group is renamed to [observability]; the host keeps its name. [caddy:children] and the two ntfy playbooks follow. Behaviour is unchanged - `hosts: monitoring` already resolved to the host - but the ambiguity is gone and the warning with it. Verified: inventory graph is warning-free, site.yml passes --syntax-check, and per-host play counts are identical before and after the rename. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/inventory.ini | 12 ++- .../services/ntfy/deploy_ntfy_playbook.yml | 2 +- .../setup_ntfy_uptime_kuma_notification.yml | 2 +- ansible/site.yml | 82 +++++++++++++++++++ 4 files changed, 92 insertions(+), 6 deletions(-) create mode 100644 ansible/site.yml diff --git a/ansible/inventory.ini b/ansible/inventory.ini index 3d4a352..73ac21e 100644 --- a/ansible/inventory.ini +++ b/ansible/inventory.ini @@ -2,6 +2,7 @@ vipy ansible_host=167.172.107.33 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua watchtower ansible_host=164.92.239.72 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua spacey ansible_host=64.227.112.128 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua +monitoring ansible_host=64.226.70.190 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua [nodito_host] nodito ansible_host=192.168.1.139 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua @@ -15,7 +16,6 @@ memos_box_local ansible_host=memos-box lan_ip=192.168.1.145 ansible_user=counter forgejo_runner_local ansible_host=forgejo-runner-box lan_ip=192.168.1.132 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua arbret_staging_local ansible_host=arbret-staging-box lan_ip=192.168.1.147 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua small_backups_local ansible_host=small-backups-box lan_ip=192.168.1.131 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -nonkeiwaisi_local ansible_host=nonkeiwaisi-box lan_ip=192.168.1.151 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua # Local connection to laptop: this assumes you're running ansible commands from your personal laptop [lapy] @@ -27,8 +27,12 @@ prd-arbret ansible_host=167.99.242.62 ansible_user=counterweight ansible_port=22 [edge] vipy -[monitoring] -watchtower +# The group is `observability`, NOT `monitoring` — there is a HOST named +# `monitoring` on line 5, and a group with the same name makes `hosts: monitoring` +# ambiguous. Ansible resolved it to the host and warned: +# [WARNING]: Found both group and host with same name: monitoring +[observability] +monitoring [vpn_control] spacey @@ -64,7 +68,7 @@ nodito_vms # Hosts that run Caddy and therefore have /etc/caddy/sites-enabled. [caddy:children] edge -monitoring +observability vpn_control [backup_store] diff --git a/ansible/services/ntfy/deploy_ntfy_playbook.yml b/ansible/services/ntfy/deploy_ntfy_playbook.yml index afd8e69..88312d8 100644 --- a/ansible/services/ntfy/deploy_ntfy_playbook.yml +++ b/ansible/services/ntfy/deploy_ntfy_playbook.yml @@ -1,5 +1,5 @@ - name: Deploy ntfy and configure Caddy reverse proxy - hosts: monitoring + hosts: observability become: yes vars_files: - ./ntfy_vars.yml diff --git a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml index 7588618..97d6c9c 100644 --- a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml +++ b/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml @@ -12,7 +12,7 @@ # What was being monitored: archive/uptime_kuma/MONITORS.md # ═════════════════════════════════════════════════════════════════════════════ - name: Setup ntfy as Uptime Kuma Notification Channel - hosts: monitoring + hosts: observability become: no vars_files: - ./ntfy_vars.yml diff --git a/ansible/site.yml b/ansible/site.yml new file mode 100644 index 0000000..fd042ce --- /dev/null +++ b/ansible/site.yml @@ -0,0 +1,82 @@ +--- +# Everything, in the order it has to happen. +# +# This file is a TABLE OF CONTENTS, not a second source of truth. It says what +# runs and in what order. It does NOT say which hosts get what — that stays on +# the `hosts:` line inside each playbook, exactly where it is today. Nothing +# moves; this file only makes the set readable in one place. +# +# What runs on a host? ansible-playbook site.yml --limit --list-hosts +# Who gets thing Y? the `hosts:` line in Y's own playbook +# What is a host? ansible-inventory --graph +# +# Run a slice with --limit, or run one playbook directly as before. Nothing here +# changes how any individual playbook behaves. + +# ── Baseline: every managed machine ───────────────────────────────────────── +- import_playbook: infra/01_user_and_access_setup_playbook.yml +- import_playbook: infra/02_firewall_and_fail2ban_playbook.yml +- import_playbook: infra/900_install_rsync.yml +- import_playbook: infra/920_join_headscale_mesh.yml +# 910_docker says `hosts: managed`, but only 5 of 11 managed hosts have or need +# Docker. Left out until it has a [docker] group — see the note in PLAN_7. + +# ── The hypervisor ────────────────────────────────────────────────────────── +- import_playbook: infra/nodito/31_proxmox_community_repos_playbook.yml +- import_playbook: infra/nodito/32_zfs_pool_setup_playbook.yml +- import_playbook: infra/nodito/34_nut_ups_setup_playbook.yml + +# ── Reverse proxy, before anything that registers a vhost ─────────────────── +- import_playbook: services/caddy_playbook.yml + +# ── Services ──────────────────────────────────────────────────────────────── +- import_playbook: services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +- import_playbook: services/fulcrum/deploy_fulcrum_playbook.yml +- import_playbook: services/datum-gateway/deploy_datum_gateway_playbook.yml +- import_playbook: services/mempool/deploy_mempool_playbook.yml +- import_playbook: services/memos/deploy_memos_playbook.yml +- import_playbook: services/forgejo-runner/deploy_forgejo_runner_playbook.yml +- import_playbook: services/phoenixd/deploy_phoenixd_playbook.yml +- import_playbook: services/headscale/deploy_headscale_playbook.yml +- import_playbook: services/vaultwarden/deploy_vaultwarden_playbook.yml +- import_playbook: services/forgejo/deploy_forgejo_playbook.yml +- import_playbook: services/lnbits/deploy_lnbits_playbook.yml +- import_playbook: services/ntfy/deploy_ntfy_playbook.yml +- import_playbook: services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml +- import_playbook: services/personal-blog/deploy_personal_blog_playbook.yml + +# ── Backups: each source dumps itself, the box pulls ──────────────────────── +- import_playbook: services/headscale/setup_backup_headscale.yml +- import_playbook: services/vaultwarden/setup_backup_vaultwarden.yml +- import_playbook: services/forgejo/setup_backup_forgejo.yml +- import_playbook: services/lnbits/setup_backup_lnbits.yml +- import_playbook: services/memos/setup_backup_memos.yml +- import_playbook: playbooks/backups.yml + +# Deliberately not here. Every playbook in the repo is either imported above or +# listed below, so this file accounts for all of them: +# +# infra/410_disk_usage_alerts.yml assert on the Uptime Kuma credentials that +# infra/420_system_healthcheck.yml were removed from the vault, so they fail +# infra/430_cpu_temp_alerts.yml before doing anything. What they deploy IS +# running on the boxes - same state the two +# nodito playbooks were in before 0f03c50. +# Add them once they are de-Kuma'd. +# +# infra/910_docker_playbook.yml says `hosts: managed`, but Docker is on 5 +# of 11 managed hosts and those 5 are exactly +# the ones that need it. Running it would +# install Docker on the Bitcoin node and the +# hypervisor. Needs a [docker] group first. +# +# infra/nodito/30_proxmox_bootstrap one-shot: bare-metal bootstrap, run once +# infra/nodito/33_..._cloud_template one-shot: builds the VM template +# +# services/ntfy/setup_ntfy_uptime_ creates a notification channel INSIDE +# kuma_notification.yml Uptime Kuma. Kuma-specific tooling, not +# a deployment. +# +# services/vaultwarden/disable_ deliberate manual actions, not convergence +# vaultwarden_sign_ups_playbook.yml +# services/personal-blog/setup_ +# deploy_alias_lapy.yml From fa9f7d10cd5598db72bfbf6708023e1639baa888 Mon Sep 17 00:00:00 2001 From: counterweight Date: Sun, 13 Sep 2026 22:29:03 +0200 Subject: [PATCH 58/67] gatus: deploy on prd-monitoring, behind Caddy basic auth First step of replacing Uptime Kuma and ntfy. Gatus runs on the new observability VPS, fronted by Caddy at status.contrapeso.xyz. Deployed as the upstream container image, not built from source. The role did build from source first - their Dockerfile is a bare `CGO_ENABLED=0 go build`, the Vue dashboard is compiled in via `//go:embed static` in web/static.go, and CGO can stay off because the sqlite driver is pure-Go modernc.org/sqlite - but that produces a binary upstream never ran, and it meant compiling the AWS SDK and gRPC on the smallest box in the estate. That load was heavy enough that unrelated Ansible tasks timed out while it ran. The cost of the container is a daemon on the machine whose job is to notice when everything else breaks; that trade is made deliberately and is written down in the role README. Pinned by DIGEST, not tag. A tag is mutable - v5.36.0 can be repushed - so pinning it alone is a weaker promise than it looks: gatus_image: "ghcr.io/twin/gatus@sha256:c5f210d0..." `docker compose pull` now either fetches exactly the reviewed image or fails. gatus_version is kept beside it only so a human can read the release; the two move together. The image is FROM scratch, so it has no /etc/passwd and its default user is root. The container runs as 10001:10001 with the host data dir owned to match, plus read_only, cap_drop ALL, and no-new-privileges. NET_RAW is added back only when gatus_allow_icmp, so the capability for icmp:// checks is a visible grant rather than something inherited from running as root. Config is a DIRECTORY, not a file. Gatus merges every *.yaml under GATUS_CONFIG_PATH - maps deep-merge, lists append - so the role owns 00-base.yaml (web, storage, ui, alerting, security) and each service will drop its own file into endpoints/, the same shape as caddy_site. A primitive defined twice is ambiguous and upstream refuses it, so anything that is not a list lives in the base file and nowhere else. Two bugs the deploy caught: * Gatus panics on a config with no endpoints ("configuration should contain at least one endpoint or suite"), so "install now, add endpoints later" is not a valid state. The role ships endpoints/00-self.yaml checking its own /health. Less circular than it looks: it proves the directory merged, the listener serves, and storage accepted a write. * web.address was carried over from the systemd design as 127.0.0.1. Inside a container that is the CONTAINER's loopback, which docker-proxy cannot reach - gatus came up healthy, self-check passing, while every connection to the published port was refused. It now always binds 0.0.0.0 inside the container; the isolation comes from publishing to 127.0.0.1 on the host. Auth is done at the edge, NOT with Gatus's own security.basic. Reading api/api.go, that middleware protects exactly four routes - the statuses endpoints. Everything else is registered on the unprotected router, including /api/v1/config, every badge, and /api/v1/endpoints/:key/uptimes/:duration and .../response-times/:duration/history, which return real data to anyone who can guess a key ("_"). Verified against the live instance: all seven routes returned 200 unauthenticated, and /uptimes/24h returned "1.000000". So the vhost uses caddy_site_body with a path carve-out rather than caddy_site_basic_auth, which has no way to exempt a path. The external-endpoint push API must NOT sit behind basic auth: it authenticates with `Authorization: Bearer `, and basic auth wants the same header. It is not unauthenticated - the handler 401s on a missing prefix, an empty token, or a token that does not match that endpoint's own. Verified end to end. All seven previously-open routes now 401. The push path distinguishes cleanly: POST with no auth gets Gatus's own "invalid Authorization header" with NO WWW-Authenticate; POST with a bogus Bearer gets 404 (key looked up, no external endpoints yet); GET on the same path gets Caddy's 401 with WWW-Authenticate: Basic, so the exemption is scoped to POST alone. The self-check still passes because it polls localhost inside the container and never traverses Caddy. The host itself was rebuilt from scratch: 01 (ok=9 changed=8), 02 (ok=12 changed=6), 910_docker (--limit, since that playbook still wrongly claims all of `managed` needs Docker), caddy (ok=13 changed=8), gatus (ok=18 changed=2). Not done here: gatus_alerting is still {} - valid, and every condition is evaluated and recorded, there is just nowhere to shout until a provider is chosen to replace ntfy. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/main.yml | 3 +- ansible/group_vars/all/vault.yml | 393 +++++++++--------- ansible/roles/gatus/README.md | 110 +++++ ansible/roles/gatus/defaults/main.yml | 83 ++++ ansible/roles/gatus/handlers/main.yml | 5 + ansible/roles/gatus/tasks/configure.yml | 54 +++ ansible/roles/gatus/tasks/main.yml | 5 + ansible/roles/gatus/tasks/service.yml | 35 ++ ansible/roles/gatus/templates/config.yaml.j2 | 56 +++ .../gatus/templates/docker-compose.yml.j2 | 44 ++ .../gatus/templates/endpoint-self.yaml.j2 | 25 ++ .../services/gatus/deploy_gatus_playbook.yml | 83 ++++ 12 files changed, 702 insertions(+), 194 deletions(-) create mode 100644 ansible/roles/gatus/README.md create mode 100644 ansible/roles/gatus/defaults/main.yml create mode 100644 ansible/roles/gatus/handlers/main.yml create mode 100644 ansible/roles/gatus/tasks/configure.yml create mode 100644 ansible/roles/gatus/tasks/main.yml create mode 100644 ansible/roles/gatus/tasks/service.yml create mode 100644 ansible/roles/gatus/templates/config.yaml.j2 create mode 100644 ansible/roles/gatus/templates/docker-compose.yml.j2 create mode 100644 ansible/roles/gatus/templates/endpoint-self.yaml.j2 create mode 100644 ansible/services/gatus/deploy_gatus_playbook.yml diff --git a/ansible/group_vars/all/main.yml b/ansible/group_vars/all/main.yml index 1dc3541..9cf3143 100644 --- a/ansible/group_vars/all/main.yml +++ b/ansible/group_vars/all/main.yml @@ -25,7 +25,8 @@ backup_pull_public_key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIOfIixKMhA9z+Nvyx6T # vars_files: - a file everyone must opt into is a file someone will forget. # ───────────────────────────────────────────────────────────────────────────── subdomains: - # Monitoring (watchtower) + # Monitoring + gatus: status ntfy: ntfy # Uptime Kuma IS still running and this subdomain DOES resolve # (164.92.239.72, HTTP 302). Only the Ansible code and the vault credentials diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index a619094..e8af624 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,194 +1,201 @@ $ANSIBLE_VAULT;1.1;AES256 -66646566313430366435336438316665363536356431633739373931326363363137653964333765 -6438363638393236306464663136663061386361363432390a623836393662636238626337373766 -39313636386137363334313465373637613363313634363230643032386132366533323934393933 -3033616238356165630a613765643762336161663863393163616636636163376537396638666139 -62376334383839373131646334383039396661396262633537333863613133346363303133353765 -61323462646434313539633833616162376637613436626365336265333863356534626339313661 -35376634386233653964353437373838343066323035366263663738643032646631633236663066 -34383535333961313439386462326264616535366434343032633639646631373930353038356330 -37626465653362366666316363333663343734663939393637373234303935363136366636306534 -65356362313662356332626530363563333337613661626531326138633936316137626334616162 -30336664323436653431393061343935386461383130356437646361333431313938393236363633 -39303063653936353736363332386366623565313861613838303735356530663138323935323066 -66613134653235353364366263343138666639396430306439396633393433636236363363393164 -66326330313530393064656331386233363466386336313430333139643830626535653538353331 -62643631366536653137393733666238386230383634356532663563343765306334383331333161 -35313336643266653231333034626139636133653236323235346238323335663934626133616664 -65326333323334663265303933353035353739323062323962396661666161333236666133343235 -65616562346131613136363932353632343935353139386234656138303261636230333264363230 -33333933623965343862393065313238306635636361373237626233353634373737336532626431 -30636163363835366330326163616539303031393561313538393338323166613739663039396566 -61613265653633386338663837396135356563353439336534633064363330303863333166393966 -35366337353636366336346436633734363833363133366561333835306433333138666134306462 -38626333663534386237303739646664663363303866656433313733393832316533383136316233 -66363234656664346465636562643832323266346631653262373166343036363532646533313636 -63303538656662383430623365383133313665393163333832626361386663346132313163313061 -39613732366435343930346639643161656165656538653531653036393764663962363832663861 -33333031353234646239386534623536376131306261653165393634383762383265313136326636 -33333430323030636164346433316337663137643133626639303536323735623637333031356633 -36613232313330636636633266636461616463376463366234303630613966363738633133636263 -64646532333764343861633935633737323639333237643831353339636235316533613266653865 -36333535333365333436653638663364386437613136313239376538323439613965343339656461 -39616239346435343665333831363134643762646532393366366533623461353964383263356630 -31373239666266613133323333323336363739346236656261323365363664303630663934653863 -34636534376633333331356162393339653762303837363765313633333535386164386439386266 -39303336646465366336316263343461613336303930653963613866663738333433326565643736 -61303631656132653064303439646637373734313364303135643963353064363061663835663632 -31623264383238323734373732393733343336323630366439303466316238336566623666303433 -34396633653762303730323334356166303134376535316131333266303438313562306135343233 -30303136333730626561656639623866666663646131383666613633663966636462663031626639 -65383732313561376261333534643937383165616137656232623664386232353337343563333262 -34633364323839353136616636366462666637313438396661303663316263353636363163633236 -64326265623262323735396637663538626139366165643437333233643830313962663735373063 -37363664363432326637386335386432616135666162363431356561333635616134626136316365 -38323539663966643262656635356334386236393538333034303565393439353931663466373966 -31376466393363396535396263336338306239613961356130613335316132313335636634353461 -35343830633036653337356131386538396139323265653230396664653533636239613239356639 -66316332323538656166393037636563626338386235633230623962373731386230383733366663 -64383064623362626639383030396339623138373935373539643335366531393635316137323266 -61383065353766626463636338383563366532323439616333623730386630656163343536646562 -63343538616139363933316266363531656661633532353765366534383735366664616565623030 -33373561386537613737653764326439366632626537346464363132303133393963313966373163 -32623239326631333338303736613465616435346535336332633163313439656232626239653836 -33343638643562626338353137346365386431663334336231656435373238323535323733623461 -37356533613738663764656466396632366430316433623334666665363463386131303832326564 -62613163366661346435343031373433343236643839396536383966363863393836623265393135 -62303838393539366166383137343737383432333235633230363036656466373837396462393564 -61353535623738616239663535333666323839666133346336646533633735623639376332656464 -35633839633665313033353834383530336164346233643033373336656265363266363534316465 -36346536383535333138313761356561323661643163393831343962633036373265643938336533 -63623135386630356262386563353634623439383631633866366661336336613038623439326264 -39353933643233323863343839656634373732363661373966313839363130646533323737646433 -33626634643964336236646233376235376230616435323930323935373730633663393435646331 -64613839366663663933626537323339613363643839313632313034633135643166323235356632 -66643437353130333831306530653834386137386632626363646563343631333235653065353730 -38323363373665623961306530316562353165656138393134653933303239663636643464306136 -66303234353537633062663735653232366639373030336330336635323065643935323936396333 -64316661323263386664323031333030326534306566333934306361353233373536356435326338 -33303561326233656233366431653237333763326434313162663165633637636262343130636133 -61366635663730653539386366353237633438323966633961303032366136336630623436393035 -62633463313338353432323939343239623630383565636162616139613664613863396165643736 -35376462333938656432373133383838623035653131646366373239306262323539313439343166 -66323830616362363263636663333566386239353765343830343465353265333262313331343434 -30313339303064313263633864623261616130343530336134333661333931663564323461353034 -64663737393430383233633330306565363230383761346362663530646238363631383062646230 -62633432623563353239353737623938663437303061623866363431343436313837343539343436 -30393737346435363563346235633438306339366639613161383763613730353437636537366565 -66396534353230613137326139333435393034346266643364646138666130656532323166623664 -36656631336636393131316361376339383333333837663263343065623937663062346666323464 -31636662386362633337306338633138623434323563643663663835656430333762366631343033 -63343861613365336434653035333132636363656139393065356330376161383039633533393366 -38326630396535383830303437666333623735393131336630663936373465326466616133343735 -34393535326335646265383333646233386434636530613464643266326134396462313762363066 -64303637386261653866623334666532323130393838313034633038356361363239386435656239 -63386163653362386565646561353238653962343033386632336364313630353063356336626338 -63356137616530643434313862643235333063363533666635346661363861383366326533646665 -66316531633163313365653337663633393661313466643331346230613539653663336637353162 -63323730386433626364333632383761323362363664306264336334613766373532386533336662 -65396336313332633539633966313733386335613236666234313664303631326361376232383036 -33663563613864643532363038303832363434386265366637376561373434643239396231643634 -61633935363234383935393333666566383766646461623061613361636531386363363238393465 -66643636633465343033306437626632373734323762666164356231333137376535313831373236 -66373435323339316533383432393766633566306635336166336336643532313430393539336332 -37613666333763396439396466313535383066636237343964306661636337383931636631633463 -64383130613964376462663837383762353634623064313431623266383839333434653537626663 -34363736643937326134616136306236656261313136366635303036646435353863306662353034 -39363266356636623135356232383630373162363330396262323734643564366232333632373739 -36656366633335626131643930366534326164646265643339623539323863393632353838633861 -36363763313039633530623736313239353036666162346432643732343438366265306164383235 -66376536653761383665316665396435646634363436376632646563633163623766366635623731 -66623831393230326566313932653536343838626566623038653534623434313239343763313433 -30376630656638623036336166346133633738303638366238666639653030323832323365353766 -33623261343664323631393166313539623734393438346562613735623636396137333733303235 -34636630643932383536633836343765373234623639643437306532353764386435663665643738 -65313065393534383935303639656534633363303235393761343330386630633734313561376433 -63623464653034336661386435386431306634613239393039386439316538343261306234393534 -30323164363734303037663134663734363636653935336539396330383931353461353966356533 -63316462376262626261623031353063353030363230653966393832626662623963306266393635 -34643565386137663435373639393537626632373535333865366165353839343732353439616265 -37373563326438656134363931623565313565666334323533653264356633353931323661663832 -35396637353163383466653765376132353561343031623039383161623034396134626562666632 -64643737306634613966663362373038396231616632636539663966313364383137326164396533 -38643266316464366161343761633332633036383537653565653139336562666461666465376230 -61656361623264366566656461373863656161393333353933326665333730356333343531653766 -36303964353134396533306630343832303961336239643462353439363962353738663731663033 -38343833323365366662343665633130306537383861613762613763646631626330613937393432 -36393338626661306130353461653565313766623762613564356466393663383339356335306436 -35373636396239653361633764366133363334643232643734383335346635386139383130363938 -35643665316139613639666436633531396537353439633839623063383134376230376235366534 -64393634376633343063646433323439303438303432323335393265366633343332616461386665 -62626132346364373931303936623963663037333665333639333966646431663635303365343630 -36316464346266313166356262346331656432363865643135613964356662366632323736613437 -37336561376334303534653237303964656665643561376237653531653764363463363464313737 -66313335383838333233663162336337373764363131643836363365353039323836346433313139 -33633636306131623335623639666130393535646536346438303435366436623132613562643066 -65623336643463333932383865353232333339343231613964373566653063643662376338316437 -65343466396530386230396236656463396566336230343838366264656161323063336264643364 -31643066366632653962323636346565313833393966343761623066336464333733646238386332 -37393534356333396661353130363462633366343535616338653330393262663066326662366530 -30383536653838333066623063306233613735363665623937373362623337323339373663303933 -64306538636533643337373766346266326238653937613366616161336333636662386238346237 -35623237333131623363376230663132373939366235376631626263393764366661346438383537 -33306233663630323863343839636166303461666135313639356637303435643062303364646132 -64656636396264303962663363376166323834393863393264323464336662393930636533363834 -61363937383766623930623537353037623031363736633332393766323238383464663531653438 -34653933623166326530343632386165666266303438623739356635346163346239643065336537 -33316631613431623030323433306133376237316535306666343433646339313965663565653130 -61303065303530616639383066623534353566366632313465316136613437383765643039313137 -64393333353834353033353532373834343831326535393837323037373136303462373530646430 -62633964626164626139373562616430613662646265396139356561666462376435643031643964 -62316339303561666435663438393838356437646336623664613863303134343261373830303730 -36303732393262363530366563343165386436313833633533623637646464616663353035363235 -31353764353135343033363562313465383563376164323766323735646336353834613264316464 -65376134346437333439376636313531666638313333313963636363353965363239323830663434 -30373234373235326635633063366232333235653935616365656135396465366564313530363161 -33363864303063363363363535323333616661323338323735323135613562366163393366393963 -31363331373661636134333063363731343831333061326431633061663064356466306438646565 -62646431613861393237313138353330323734636666333934666437343665653965316436636464 -64333432313561656339366533363765343365613139626331396636376535326165623735346633 -61666364373465613438643032316362303335333431306166393766663032666665666564313837 -38323232373439376563633939623835656464313361333666623336666461383330663764666166 -34343638646637646338313737336566623937616666626637386639346530616439316635396661 -61656439346231663436386665653031643334393734346163353963633333646239623536333730 -39363731316361633139326637303839633633633738633737326465343664336261646337343831 -33623261363136613935316635663434633962613635616333383238613965646637363735646264 -39626635323532396339653861383137336136346539303931323262663635366432376365623732 -35356639376161323937306136376236643431643863393630633662346531343230643636666633 -33636139323663353835313663656162656264313739663066616663346639326434353762666330 -62303235363964323964663664353265316461373637346230646364393065653763646636303664 -62623735633234303063306531393662326363333234333763613335666233373131303334366665 -30633037323934633232316366663563326562353834643132623031613138323835373933376536 -37323638396561643363663235333632323234303734643637663261393736613738346433333539 -33623865363862653731366466333530343561653661653638353738346438633236343730326239 -35363837393736343632663765626635613865656165396336646538376434653537633037616566 -37393766636661383538386164363630626438616365666532333039663464303835373134333432 -30386439333732666266666331633566326463316438333639336665613933343963363562636335 -31356465303535663233656466616462646161666538346334323335396561656634396631356262 -62323561316261653134363436636333353665613165323935353439666164363163303530353964 -66633536306432323335306530323733623536383366653566653339383161383434613330343934 -63666131333963356336333763386338343437613035313036393766636535386234613665636632 -34383032633931663633386636343233646131393066393766366635343935356538306231336232 -32353836353335363731353234303337343737363064636463613132393563316136626539326264 -30343230343166393466353664613438383462623734363038363962306437663938306235316235 -64373433626366636133373964633131636534333839623033383163613561363430343534313966 -63366561613939623334653434616238616136353665386665346366666236326363376135373363 -36343965343131356662373834366535313963333366633536343930633561626230373736303030 -65373063386232316332343963343631333162376334636665633531313936303232323933313263 -31373639613365396533303431376532643434626538653538356630626532343530663163366232 -39376566353361333639343237653937303161646164613831643235613139373961663566333630 -62663430313832663432306634356464623838303835373962373533313565663735326535326461 -66653939623836666461626631656534616162323433356431303366306635323938373138613031 -66366461353133386436633765386262373736633539643637373237326235633764353132636561 -63343438666361303364333732303738323463353832613438636663363332333132373361633439 -64373334386661626364393664346262653739613134346365636132613538666339393036333163 -64373363643031393062316135336561303665623465336464356165313466633462666331303030 -62626535396634623366313662643066313462636330343237316433646434313737626562346237 -65393161333336386430353938336536336461326132393933343335656131393664356463376631 -38373862363862343839666161663136336534633862303636653931363136303337353664393737 -35316138353361306133636463366463333137616562366135653230343039643365646665636364 -65366564643738356539663331613439656536376263306362366165306230373336386166613865 -33626631623364663062396466626531666238616436326163643533386366633164633262356431 -32313762336561396531386535383434396135633066316334626632633431356363373034663135 -33323166373562326237636265353634623236666364643837353232306263643132343761373230 -63626331393461653662383666646332303632386634313034313664346432656265 +39383462626535653332386339666338643666353535313064613633353230366534396665376235 +3061353662623030623038663063646661616537306531340a346430623235633264346163396661 +61333533646538373838383231353338313262656466613664393132356536396134306265633036 +3861643930326163610a303834613330326536616432646463323731666536383563306162616130 +61613665313665643265393938383664353538653937616230316136643765653632313932643235 +38346639646630396566656165376639383034323439386563376563666662363866303639343230 +37313938613261313962303666666431383864383963333833316531333065323630306165623562 +34303736646663373230303733356536656634656330316362303066303939376230393330376338 +30653666376330303234376132343731376464383061306139643237393463613565613561363335 +63336366613736616336643937326663383161376534626432646437633136646339313566626335 +32303638616466376535373331363564386534396164323538363664623965626635653438613436 +34313466363939343934636635363762626662636462356332646232633539323034646338643139 +34626462303937633032396334643232336363663732636132316136666232623865663434336433 +32363164626635356366323331366634613532666530643233643531613264393639333065396139 +39306132383462633138626231383463623335363130303565336662343933643235303331333335 +30373665303665393532613333383539323332363636366363616236343533373663333864653961 +63613465656465653763313635343263636430653964626662666434666634313664363930656165 +35313265643031653861653237643763663332383266646463633732616666346638333937613339 +61653265663765343764303439663230326234363961303835353535346332666561643266386632 +33633437336639663063373737356536326332336631303834363838346634646535366331616661 +38633235336361633132306330393131643530626661663733366132633939633637656231623438 +36646332363065643930343831316238333566323234643435346530323038626562666165373934 +35376136343566656436343438336363363364303965363533393330393430316566313033313737 +33303536323235346437316636643566306465393031353439666564313165663230643732323563 +38386265633938316462343636636666343638356338393061323134303837393231653435653833 +32616433343465623536653165623435383933656534366664343461363662313736623061616163 +65343664643736316337356134336236663434353635633032626664633333356661646433353566 +39643664366132623363336339313739346634616439396535346533396438646433323538653933 +36373639646566353931643630656432316166306531363562343666353633633063346263626532 +65343437656564633466306166366535623265363566323461626237333862623465653163643463 +38316338353233653131373336306537386638616462366463306464616566623932323633316631 +32386231313430343765306130616361343537393333616466663763633736303131356261626137 +38396634613734356465313865326361343432633066393861383363343262306662316662363330 +61393035666437366635396361313831626138396337616630346339643133366363616264393635 +63336339333862353838383437346165326238313632323862616161303330376464306265306661 +34623530656331623863376661373032333462616462646630303630626631623830343637386133 +61383236636534643932393262316330326161323435316162666566663235336639373431386463 +65363336313439323064333633353236313537613132396632343665616636396632363362366132 +62386538336162333062306136383737323336646163393161316164363561306461373731653334 +64313765313330653762336633396262626361353730353737646534383262663330363335633539 +33646534643231663464376638373934613432623639643061613262383635323133326564346263 +38303863333638353864636436623938326538323765623165633465353433663638363736316432 +66626264646335376432376561336261336633643039663533343562393736303164643833346566 +36656361356130303931666539363231633232393731653262666162396163376230643635636163 +35376163653036383364393638343663383937353239663030396663303431356632346339313931 +34326638333766353763623433393938663732373538323735323166613439663132653961366132 +63333465653035336238633665393134633436386263383862386166633436633834393034623231 +31613362346239313639346262633566373662623261343939313434616336663337306632656434 +37373331333163343035353731363462613436613833386561366430343336386261373835663237 +35643032323065313835353637666663323337626330353138386661383039383162336437343764 +31643037386465623138393236613166366335636633383762623762336535323936363134396436 +66323339383833613363653331643032616438383238643666653938613964393464613434623138 +32343835653332386135653762326238313833653264396266383830633536396131373839343332 +62316132326361646633653732663063366663646366366633353133663963356637663137663263 +62636261336463626439616334633464373566373864393965623166633733643730393664393131 +38636333326465386132636139356339613730366539323762653564616639656630623435656262 +39333331353933396534323562333665373233616135386336353334383436356630373931653237 +66656461353239346435643034643166363765623631373736656233373965373264636566313139 +63613839636565613663623831636465616335633636613437363066353961396631306332636537 +38393230616132316266613331643534363764666637663532353539333661636331623364623433 +31363433326536646439336131633961303737633364623961306466643836653133363235326439 +37616461346433623130616432353035653464366136353062633732313362643033323830626163 +64343866386162313133393637343365336363386563653436336438653836643831333064613461 +32646133636130303166333466363139386330653030326230626136643830666436653036323061 +61306530323666616332373635633132346635376237393237636638346433393131323533643761 +32613230663161643238623865613333323263303831366237386331353935636532343264653134 +61636136636538666461373537376539343136363466343766613961386231313336633438373965 +35666130373163373236396638303738306538653238343838386130663865366137613730616161 +31313330343262393632316635326134646662323535366338656434363162363461646362663761 +63653266653835333762353433633562346164623661653762666135363435376665663437333363 +63646663343030373566383665383864383463366363353630366138366433363362346634383633 +39396235303830626138363539333138363230313463303062383562633234616230333365333831 +66323632656336303231653138636164396532616265346261663839346235643439616466643438 +61653837336532316230313664623233353634643037653138313936653261363537656365323961 +66323563376363396336623164643866373833643138646361383132363537323862336262633766 +35643235613033333530346432376639353266346433386439323830363237666464306535386463 +36303032346234336335613764383165326536643439656563326465366136653964623066343036 +39333439343932346162373030343332373135343064346336353431303330666561623963366465 +38336162303337303235636131656564666536336139393537323831633739643531333838653432 +31633535386230306335616564653832656361643339386638623937326635643465643564623833 +34376633643830316635646131373039653932626231646536386165623435353861353639343734 +65393234646662363435636536346265336337616235376466613865383965316234383434303930 +31363932633765653335393262316363356265643533663762356235386437643238316334356131 +62333836666430616632613962373635633333333135656338323636346432636131613165663830 +34616338386232633066323131663464393963386538393632626265623133363637393537626535 +33303232643465353365626364306663363233333965643464393734626634353065663637363164 +66396261643531386463363931343531663733316336313537316538326436386662616338613765 +36363037323436353035353638613931373364393961336637636438306263376565346234366539 +62353530313335303734643731393636303532343362626361366162663362356463376636646661 +31633161633134626164363430306234366130396235613238646661393034656638313333343766 +39383266623234653338396637336435666134653164383463373734616364323731616462376634 +62643039623931393736383261393131333261623130333363636462343830626435623336666161 +38663936303531353932316430626261316437333536313230353431393262616463653430643136 +32626539316330323664643533653661323038313063623262346264393333343030656365653566 +38303835363837383539333265613865663930633535373230653763323437363535656265363633 +64653230646363373032653032336163353033336539303964663830323935393732393538373835 +34323938616333346139316361376463383531393134316262376332636637353834623937333264 +32633135386232656662383838623238323336633638636631386361306238333035636232376432 +39383634396635343032313164636132363331373830656632626461643464343631363838323233 +61653530616666313635383364633738346134356533336431313935353364376537666635633936 +64303764386538363438393430653236653862653165303631383434613037303639626461623066 +63383966343437646433656531333761393430636236326362656137663861663938323333313238 +34323433653363353565396636623030323633303536303366323030386566313661373063353435 +36346232353838663962353436316133643331383065346536663961373665343338373761653439 +63353463326130356565626164613836393931383236363638643565386230386363363236366536 +35653366393139616362633737616165663438653738636337313861313035353430393962373264 +64303138373339616436623664663638623666336637653662313131646630376134323437616166 +36366330303836396133393764653462316532313434633738316335626430653133633162623734 +31633465623766343566393431306535633033363937376638643865306239316565366162613366 +36323439663836373630643138303235313636663431663738653436646134636230393535623236 +30393036336137346630623035626239313661376539333237313261323438386632303939336566 +31346433623235326630613165323762363835343939623666613638616164626236646433373864 +39623866373238356138336266313135393063373733373332303338343965303538316431663462 +65613766623538633138616632366436316230646162323432313434393938356635663836633036 +62356631626138343436393238306539663532353431366162336130626334633230653638323134 +62313830343730653230346233626633386433313131663233303435373561343036393838633830 +61346135313130323464643531646139393633386461666630613637363833393466353737306530 +33653331346464613431643735396661346431653933346563643531626638653932316238356632 +30343864353431393231386639343237373130376639343236613337643739316132333133373962 +34393962393065363133353338613739613332653365316335306663643331383138323537613866 +35336339613138323031366436616139306339346363663662316265306161343935376539376166 +31316264353636343463636230646431333164626135663833316661363261663332366335326130 +36393438626531663538313931623132303332366538396666303939663061353865633936343162 +61303533383439626636646230623931383636633136336636643436646662396630303637333535 +63623661376130303134633761393835643266326133616236363464326238376466326266633763 +62656637663265363338636137346138616431616139653266386432373461316634326339333932 +31373562366132326635366164353637626334353464323764636463316665373936643030333539 +35646634643339653564303835333163383936386335353139636563363939323632343562633662 +65663033623232393430396139626135326236663463333639353338633736356430323036356537 +35326538616664386233333264313637626436613033643936343063383261653333353339306538 +38313631663263303230363830623235343732626638373535316537343938633632633534386365 +36623336343866383132373162626565313665393964663562653463396130623238643631363838 +38313363666635653134353663653461343135313138636434376362303639356362613639346238 +39343432613164643265333231366139326462313363633234313832386132323239323166613931 +37666135313036313236643261333739306362626236613435343766303162373464306531326333 +66363664306535666235636234346164373064373863646562353331346536613764646633333861 +65323361666430336462656238633735366661666662643632646562366535346432346233663330 +30336261636239613932303032653962376231333536633830646364663135626561336339393364 +30633532616163616363623835306335313163636337353263333836326261643334636666333836 +66653731316639376162386363303766323734653938373866306264376462643038386436323438 +62396136353961613633316661666136323037613863383338613633386536336330333363626237 +66376266623963353439613433633037623135333062636436656663633034343437346262333534 +65333331633037396239653838383462386136623764343436623737643564663261313834333133 +38343934636164323135653435653835333331623864383133313965373065643439623361613631 +62393036356164313665326537626363633839313935663636613434363934616135323036346131 +32363033323862613562353364373035653631366431303865646134316438366266643965623536 +37633835386439343932663861666230653463336332343864363035346135643535343535366135 +61653262363937373036616661333639383362383337316561303163643639306466353334353866 +38653862633732333962616135383234336233383434643264396461646264363134353531356537 +31643030386437303563313033373366366565633435616435643164626334643364613535363232 +64623835393434316131303237393038323930633861396431643833346636353534663362336264 +35326337303036303737336333636432353033366461653435626464646434646461303337643437 +34666635626335333133326663353234393637386330393964633836666364336164633030643534 +33346335643962323866306634383335613463633136636362396464383135666162333866333037 +34383461306332306132303063346634643838363835393564646537663236343432313766336265 +30336537313735386261376365623461303337303430613461373866366665623332343633643439 +62663366323165383463333864653132363632636134333932393333366466356536336666393330 +37346361393831363633623735363032646464303765353935643265373463323366353461316535 +33366265333863613531653663326162383431353932366231396332306234363963653566656130 +32313831653634353865313561363966653633313535363464326664396432623561653738313561 +62626239383562626139623865353039643365613565386337313666386364656537373165663833 +36353264613462393365396562636138663535613433653461303935646438653039653736616564 +66643737643437313366333830303839333965626164646637326331626265323035626464326462 +61343861316532666461336533656430633261393732323232313864623832323662613334386334 +65653837366532613538373138303633346566663330613965633963653864666330346532313038 +37616163626235303364303534383930376262643539633630336162383864386437633162323332 +37393236346334646139373130653438393663393335336331316336386236303535366237616330 +61346663323234626637343532666135633133343234393064323033623832613264623337356139 +39373530366564333437383266333636666363393461303034346566626562393065633732643564 +61393936316330313137386338323461353339323331383533376362323833666635383836333736 +65383738373537633164626237626364623866663031313864633032353137353331393765393339 +62653337623538623664326563316266613263376536653138656632303939626133653964616537 +39613763383938633163323234356139313261613332373861393462323930353066333134306532 +36303061363763316632616134353633323739633665313962623364356562366535326339646635 +62333838373833333133653338366638386566366337303232353537303961373030303464613964 +39626538353033383430333132306637363633353135333232613462656634626438646139653266 +61636365623634396438633731616665623333353735663036303664333461636536313733363232 +61656333613736363039613865383566343336346466366165323737336233303239333163326437 +66373231323333316630653039343731323332656131633131663635636433313732663935303232 +64346664353836383864333163353834313732376638303036323235303034373666653666383264 +35316537646237373535366262656532623331393566316235396631353030663961323238376236 +64303766636134626236646530616331323334613233633932653861333961346138646266363864 +36616563313131633931376230613861623331373839353733353730363234343339303037653663 +63646561346466386131626361326533363739383832383437393033376261323363643966396662 +63323234343931326663636662363837313665383962373965623335346330623735346665306364 +64333239303031666262316239613033303837386164333034616461316237376431643862366231 +63613935653533373530346131346265363565343166656563303164663132346430616234346235 +38343363633063326233306261643466383262346661363933316238303836373831623438653238 +39646438306639383561646666646137356561633131363563663066613263333234636466366636 +63323065346430353834636332383536643061356464333863356666333264356633306131303464 +34343230396339646466663963613333356235316638636463643333346262613637633936623839 +30666430353261363432653162353166326263396333653633663764376535643331336661626666 +64653433666638313838353766376663633564326633376461336234366534656531656638323164 +33636130333032393030633131646532323766383062646431646461353035636338363865366564 +32373664636362613539356665316461326638363631316664626435333238653135376134303035 +32396631666466363338626131393562316638393531633438613133623734323461326232383665 +65356538616538363065326264306236613330303439636232316339363333306238393766363361 +39396634366333636637643136653164623165613733373039366335323061636332363636336330 +61306237646637613961326564626334316334623835616365623132303039373930316137653433 +38306134383738346662 diff --git a/ansible/roles/gatus/README.md b/ansible/roles/gatus/README.md new file mode 100644 index 0000000..129057c --- /dev/null +++ b/ansible/roles/gatus/README.md @@ -0,0 +1,110 @@ +# gatus + +Deploys [Gatus](https://github.com/TwiN/gatus) — health checks, a status page, +and alerting — **as the upstream container image**, on the +`observability` group. + +## Why the container + +Upstream publishes **no binary release assets**. The image is the only artefact +they ship and therefore the only one they test, so it is what `docker run` in +their README gets you, and it is what this role deploys. + +Building from source is entirely possible — their Dockerfile is a bare +`CGO_ENABLED=0 go build`, the Vue dashboard is compiled in via `//go:embed +static` in `web/static.go`, and the sqlite driver is pure-Go +`modernc.org/sqlite` so nothing needs linking. This role did that at first. The +reasons it doesn't now: + +- It produces a binary upstream never ran. +- It means compiling the AWS SDK, gRPC and the Google API libraries on the + smallest box in the estate. On this VPS that load was heavy enough that + unrelated Ansible tasks timed out while it ran. + +The cost of the container is a daemon on the machine whose job is to notice +when everything else breaks. That is a real trade, made deliberately. + +## Pinned by digest, not by tag + +```yaml +gatus_image_digest: "sha256:c5f210d0…" +gatus_image: "ghcr.io/twin/gatus@{{ gatus_image_digest }}" +``` + +A tag is mutable — `v5.36.0` can be repushed — so pinning the tag alone is a +weaker promise than it looks. The digest is a content address: if it resolves, +it is byte-for-byte the image reviewed here, and `docker compose pull` either +fetches exactly that or fails. `gatus_version` is kept alongside it purely so a +human can read which release it is; **the two must be updated together.** + +## What `FROM scratch` means for running it + +The image has no `/etc/passwd`, so there is no user to drop to by name and the +default is root. The compose file runs it by numeric id (`10001:10001`) and the +data directory on the host is owned to match. Everything else is locked down to +approximate what the systemd unit used to do natively: + +| systemd | compose | +|---|---| +| `ProtectSystem=strict` | `read_only: true` | +| `NoNewPrivileges=true` | `security_opt: [no-new-privileges:true]` | +| `CapabilityBoundingSet=` | `cap_drop: [ALL]` | +| `AmbientCapabilities=CAP_NET_RAW` | `cap_add: [NET_RAW]` (for `icmp://`) | + +## Configuration is a directory, not a file + +`GATUS_CONFIG_PATH` points at `/opt/gatus/config`, and Gatus merges every `*.yaml` +underneath it — maps deep-merge, lists append. This role owns exactly one file: + +``` +/opt/gatus/config/00-base.yaml web, storage, ui, alerting, security (this role) +/opt/gatus/config/endpoints/*.yaml one file per service (gatus_endpoint) +``` + +**A primitive defined in two files is ambiguous and upstream refuses it.** So +anything that is not a list belongs in `00-base.yaml` and nowhere else. Endpoints +are lists, so each service's file appends cleanly — the same shape as +`caddy_site`, where each service contributes its own vhost. + +## Pull and push + +Gatus polls. For anything with a reachable HTTP or TCP surface that is the +better check, because it tests the path a user actually takes. The monitoring +host joins the headscale mesh via `infra/920`, so internal boxes are reachable +by MagicDNS name and can be polled directly rather than having to report in. + +For state with no pollable surface — ZFS pool health, UPS mains status, disk +usage, backup freshness — Gatus has **external endpoints**, a push API: + +``` +POST /api/v1/endpoints/{group}_{name}/external?success=true&error=&duration= +Authorization: Bearer +``` + +with `heartbeat.interval` to alert when nothing reports in. That is the same +shape as the generic `healthcheck_push_url` already wired into every service +role, so those scripts need a URL, a POST, and an auth header — not a rewrite. + +## Variables + +See `defaults/main.yml`. The ones that matter: + +| Variable | Default | Note | +|---|---|---| +| `gatus_version` / `gatus_image_digest` | `v5.36.0` / `sha256:c5f210d0…` | must move together | +| `gatus_bind_address` | `127.0.0.1` | **never bind publicly** — the push API shares this listener | +| `gatus_storage_type` | `sqlite` | `memory` loses all history on restart | +| `gatus_alerting` | `{}` | pass-through; any provider Gatus supports | +| `gatus_allow_icmp` | `true` | adds back `NET_RAW` for `icmp://` checks | + +`gatus_alerting` empty is valid and is the current state: every condition is +still evaluated and recorded, there is just nowhere to shout yet. + +## Verifying + +```bash +docker ps --filter name=gatus +docker logs gatus --tail 50 +curl -s localhost:8080/health +ls /opt/gatus/config/endpoints/ +``` diff --git a/ansible/roles/gatus/defaults/main.yml b/ansible/roles/gatus/defaults/main.yml new file mode 100644 index 0000000..95c41ae --- /dev/null +++ b/ansible/roles/gatus/defaults/main.yml @@ -0,0 +1,83 @@ +--- +# Gatus, deployed the way upstream distributes it: the container image. +# +# Upstream publishes NO binary release assets - the image is the only artefact +# they ship, and therefore the only artefact they test. Building from source is +# possible (`go build` alone is enough; the Vue dashboard is compiled in via +# `//go:embed static`, and CGO_ENABLED=0 works because the sqlite driver is +# pure-Go modernc.org/sqlite) but it produces a binary upstream never ran, and +# it means a full compile of the AWS SDK and gRPC on the smallest box in the +# estate. + +# ── Image ──────────────────────────────────────────────────────────────────── +gatus_version: "v5.36.0" +# Pinned by DIGEST, not by tag. A tag is mutable - `v5.36.0` can be repushed - +# so pinning the tag alone is a weaker promise than it looks. The digest is the +# content address: if it resolves, it is byte-for-byte the image reviewed here. +# Both must be updated together; the tag is kept only so humans can read it. +gatus_image_digest: "sha256:c5f210d095fa78e6efaa20ffeb14803f2ba4f10615e16a6d12087697149617f0" +gatus_image: "ghcr.io/twin/gatus@{{ gatus_image_digest }}" + +# ── Paths (host side) ──────────────────────────────────────────────────────── +gatus_dir: /opt/gatus +gatus_config_dir: "{{ gatus_dir }}/config" +gatus_data_dir: "{{ gatus_dir }}/data" + +# Gatus merges every *.yaml under GATUS_CONFIG_PATH and its subdirectories: +# maps deep-merge, lists append. That is why this role ships a config DIRECTORY +# rather than one file - each service contributes its own endpoint file, the +# same way each service contributes a vhost through `caddy_site`. +# +# Primitives must be defined exactly once across all files or the merge is +# ambiguous, so everything that is not a list lives in the base file and +# nowhere else. +gatus_base_config_file: "00-base.yaml" +gatus_endpoints_dir: "{{ gatus_config_dir }}/endpoints" + +# ── Identity ───────────────────────────────────────────────────────────────── +# The image is FROM scratch, so it has no /etc/passwd and no user to drop to by +# name. Run it by numeric uid/gid instead, and own the data volume to match. +gatus_uid: 10001 +gatus_gid: 10001 + +# ── Web ────────────────────────────────────────────────────────────────────── +gatus_port: 8080 +# HOST-side address the container's port is published on. Gatus itself always +# binds 0.0.0.0 inside the container - see the note in config.yaml.j2. Never +# publish this on 0.0.0.0: the external-endpoint push API shares the dashboard's +# listener, and Caddy is what should be in front of both. +gatus_bind_address: "127.0.0.1" +gatus_ui_title: "Status" +gatus_ui_header: "Status" + +# ── Storage ────────────────────────────────────────────────────────────────── +# sqlite, not memory: history has to survive a restart, or the dashboard lies +# about uptime after every deploy. Path is INSIDE the container. +gatus_storage_type: sqlite +gatus_storage_path: "/data/gatus.db" +gatus_storage_caching: true + +# ── Alerting ───────────────────────────────────────────────────────────────── +# Pass-through: rendered verbatim under `alerting:`, so any provider Gatus +# supports works without touching this role. Empty means "check and record, +# alert nowhere" - valid, and the default until a provider is chosen. +gatus_alerting: {} +gatus_default_alerts: [] + +# ── Security ───────────────────────────────────────────────────────────────── +# gatus_basic_auth: {username: admin, password-bcrypt-base64: "..."} +gatus_basic_auth: {} + +gatus_maintenance: {} +gatus_log_level: INFO + +# ── Self-check ─────────────────────────────────────────────────────────────── +# Gatus panics on a config with no endpoints, so the role always ships one. +# Turning this off is only safe once another file in endpoints/ provides one. +gatus_self_check: true + +# ── ICMP ───────────────────────────────────────────────────────────────────── +# Gatus supports icmp:// endpoints. Raw ICMP needs CAP_NET_RAW, which the +# container would get free only if it ran as root; it does not. Set false if +# you never use icmp:// checks and want the capability dropped entirely. +gatus_allow_icmp: true diff --git a/ansible/roles/gatus/handlers/main.yml b/ansible/roles/gatus/handlers/main.yml new file mode 100644 index 0000000..25ad865 --- /dev/null +++ b/ansible/roles/gatus/handlers/main.yml @@ -0,0 +1,5 @@ +--- +- name: Restart gatus + ansible.builtin.command: + cmd: docker compose up -d --force-recreate + chdir: "{{ gatus_dir }}" diff --git a/ansible/roles/gatus/tasks/configure.yml b/ansible/roles/gatus/tasks/configure.yml new file mode 100644 index 0000000..94f71f0 --- /dev/null +++ b/ansible/roles/gatus/tasks/configure.yml @@ -0,0 +1,54 @@ +--- +- name: Assert Docker is available + ansible.builtin.command: docker --version + register: gatus_docker_check + changed_when: false + failed_when: gatus_docker_check.rc != 0 + +- name: Create the gatus directories + ansible.builtin.file: + path: "{{ item.path }}" + state: directory + owner: "{{ item.owner }}" + group: "{{ item.group }}" + mode: "{{ item.mode }}" + loop: + - {path: "{{ gatus_dir }}", owner: root, group: root, mode: "0755"} + # Config is root-owned and world-unreadable: it holds external-endpoint + # tokens. The container mounts it read-only and reads it as gatus_uid, so + # that id needs group access - hence the group ownership below. + - {path: "{{ gatus_config_dir }}", owner: root, group: "{{ gatus_gid }}", mode: "0750"} + - {path: "{{ gatus_endpoints_dir }}", owner: root, group: "{{ gatus_gid }}", mode: "0750"} + # Data is the one path the container writes to, so it must be owned by the + # numeric id the container runs as. `read_only: true` makes everything else + # in the container immutable. + - {path: "{{ gatus_data_dir }}", owner: "{{ gatus_uid }}", group: "{{ gatus_gid }}", mode: "0750"} + +- name: Write the base gatus configuration + ansible.builtin.template: + src: config.yaml.j2 + dest: "{{ gatus_config_dir }}/{{ gatus_base_config_file }}" + owner: root + group: "{{ gatus_gid }}" + mode: "0640" + notify: Restart gatus + +# Without this the container crash-loops on an empty config. See the template. +- name: Write the gatus self-check endpoint + ansible.builtin.template: + src: endpoint-self.yaml.j2 + dest: "{{ gatus_endpoints_dir }}/00-self.yaml" + owner: root + group: "{{ gatus_gid }}" + mode: "0640" + when: gatus_self_check | bool + notify: Restart gatus + +- name: Write the docker compose file + ansible.builtin.template: + src: docker-compose.yml.j2 + dest: "{{ gatus_dir }}/docker-compose.yml" + owner: root + group: root + mode: "0644" + notify: Restart gatus diff --git a/ansible/roles/gatus/tasks/main.yml b/ansible/roles/gatus/tasks/main.yml new file mode 100644 index 0000000..e1b67bf --- /dev/null +++ b/ansible/roles/gatus/tasks/main.yml @@ -0,0 +1,5 @@ +--- +# import, never include: dynamic includes are opaque to --list-tasks, which is +# the primary verification tool in this repo. +- ansible.builtin.import_tasks: configure.yml +- ansible.builtin.import_tasks: service.yml diff --git a/ansible/roles/gatus/tasks/service.yml b/ansible/roles/gatus/tasks/service.yml new file mode 100644 index 0000000..bfcca3b --- /dev/null +++ b/ansible/roles/gatus/tasks/service.yml @@ -0,0 +1,35 @@ +--- +# Pulling by digest means this either fetches exactly the reviewed image or +# fails. There is no "latest wins" path. +- name: Pull the pinned gatus image + ansible.builtin.command: + cmd: docker compose pull + chdir: "{{ gatus_dir }}" + register: gatus_pull + changed_when: "'Downloaded newer image' in gatus_pull.stderr or 'Pull complete' in gatus_pull.stderr" + +- name: Start gatus + ansible.builtin.command: + cmd: docker compose up -d --remove-orphans + chdir: "{{ gatus_dir }}" + register: gatus_up + changed_when: "'Started' in gatus_up.stderr or 'Created' in gatus_up.stderr or 'Recreated' in gatus_up.stderr" + +- name: Flush handlers so a config change is live before it is verified + ansible.builtin.meta: flush_handlers + +- name: Wait for gatus to answer + ansible.builtin.uri: + url: "http://{{ gatus_bind_address }}:{{ gatus_port }}/health" + status_code: [200, 404] + register: gatus_health + until: gatus_health.status in [200, 404] + retries: 12 + delay: 5 + +- name: Assert gatus is running + ansible.builtin.assert: + that: + - gatus_health.status in [200, 404] + fail_msg: "gatus did not come up on {{ gatus_bind_address }}:{{ gatus_port }} - check `docker logs gatus`" + success_msg: "gatus is answering on {{ gatus_bind_address }}:{{ gatus_port }}" diff --git a/ansible/roles/gatus/templates/config.yaml.j2 b/ansible/roles/gatus/templates/config.yaml.j2 new file mode 100644 index 0000000..67d6244 --- /dev/null +++ b/ansible/roles/gatus/templates/config.yaml.j2 @@ -0,0 +1,56 @@ +# {{ gatus_base_config_file }} — managed by Ansible (roles/gatus) +# +# BASE CONFIGURATION ONLY. +# +# Gatus merges every *.yaml under GATUS_CONFIG_PATH: maps are deep-merged and +# lists are appended, but a primitive defined in two files is ambiguous and +# upstream refuses it. So `web`, `storage`, `ui`, `alerting` and `security` are +# set here and MUST NOT appear in any other file in this directory. +# +# Endpoints are lists, so they append cleanly. Each service drops its own file +# into endpoints/ via the gatus_endpoint role - the same shape as caddy_site. + +web: + # 0.0.0.0 is the CONTAINER's interface, not the host's. This must not be + # 127.0.0.1: that is the container's own loopback, which docker-proxy cannot + # reach, and gatus comes up healthy while the published port refuses every + # connection. + # + # The isolation comes from the port mapping in docker-compose.yml, which + # publishes to {{ gatus_bind_address }} on the host. Caddy fronts that, and it + # matters because the external-endpoint push API shares this listener with the + # dashboard. + address: 0.0.0.0 + port: {{ gatus_port }} + +storage: + type: {{ gatus_storage_type }} +{% if gatus_storage_type != 'memory' %} + path: {{ gatus_storage_path }} +{% endif %} + caching: {{ gatus_storage_caching | bool | lower }} + +ui: + title: {{ gatus_ui_title }} + header: {{ gatus_ui_header }} +{% if gatus_alerting %} + +alerting: +{{ gatus_alerting | to_nice_yaml(indent=2) | indent(2, true) }} +{% else %} + +# No alerting provider is configured yet. Gatus still evaluates every condition +# and records every result; it simply has nowhere to shout. Setting +# `gatus_alerting` is the single change needed to wire one up. +{% endif %} +{% if gatus_basic_auth %} + +security: + basic: +{{ gatus_basic_auth | to_nice_yaml(indent=2) | indent(4, true) }} +{% endif %} +{% if gatus_maintenance %} + +maintenance: +{{ gatus_maintenance | to_nice_yaml(indent=2) | indent(2, true) }} +{% endif %} diff --git a/ansible/roles/gatus/templates/docker-compose.yml.j2 b/ansible/roles/gatus/templates/docker-compose.yml.j2 new file mode 100644 index 0000000..67d7573 --- /dev/null +++ b/ansible/roles/gatus/templates/docker-compose.yml.j2 @@ -0,0 +1,44 @@ +# Managed by Ansible (roles/gatus) +services: + gatus: + image: {{ gatus_image }} + container_name: gatus + restart: unless-stopped + + # The image is FROM scratch: no /etc/passwd, so there is no user to drop to + # by name and the default is root. Run it by numeric id instead. + user: "{{ gatus_uid }}:{{ gatus_gid }}" + + ports: + # Loopback on purpose. Caddy fronts this, and the external-endpoint push + # API is served from the same listener as the dashboard - publishing + # 0.0.0.0 would put both straight on the public internet. + - "{{ gatus_bind_address }}:{{ gatus_port }}:{{ gatus_port }}" + + environment: + GATUS_CONFIG_PATH: /config + GATUS_LOG_LEVEL: "{{ gatus_log_level }}" + + volumes: + - {{ gatus_config_dir }}:/config:ro + - {{ gatus_data_dir }}:/data + + # Hardening. The systemd unit this replaced got most of it from + # ProtectSystem/NoNewPrivileges/etc; these are the container equivalents. + read_only: true + security_opt: + - no-new-privileges:true + cap_drop: + - ALL +{% if gatus_allow_icmp %} + cap_add: + # icmp:// endpoints need raw sockets. Dropped above with ALL, added back + # explicitly so the grant is visible rather than inherited from root. + - NET_RAW +{% endif %} + + logging: + driver: json-file + options: + max-size: "10m" + max-file: "3" diff --git a/ansible/roles/gatus/templates/endpoint-self.yaml.j2 b/ansible/roles/gatus/templates/endpoint-self.yaml.j2 new file mode 100644 index 0000000..197ae56 --- /dev/null +++ b/ansible/roles/gatus/templates/endpoint-self.yaml.j2 @@ -0,0 +1,25 @@ +# 00-self.yaml — managed by Ansible (roles/gatus) +# +# Gatus refuses to start with no endpoints at all: +# panic: error parsing config: configuration should contain at least one +# endpoint or suite +# +# So the role ships one. Checking its own listener is not as circular as it +# looks: it proves the config directory parsed, the container is serving, and +# the storage backend accepted a write. If this row is missing from the +# dashboard, the dashboard is not telling you the truth about anything else. +# +# It is also what keeps `gatus` deployable before any service has contributed +# an endpoint file of its own. + +endpoints: + - name: gatus + group: infrastructure + url: "http://localhost:{{ gatus_port }}/health" + interval: 60s + conditions: + - "[STATUS] == 200" +{% if gatus_default_alerts %} + alerts: +{{ gatus_default_alerts | to_nice_yaml(indent=2) | indent(6, true) }} +{% endif %} diff --git a/ansible/services/gatus/deploy_gatus_playbook.yml b/ansible/services/gatus/deploy_gatus_playbook.yml new file mode 100644 index 0000000..704ada9 --- /dev/null +++ b/ansible/services/gatus/deploy_gatus_playbook.yml @@ -0,0 +1,83 @@ +--- +# Gatus: health checks, status page and alerting for the whole estate. +# +# Built from source and run under systemd - upstream publishes no binaries, and +# their Dockerfile shows the runtime needs nothing but the static binary and a +# CA bundle. See roles/gatus/README.md. +- name: Deploy Gatus on the observability host + hosts: observability + become: yes + roles: + - gatus + +# The dashboard is bound to loopback; Caddy publishes it. +# +# Auth is done HERE, at the edge, and not with Gatus's own `security.basic`. +# Gatus's security middleware protects exactly four routes (api/api.go): +# +# /api/v1/endpoints/statuses +# /api/v1/endpoints/:key/statuses +# /api/v1/suites/statuses +# /api/v1/suites/:key/statuses +# +# Everything else is registered on the UNPROTECTED router, including +# /api/v1/config, every badge, and - the part that matters - +# /api/v1/endpoints/:key/uptimes/:duration and .../response-times/:duration/history, +# which return real per-endpoint data to anyone who can guess a key. Keys are +# just "_". So Gatus's own auth makes the dashboard render empty +# while leaving the data readable, which is worse than it looks. +# +# The one route that must NOT sit behind basic auth is the external-endpoint +# push API. It authenticates with `Authorization: Bearer `, and basic +# auth wants `Authorization: Basic <...>` - same header, two schemes, and the +# push clients lose. It is not actually unauthenticated: the handler 401s on a +# missing prefix, an empty token, or a token that does not match that endpoint's +# own. Upstream's comment on the route says exactly that. +- name: Publish the Gatus status page through Caddy + hosts: observability + become: yes + tasks: + - name: Require the dashboard credentials to be set + ansible.builtin.assert: + that: + - gatus_dashboard_username is defined + - gatus_dashboard_username | length > 0 + - gatus_dashboard_password_hash is defined + - gatus_dashboard_password_hash.startswith('$2') + fail_msg: >- + gatus_dashboard_username and gatus_dashboard_password_hash must be in + the vault. Generate the hash on the observability host, which runs + Caddy natively, so the bcrypt cost and format match what verifies it: + caddy hash-password --plaintext 'your-password' + then: ansible-vault edit group_vars/all/vault.yml + + - name: Configure the Caddy vhost for Gatus + ansible.builtin.include_role: + name: caddy_site + vars: + caddy_site_name: gatus + caddy_site_domain: "{{ subdomains.gatus }}.{{ root_domain }}" + # caddy_site_body rather than caddy_site_upstream + caddy_site_basic_auth, + # because that pair applies auth to the whole site with no way to carve + # out the push path. `handle` blocks are mutually exclusive and first + # match wins, so the push API gets a route of its own. + caddy_site_body: | + @push { + path /api/v1/endpoints/*/external + method POST + } + + # Push API: Bearer-authenticated by Gatus itself. No basic auth here, + # or the Authorization header collides. + handle @push { + reverse_proxy 127.0.0.1:{{ gatus_port | default(8080) }} + } + + # Everything else: the dashboard, the config endpoint, the badges and + # the uptime/response-time history. + handle { + basic_auth { + {{ gatus_dashboard_username }} {{ gatus_dashboard_password_hash }} + } + reverse_proxy 127.0.0.1:{{ gatus_port | default(8080) }} + } From ede407ebe4eabba1265aa6f89f13422005a1e9d4 Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 08:54:10 +0200 Subject: [PATCH 59/67] monitoring: recover host checks for the whole estate, reported to Gatus Five checks, 27 endpoints, replacing what Uptime Kuma used to watch: is it up every 5min, all hosts is disk full daily, all hosts is CPU hot every 5min, nodito is ZFS broken daily, nodito is UPS online every 5min, nodito Two roles, kept separate so neither knows about the other - they meet at a URL and a token, the same way caddy_site and each service meet at a vhost: roles/gatus_endpoint runs on the observability host, writes ONE file into /opt/gatus/config/endpoints/. Gatus merges every *.yaml there and appends lists, so callers compose without coordinating. roles/healthcheck runs on the monitored host: a check script, a systemd service, a timer, and an optional push. Ships a library of check bodies under templates/checks/. Everything PUSHES. Gatus never reaches out, which matters because nodito and its VMs are behind NAT, and because four of the five checks are internal state with no pollable surface at all. Liveness pushes too, deliberately: a heartbeat proves the host is running AND can reach the internet, where an ICMP probe from one vantage point only proves it answers pings from there. And since Gatus alerts when a heartbeat window expires, a check that stops running raises the alarm by itself - a dead timer looks exactly like a dead host, which is the correct reading. One bearer token per host, generated straight into the vault and never printed. A token only writes results for its own host's endpoints, so a compromised host can lie about itself, which it could do anyway. Three things learned from the source that shaped this: * Gatus polls its own config every 30s and reloads (main.listenToConfigurationFileChanges), so gatus_endpoint needs no restart handler - writing the file IS the deploy. * ...but on a reload it panics if the new config fails to parse, unless skip-invalid-config-update is set. Endpoint files are contributed by other playbooks, so one malformed file would take the monitor down at the worst possible moment. Now set. * The push URL uses a key Gatus computes, not the name you write: sanitize(group) + "_" + sanitize(name), lowercased with / _ . , space # + & replaced by "-" (config/key/key.go). So knots_box_local is knots-box-local in the URL. The playbook derives it rather than hand-writing. storage: maximum-number-of-results 900, up from upstream's 100. Gatus bounds the database by COUNT and trims inline on insert, so there is no retention job and no way to fill a disk - but history depth is then a function of check frequency, and 100 results at a 5-minute interval is 8 hours. 900 is ~3 days of liveness and ~2.5 years of the daily disk check. The uptime table is separate and its 30-day retention is hard-coded upstream. A bug worth recording: the first deploy shipped five scripts that all died with "syntax error: unexpected end of file". Jinja strips an included template's trailing newline and trim_blocks then eats the newline after {% endif %}, so the closing brace of check() landed on the same line as the body's last statement - `return 0}`. Every check was broken and the deploy still reported failed=0, because the role's "run once" task has failed_when: false and reports the result as a debug message nobody read. The blank line that fixes it is now load-bearing and commented as such. Verified by triggering every unit by hand rather than waiting on timers: all checks exit 0 on all hosts, and Gatus shows 26 UP / 1 DOWN. The one DOWN is liveness_watchtower, which is honest - that host currently refuses SSH (TCP connects, no banner exchange) and is excluded from this deploy. It is also the box still running Uptime Kuma. Known waste, not yet fixed: healthcheck installs its dependencies per CHECK rather than per HOST, so apt runs 29 times estate-wide for a curl that is already present, and daemon_reload runs 4x per host. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 453 ++++++++++-------- ansible/infra/400_host_monitoring.yml | 175 +++++++ ansible/roles/gatus/defaults/main.yml | 19 + ansible/roles/gatus/templates/config.yaml.j2 | 9 + .../roles/gatus_endpoint/defaults/main.yml | 28 ++ ansible/roles/gatus_endpoint/tasks/main.yml | 22 + .../templates/endpoints.yaml.j2 | 40 ++ ansible/roles/healthcheck/defaults/main.yml | 38 ++ ansible/roles/healthcheck/tasks/main.yml | 78 +++ .../templates/checks/cpu-temp.sh.j2 | 28 ++ .../templates/checks/disk-usage.sh.j2 | 19 + .../templates/checks/liveness.sh.j2 | 6 + .../templates/checks/ups-status.sh.j2 | 18 + .../templates/checks/zfs-health.sh.j2 | 52 ++ .../templates/healthcheck.service.j2 | 16 + .../healthcheck/templates/healthcheck.sh.j2 | 68 +++ .../templates/healthcheck.timer.j2 | 18 + 17 files changed, 887 insertions(+), 200 deletions(-) create mode 100644 ansible/infra/400_host_monitoring.yml create mode 100644 ansible/roles/gatus_endpoint/defaults/main.yml create mode 100644 ansible/roles/gatus_endpoint/tasks/main.yml create mode 100644 ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 create mode 100644 ansible/roles/healthcheck/defaults/main.yml create mode 100644 ansible/roles/healthcheck/tasks/main.yml create mode 100644 ansible/roles/healthcheck/templates/checks/cpu-temp.sh.j2 create mode 100644 ansible/roles/healthcheck/templates/checks/disk-usage.sh.j2 create mode 100644 ansible/roles/healthcheck/templates/checks/liveness.sh.j2 create mode 100644 ansible/roles/healthcheck/templates/checks/ups-status.sh.j2 create mode 100644 ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 create mode 100644 ansible/roles/healthcheck/templates/healthcheck.service.j2 create mode 100644 ansible/roles/healthcheck/templates/healthcheck.sh.j2 create mode 100644 ansible/roles/healthcheck/templates/healthcheck.timer.j2 diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index e8af624..d68ae8b 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,201 +1,254 @@ $ANSIBLE_VAULT;1.1;AES256 -39383462626535653332386339666338643666353535313064613633353230366534396665376235 -3061353662623030623038663063646661616537306531340a346430623235633264346163396661 -61333533646538373838383231353338313262656466613664393132356536396134306265633036 -3861643930326163610a303834613330326536616432646463323731666536383563306162616130 -61613665313665643265393938383664353538653937616230316136643765653632313932643235 -38346639646630396566656165376639383034323439386563376563666662363866303639343230 -37313938613261313962303666666431383864383963333833316531333065323630306165623562 -34303736646663373230303733356536656634656330316362303066303939376230393330376338 -30653666376330303234376132343731376464383061306139643237393463613565613561363335 -63336366613736616336643937326663383161376534626432646437633136646339313566626335 -32303638616466376535373331363564386534396164323538363664623965626635653438613436 -34313466363939343934636635363762626662636462356332646232633539323034646338643139 -34626462303937633032396334643232336363663732636132316136666232623865663434336433 -32363164626635356366323331366634613532666530643233643531613264393639333065396139 -39306132383462633138626231383463623335363130303565336662343933643235303331333335 -30373665303665393532613333383539323332363636366363616236343533373663333864653961 -63613465656465653763313635343263636430653964626662666434666634313664363930656165 -35313265643031653861653237643763663332383266646463633732616666346638333937613339 -61653265663765343764303439663230326234363961303835353535346332666561643266386632 -33633437336639663063373737356536326332336631303834363838346634646535366331616661 -38633235336361633132306330393131643530626661663733366132633939633637656231623438 -36646332363065643930343831316238333566323234643435346530323038626562666165373934 -35376136343566656436343438336363363364303965363533393330393430316566313033313737 -33303536323235346437316636643566306465393031353439666564313165663230643732323563 -38386265633938316462343636636666343638356338393061323134303837393231653435653833 -32616433343465623536653165623435383933656534366664343461363662313736623061616163 -65343664643736316337356134336236663434353635633032626664633333356661646433353566 -39643664366132623363336339313739346634616439396535346533396438646433323538653933 -36373639646566353931643630656432316166306531363562343666353633633063346263626532 -65343437656564633466306166366535623265363566323461626237333862623465653163643463 -38316338353233653131373336306537386638616462366463306464616566623932323633316631 -32386231313430343765306130616361343537393333616466663763633736303131356261626137 -38396634613734356465313865326361343432633066393861383363343262306662316662363330 -61393035666437366635396361313831626138396337616630346339643133366363616264393635 -63336339333862353838383437346165326238313632323862616161303330376464306265306661 -34623530656331623863376661373032333462616462646630303630626631623830343637386133 -61383236636534643932393262316330326161323435316162666566663235336639373431386463 -65363336313439323064333633353236313537613132396632343665616636396632363362366132 -62386538336162333062306136383737323336646163393161316164363561306461373731653334 -64313765313330653762336633396262626361353730353737646534383262663330363335633539 -33646534643231663464376638373934613432623639643061613262383635323133326564346263 -38303863333638353864636436623938326538323765623165633465353433663638363736316432 -66626264646335376432376561336261336633643039663533343562393736303164643833346566 -36656361356130303931666539363231633232393731653262666162396163376230643635636163 -35376163653036383364393638343663383937353239663030396663303431356632346339313931 -34326638333766353763623433393938663732373538323735323166613439663132653961366132 -63333465653035336238633665393134633436386263383862386166633436633834393034623231 -31613362346239313639346262633566373662623261343939313434616336663337306632656434 -37373331333163343035353731363462613436613833386561366430343336386261373835663237 -35643032323065313835353637666663323337626330353138386661383039383162336437343764 -31643037386465623138393236613166366335636633383762623762336535323936363134396436 -66323339383833613363653331643032616438383238643666653938613964393464613434623138 -32343835653332386135653762326238313833653264396266383830633536396131373839343332 -62316132326361646633653732663063366663646366366633353133663963356637663137663263 -62636261336463626439616334633464373566373864393965623166633733643730393664393131 -38636333326465386132636139356339613730366539323762653564616639656630623435656262 -39333331353933396534323562333665373233616135386336353334383436356630373931653237 -66656461353239346435643034643166363765623631373736656233373965373264636566313139 -63613839636565613663623831636465616335633636613437363066353961396631306332636537 -38393230616132316266613331643534363764666637663532353539333661636331623364623433 -31363433326536646439336131633961303737633364623961306466643836653133363235326439 -37616461346433623130616432353035653464366136353062633732313362643033323830626163 -64343866386162313133393637343365336363386563653436336438653836643831333064613461 -32646133636130303166333466363139386330653030326230626136643830666436653036323061 -61306530323666616332373635633132346635376237393237636638346433393131323533643761 -32613230663161643238623865613333323263303831366237386331353935636532343264653134 -61636136636538666461373537376539343136363466343766613961386231313336633438373965 -35666130373163373236396638303738306538653238343838386130663865366137613730616161 -31313330343262393632316635326134646662323535366338656434363162363461646362663761 -63653266653835333762353433633562346164623661653762666135363435376665663437333363 -63646663343030373566383665383864383463366363353630366138366433363362346634383633 -39396235303830626138363539333138363230313463303062383562633234616230333365333831 -66323632656336303231653138636164396532616265346261663839346235643439616466643438 -61653837336532316230313664623233353634643037653138313936653261363537656365323961 -66323563376363396336623164643866373833643138646361383132363537323862336262633766 -35643235613033333530346432376639353266346433386439323830363237666464306535386463 -36303032346234336335613764383165326536643439656563326465366136653964623066343036 -39333439343932346162373030343332373135343064346336353431303330666561623963366465 -38336162303337303235636131656564666536336139393537323831633739643531333838653432 -31633535386230306335616564653832656361643339386638623937326635643465643564623833 -34376633643830316635646131373039653932626231646536386165623435353861353639343734 -65393234646662363435636536346265336337616235376466613865383965316234383434303930 -31363932633765653335393262316363356265643533663762356235386437643238316334356131 -62333836666430616632613962373635633333333135656338323636346432636131613165663830 -34616338386232633066323131663464393963386538393632626265623133363637393537626535 -33303232643465353365626364306663363233333965643464393734626634353065663637363164 -66396261643531386463363931343531663733316336313537316538326436386662616338613765 -36363037323436353035353638613931373364393961336637636438306263376565346234366539 -62353530313335303734643731393636303532343362626361366162663362356463376636646661 -31633161633134626164363430306234366130396235613238646661393034656638313333343766 -39383266623234653338396637336435666134653164383463373734616364323731616462376634 -62643039623931393736383261393131333261623130333363636462343830626435623336666161 -38663936303531353932316430626261316437333536313230353431393262616463653430643136 -32626539316330323664643533653661323038313063623262346264393333343030656365653566 -38303835363837383539333265613865663930633535373230653763323437363535656265363633 -64653230646363373032653032336163353033336539303964663830323935393732393538373835 -34323938616333346139316361376463383531393134316262376332636637353834623937333264 -32633135386232656662383838623238323336633638636631386361306238333035636232376432 -39383634396635343032313164636132363331373830656632626461643464343631363838323233 -61653530616666313635383364633738346134356533336431313935353364376537666635633936 -64303764386538363438393430653236653862653165303631383434613037303639626461623066 -63383966343437646433656531333761393430636236326362656137663861663938323333313238 -34323433653363353565396636623030323633303536303366323030386566313661373063353435 -36346232353838663962353436316133643331383065346536663961373665343338373761653439 -63353463326130356565626164613836393931383236363638643565386230386363363236366536 -35653366393139616362633737616165663438653738636337313861313035353430393962373264 -64303138373339616436623664663638623666336637653662313131646630376134323437616166 -36366330303836396133393764653462316532313434633738316335626430653133633162623734 -31633465623766343566393431306535633033363937376638643865306239316565366162613366 -36323439663836373630643138303235313636663431663738653436646134636230393535623236 -30393036336137346630623035626239313661376539333237313261323438386632303939336566 -31346433623235326630613165323762363835343939623666613638616164626236646433373864 -39623866373238356138336266313135393063373733373332303338343965303538316431663462 -65613766623538633138616632366436316230646162323432313434393938356635663836633036 -62356631626138343436393238306539663532353431366162336130626334633230653638323134 -62313830343730653230346233626633386433313131663233303435373561343036393838633830 -61346135313130323464643531646139393633386461666630613637363833393466353737306530 -33653331346464613431643735396661346431653933346563643531626638653932316238356632 -30343864353431393231386639343237373130376639343236613337643739316132333133373962 -34393962393065363133353338613739613332653365316335306663643331383138323537613866 -35336339613138323031366436616139306339346363663662316265306161343935376539376166 -31316264353636343463636230646431333164626135663833316661363261663332366335326130 -36393438626531663538313931623132303332366538396666303939663061353865633936343162 -61303533383439626636646230623931383636633136336636643436646662396630303637333535 -63623661376130303134633761393835643266326133616236363464326238376466326266633763 -62656637663265363338636137346138616431616139653266386432373461316634326339333932 -31373562366132326635366164353637626334353464323764636463316665373936643030333539 -35646634643339653564303835333163383936386335353139636563363939323632343562633662 -65663033623232393430396139626135326236663463333639353338633736356430323036356537 -35326538616664386233333264313637626436613033643936343063383261653333353339306538 -38313631663263303230363830623235343732626638373535316537343938633632633534386365 -36623336343866383132373162626565313665393964663562653463396130623238643631363838 -38313363666635653134353663653461343135313138636434376362303639356362613639346238 -39343432613164643265333231366139326462313363633234313832386132323239323166613931 -37666135313036313236643261333739306362626236613435343766303162373464306531326333 -66363664306535666235636234346164373064373863646562353331346536613764646633333861 -65323361666430336462656238633735366661666662643632646562366535346432346233663330 -30336261636239613932303032653962376231333536633830646364663135626561336339393364 -30633532616163616363623835306335313163636337353263333836326261643334636666333836 -66653731316639376162386363303766323734653938373866306264376462643038386436323438 -62396136353961613633316661666136323037613863383338613633386536336330333363626237 -66376266623963353439613433633037623135333062636436656663633034343437346262333534 -65333331633037396239653838383462386136623764343436623737643564663261313834333133 -38343934636164323135653435653835333331623864383133313965373065643439623361613631 -62393036356164313665326537626363633839313935663636613434363934616135323036346131 -32363033323862613562353364373035653631366431303865646134316438366266643965623536 -37633835386439343932663861666230653463336332343864363035346135643535343535366135 -61653262363937373036616661333639383362383337316561303163643639306466353334353866 -38653862633732333962616135383234336233383434643264396461646264363134353531356537 -31643030386437303563313033373366366565633435616435643164626334643364613535363232 -64623835393434316131303237393038323930633861396431643833346636353534663362336264 -35326337303036303737336333636432353033366461653435626464646434646461303337643437 -34666635626335333133326663353234393637386330393964633836666364336164633030643534 -33346335643962323866306634383335613463633136636362396464383135666162333866333037 -34383461306332306132303063346634643838363835393564646537663236343432313766336265 -30336537313735386261376365623461303337303430613461373866366665623332343633643439 -62663366323165383463333864653132363632636134333932393333366466356536336666393330 -37346361393831363633623735363032646464303765353935643265373463323366353461316535 -33366265333863613531653663326162383431353932366231396332306234363963653566656130 -32313831653634353865313561363966653633313535363464326664396432623561653738313561 -62626239383562626139623865353039643365613565386337313666386364656537373165663833 -36353264613462393365396562636138663535613433653461303935646438653039653736616564 -66643737643437313366333830303839333965626164646637326331626265323035626464326462 -61343861316532666461336533656430633261393732323232313864623832323662613334386334 -65653837366532613538373138303633346566663330613965633963653864666330346532313038 -37616163626235303364303534383930376262643539633630336162383864386437633162323332 -37393236346334646139373130653438393663393335336331316336386236303535366237616330 -61346663323234626637343532666135633133343234393064323033623832613264623337356139 -39373530366564333437383266333636666363393461303034346566626562393065633732643564 -61393936316330313137386338323461353339323331383533376362323833666635383836333736 -65383738373537633164626237626364623866663031313864633032353137353331393765393339 -62653337623538623664326563316266613263376536653138656632303939626133653964616537 -39613763383938633163323234356139313261613332373861393462323930353066333134306532 -36303061363763316632616134353633323739633665313962623364356562366535326339646635 -62333838373833333133653338366638386566366337303232353537303961373030303464613964 -39626538353033383430333132306637363633353135333232613462656634626438646139653266 -61636365623634396438633731616665623333353735663036303664333461636536313733363232 -61656333613736363039613865383566343336346466366165323737336233303239333163326437 -66373231323333316630653039343731323332656131633131663635636433313732663935303232 -64346664353836383864333163353834313732376638303036323235303034373666653666383264 -35316537646237373535366262656532623331393566316235396631353030663961323238376236 -64303766636134626236646530616331323334613233633932653861333961346138646266363864 -36616563313131633931376230613861623331373839353733353730363234343339303037653663 -63646561346466386131626361326533363739383832383437393033376261323363643966396662 -63323234343931326663636662363837313665383962373965623335346330623735346665306364 -64333239303031666262316239613033303837386164333034616461316237376431643862366231 -63613935653533373530346131346265363565343166656563303164663132346430616234346235 -38343363633063326233306261643466383262346661363933316238303836373831623438653238 -39646438306639383561646666646137356561633131363563663066613263333234636466366636 -63323065346430353834636332383536643061356464333863356666333264356633306131303464 -34343230396339646466663963613333356235316638636463643333346262613637633936623839 -30666430353261363432653162353166326263396333653633663764376535643331336661626666 -64653433666638313838353766376663633564326633376461336234366534656531656638323164 -33636130333032393030633131646532323766383062646431646461353035636338363865366564 -32373664636362613539356665316461326638363631316664626435333238653135376134303035 -32396631666466363338626131393562316638393531633438613133623734323461326232383665 -65356538616538363065326264306236613330303439636232316339363333306238393766363361 -39396634366333636637643136653164623165613733373039366335323061636332363636336330 -61306237646637613961326564626334316334623835616365623132303039373930316137653433 -38306134383738346662 +63323431376238353966386463626539656230326233323861656165386335383832316631353236 +3731396264313366313166653861313736666435346537630a303363346131366433313264626135 +30633836613636393833333239666364623962383763343434353463343739633033383433306664 +6339616335366562310a393463663239333530633034373462313537376266393937373033346233 +62333331613063356535353134613538636663383166353731616633336366653864656333356330 +37323139616436353065363866633139323764336432656663326236636466356432333132656530 +39396566656161373365653738316162653134393234613130356561663464623564316161383566 +32356161646631623134313733343030353064666635346134653361366635316362316564323562 +38623236383137383665623934336631636537343361656539393538346665613234333638346466 +61653434653430636534636137613236326330356234616137303130666363323861623230626662 +37363839333334373561323863353563303861636137613338303736343064663232663064303466 +62396466363966303339643938653930313163323561633835396562636363633633646163363461 +61636331316161386630393766326431316239326563633033376538323330303863646162303332 +62306439666232626533306238343338343938626461363939326363333137306430643033373363 +65393164626166366663656637373330633939326361653261336339393135363934376164303537 +33326631396161393936656563336636643132666232343035633465383632613661633135343165 +33656162656334353636303130613231393835626166633666316561343165333138376439356539 +35356461353832643166613437343633346563393636323631353034366233323566613039343030 +65356438656664366434343638643963396563663434623961663432653334646639653435343262 +31366465366337633638393061643139643138316536396333653035613132623230646561373465 +65366332386565623161303563386237666538633433386438623535386564633937386434343264 +39653335343235386439383964356664313339356531323732363362643566363964653634393039 +37643338316264633733383531633934373132643034653433316438303962356139376364306466 +37613232636365623732313233303766373566623162303965353163373131363763346135313230 +39316466613136323039373765613336333835323465323737323535393736366433343664616539 +32303839376538393238613763366433346433643436613662383062306366643561626164363430 +38353738653734393237663765613733633230343965363732643336333337646438623562303561 +34373031623466343539663738613561346665316163623631643236633236633765616361623764 +35343465346435656533336464393633616663343239343162343664363763373736656431663433 +62306565643862613065386265613036326336353130633530373166343966343033346665346235 +37333761613437393236393935646239373930613239323639616564383336376664386532366236 +31313862363232656537613733666239373433343835633964333164633335656436373766353838 +36303234353333333233313531396464336262663864633638306236343633336432333737376539 +36623439336133356362626632663966336162646636393932353537356638336337346663303230 +66623964656133333534313231656337336463346363363635396135393062643736336133393134 +66376233376161373561386531633639376137643732613535333066646666616133623963323137 +63356565383430656535633639643363653232323737363566663832633931616230343132663563 +37633435626266393937373831366632643535623634386131343065353036653163326464666364 +39613732653739643436303665363132336235333339653630353335646432306432643235376139 +33643935366365383232646234313436383534353130633039376562363939363033643830303936 +64376131346631306263363137316435343661393562386638363636336261623831616232313934 +30373133633034626430313936373537323366626562323239623538306130623135333562333333 +33303761623232666165323438643364313530316563396466653331636531366166633964353033 +61633861363833613264353036383766343233646439393764636664353938656335623330323062 +64306565626630326632383963353266383464393530623639333836663739306132346336373936 +36636532623032666433303339643062383663646536636663383662646237323336386366386138 +31643930356330343138323132613238333837356137643061346364653134346165383661366663 +31643630353631656131633430323838383233323936623661613466663761313264383338373636 +63323062646334666665633662366364383434393561313863653862633533356166396336303133 +61383637326535656539656538353238326337353138616533313534643131346163356533396331 +30303031656133623735366163323664626362356365663730306438653132396461393539386138 +65343865633765306337663830343931376265666362653361616662666665386138313462633432 +32623436643239363437373161353162626663623332373163656133613465333139356564383130 +36653433653535326433313337636635336132663636383764653237356432313362393535306630 +32663738663835336131353737303262313966663264303264313864353663323733653334313263 +32333866613335666436303236393536633835653837363133666437303736303630633939616232 +65316337646665343062353864623839613263356335353938386136666166363565303062633331 +39633761316630656662636136393834626564616139396336663363373931666132383062626536 +61376462333131633634383130393765353037653536653837373032663636623031316530613961 +63633861613366373235653735616333303262333765636531353734633664383766623261636238 +32623132336137666366643035306562643833333537633666343637366230623366666230366638 +37333238336435636235333063636538666635666438383239336164613536666262343132646634 +38633132343966663564326138346561313731623966356435653937383431396261313138656261 +65326265323333366538656335306431396666643663313561663964626436356464356336323661 +66663763396166313831623534353131633566346634613738316634653935373038303730336662 +34656634343834623930623833386632306136323738646535396539643463393861343832333065 +35333464316163626130366233663439633435333462353534616530316464666537613436386637 +37363939333834626638396337636234343561313565613636643463393132626466333632636463 +31326261656537616130613634323736313132633361653162666631303965613036653434653236 +33313037313034356139353261666138386636663637353831666330616563323536303834623863 +30346539376536396337313561396561363237613865623533633063613936316162353138306164 +64646634396236646535633135613036376538343964363663616234666532386165643938396436 +36333832313136363965383334323430313630663336343562626365633461366133653931376139 +38613364383236393634386436643733333433336563306138356337636337623239646164306539 +61336433383339353861313131343166353265393132356631366230343438646239323730613865 +35333438326236353264386436373338303532646336333161636232393235653830303237323137 +36323464303634663839326230613539376238323931663137616363663434373662636535373266 +38363261653937633363316665613130386135626135333662356261323462303939383962363462 +37663333386336626261663464393561666135633537616365376665393664346466633932656635 +65343762656162613831643265633562373865616662313631313034363964626538383633656363 +61363639343663643638353935643635643663636139313963303739343830613336343661663733 +32323463626264663639396130343930383162373133386232613635633934393334626631303561 +64346636636330623538343534663037363635323138343566303664613337346638353637306336 +36386432326661316633303764323761303363646337303539633831666665373833666663333766 +62363461616462383663316434373561653063323062356131643930633838336364643065613838 +66386565373635396138623832646137636236646335386463326265326565353566346136306564 +36323666333331396636353762363866616239343362313431313765373334386637633366623936 +31333432663331623232373330306430626264333761373362303365656262303164656439623637 +36663065376137336435646232323262646439623266333534623766353035376663353535333137 +38373335313833633239313963613439316433643832653938386434323666333263373437663337 +32613661666664366636616562396232366237333237316430346565653066623561636263343265 +34313137643933366239353836383265633030373636393232663534393036343130633438643331 +36643236636232323532363036646563646436613630323638356534373831373737386139393664 +62616133323861333265626665616331333665356666643734666537356565393334373036663330 +66346361363361303538663936616664663864316530303338316634616636353235353161366135 +33363630626566336363336630656331343437393666353262396137386163633134343238623361 +61306163333831626164326461616630636266633666383765356539656536363133393537333263 +66646635343437643762373432333164626334333730373762336565636533333132363965636334 +64316232623437353263656131633335396166616133653234623432386234666531356431373333 +63316462643464626362646463323935303765643061343535393065333961303931663564353735 +33336565383332303637383538316237343430626432393533376166323565303263643435313136 +30626165623765346430373165323030386635376338666235306534643730656564653531633765 +36663864396638666165303837646236386434326166663365356164353036646464363132356261 +66373861343834643161343665616139373466346130353233623135656663373230333630306132 +66653233623632303463343261653764636230333038623936353138356565623061643432346265 +36326263613833386237333566333534636238336539303638643233376331616331626636666635 +66303535383838336135306439306239323531343331343832376636363931626663333337616464 +61656532623437363634343039313534356565383361303562336666383561333235303939303339 +65326336633631313133396263326139636535346166373333343934316435363435353236363331 +30396231383336326461326462393633353739346463653636393331313534613466363738376263 +62663264643833333032343839373165353262663637376635373430333631316532336335653038 +32303835333461363635306664653331613164316264613632326131633639666263633234306336 +30656562383439333639623534303164323231626337396362373235383530323731626139333335 +38386238396565643533303030393364373735376564373765373632386335613432313735336132 +63306231373038366131353934393735323932633233646439383666366333653535373634386333 +31643130376266626236353937633235623765323462396130383831376366313337643939363366 +62326439623135393539366133386337353964306637343238633730363639373831633662653565 +64333134336666643465633565333765613835373765663664653331353935666437633566386165 +61656239653264646530306165396234626365326238616566303831353935626362366338613339 +30303364356137323935626662636663383761383935343531666461656537643333346637666365 +37663836386466303433313339663531373432643732333461613739636233336566356139323934 +36663965626330333436373764613730393365616166653866306336633939393765633564626331 +31353131663038323235396564623234396138386237663030316530353337373934323232633433 +39363931656466313639346265363466646631363032353363383662306436346162363431353833 +36303934323364363465343236633064336239643565393034353934373239376139386535373061 +34643365393866636430356338626438386136646161376538363762336265653632643334336331 +38646131386463666135376162613864636534366264336630356135316637393135646630333733 +64343731633836306565623238313936656164653038393236356130306162346331396231393436 +38303435383963626434323230643832303838646239646263353737323166613965623734313933 +38313164313062373263326138656239366135383361616436333539356331383064363562376366 +31333838313664653432373937306239363631626234336136396364396166656234623365616237 +36363434393537396639633062626639353738386232393066333034343132303831366362623031 +36643862376637643739626662656239633361653933313130646661656332316535396230386162 +39373738366138373130643636663339613732626532383465316365363638386361623838656630 +66636262666536303739616361343763366135313835353938323330363635343135633138306361 +61346433336366643430646334336136346136646166363962613336366239653236373135376138 +61316464646264316532333839626634623165336334323836643130323137303632353232616631 +32653036313133333237363233653932366334623133646565343461373132306365313331373335 +33623561623163383562366137663733343833333136353738386535313439386164346164383565 +32393331373130633065323266623465613732353431623234653435383133373537363436346463 +63323666666136613034643237623463383462326334343334303731363961303438646330646562 +64336132663561353565376361306331363138306638353834613231323331356238343730616164 +35666432626436376539373633383466343732616430333661613865653530366439626131303233 +63353633613232636238316532633231633539633363623734316337633764333430366466626430 +34336534616537656531613538356231343061326462333663653662343762666131393765306462 +34353435626337656362663637613763623537623534666336643833643963626265303266623434 +32323836663432313438363530663839653637346562626566316163373137363063633564323731 +64336638616661366437656431396435303439373132383330646635306435646138633531336461 +39663366643961303832323838643530653632383230373366396531646639336138373434396431 +32383064313533393735336235356432646162353236343365313435313666363935333835336331 +39316330306336396637613634306531346531393036326536653961313438623664643362623937 +36363535383562663535343332363064326530636335343963366166333365613234643035343764 +62366334623332653931393836303934333139636635363638303734663230363537663338383461 +34366661313931316639303762373131636530363232386263313361666561313034343033626333 +63386630396136316338653134306632323664396466343361306636623164616562393430353732 +32646537653963383639616530646561653065386631386138616130343936333935393861643161 +36613166633561646138336539373731373939633234633861636634396563646137316134636463 +30363133633131303632316634376361623637373133393234393362386338626536653438666539 +31633837353935346438316133333136326237303430313533643265363966396537373161346236 +33656163636338363565653834636461333837653433633639393361636565613337653766353765 +36613834633631346136366232663539636566666339343939383732306537396465373932646434 +31663736323535353763623633366535333262636234316363366537393531666430633631373364 +36393766343863353135333864656536393739636563363631306362336665616636303832666330 +66623336663566316330333337343064626366323563613463646338613433623637363064656639 +63626465613963333932346131656639653239353034613862323565666436316338666563623836 +37356465633635363834316564313839366137656232346637663231373261393035623035643065 +30316663613732616237376265363561326164313631366466653139323534623531323537623164 +35383365336438356538343934356134646337626635326266336533386262323262323734663534 +35313239393232393435346531643138376362336437323963613933623739343463613831326539 +61643939393461386534656664366361363062643939393962376664353563643635373336333266 +39383132363964656438623031366336306362353736633634353033656539376233343131653030 +62653736663637653166373930313664333930346262396339386432353066646465373937393964 +35656538373064643936323731393436386665353838313563613832323834316539366163373638 +30623365333266393331396135613663393139303136663636383766613731636435386635666132 +30633664303830353137393438303666656235616332393132613364333238646566303363363365 +61636132663366666139363965663239623764366334333432633830316636303263373331313931 +30323337383931353363343231656632663534323435353338646664393635323733623166343830 +39633865336666363635346462306261613935383666396261653531396164383961383830333434 +39303063383365306435303430633834386566316332313864373664383766646463393734343961 +34656339343761613863363831393933396438636332333339393433636435316330393634353032 +66303139333933393164363735303534346332336564333366343032353631396633366337333831 +63616637313262316335386230343034323038613530663432353764656337343638383135316361 +32376339333235316335656638613434306633316631376230383434303234666532303662333630 +32626562656363633837316335643162363232623265396362653264313934336566366130376634 +63633132626539353138373263313135633265623761393063383136376130646132643664343431 +32616136306335336636343434366236376536663730643638636234623136383262643766613137 +34666431333265633063356566623139666266643138653365613961613532343337346161643337 +65316332393762386133633030366534353763323738303437636537326137306365626632383630 +34363566643436376535386134383638313237643934303931653839616637643036373134363830 +66656463303331626430663063356239373434386361323139343838333763343565396637356561 +63386436353934646336616537323233336339326261653466653135653735316463653666343231 +36343261616338303864343162646338323339633634663032306433313138376539643236393462 +35333734306662393836386231306466616133383762353464343965376132643634393237396164 +34373831316634626133353735376661633464373238383561316565616331666431356165623039 +34386130363761663365666238323534393933313866333030643132346563646463393063333764 +64646432616231343032633965643337363234393435346430303931633665623362306465383962 +39383437623761613339353733643862376462623166636264303833666437393231656437663764 +65313065323134623265333466373461306639326465326362646532643037333230373837653230 +62383033363561343735623639626565333666636563306639643139666134383762353261343864 +34386334306237363762313465333163643037626534613830663463653563663537396231356431 +36646139333534643765653732303330343532373933656537633465326632663539373638323865 +33386363333065373565336566613333313738313437613764663664633032393865313331633433 +65336664386437353034366638643533366365393864363033366363626630336132313038653430 +30396539313865646333666462643061666566313033613634326537363132663462623861666137 +66386663633634396163666164316631393635623938333137653361363963303235363732623930 +30376632376264343036643461613438396531373865613162643734303963373539363464343337 +39653066626635626164613435643738616534623064306338303734653830633737386531393435 +61303461666666363230643638626230353839653962663439353166346635376438623832663637 +35663738356136363165653133323666383935363566616135376161356165316239373661643039 +35353666326336623561356635343232393137383536356565353762616639383932313464383063 +35393734306636313561383133393961646132633639363136326332366338633535666139643634 +39303735653564656232326339313238383137333630383539623139356539323561353366323131 +37393130356639623133653134623131633031393633386330653034353166376466633739303638 +66613534393331383438326230313936653532336134643632306634323530343630363236646235 +38656463343932333637613866386232376162623939626466303063306466623132643731623338 +63656662376132653238343537343265373236306134376565323961323264383830363065323738 +33366137323535633732316563386237646230613339666337386162633166353533346565393337 +34316233323734643262663532373964346331393031356134373266616265373564386236616237 +38373730623261346365353363383833633964383132623361343333373637386339313262396633 +37666437613361646337323039343938353466346538356138376130313337633533666137626266 +66313838323638626230306138623135363266346366316339313164626233656237376266353762 +65356461346239613066613038353735626233383939666330393131363064363337356330363435 +39656633616166356430343564366433323864333236623934356335336661346338653834363362 +39343435653934643132363931363334366631646463306261666537303938363633336464666537 +38336632643561633461383965333264376462306131666232623735313265373832343762393434 +61386534393034353363613230626330393234326234363837393738376634633561613562323137 +66613930306330623434323539366333663364396165663466303365653331363431656365353266 +63623430353734616439393735623564626638313336613636383438363531306234663939383066 +37623165333233643465363334663433663034636438613433633966306334376462316233343338 +61333530376236333134306164323263633266376666663030303438646335393661316137646237 +32383865636433366635323134663938383339393933656438633662333334313264643338636563 +64343533643432386164333630366531383434333231393762613136616435376534653530346364 +64396531383363376437613366613066633534616163396133323835323431353034373563306536 +35616335636463623565636534346435643463376330333962353261663062613034653863373834 +37653566383164373263616265643536343037346464633930303935393337336333616338633730 +35316166656164356535613364386366373666306131373063376465663935303530666432383435 +65393934396639333765313933643263306337623635623930656430343361653861653039323861 +62326165313038313137323539343934366134383630363632653939633331626566653663613666 +30336161616136613034353133663738646464306164663931373365343664373337303564643565 +31623132396431336130396236656136656335336236306364656164353431633136343732663631 +37316366633837323961643338343538653933306664356236373165333464643032623864633538 +32613030313563343930653863346638393662303030396537343264326234643735323532376337 +61313936623663323961333364306664613331626233393430626632373765333832616136333065 +65323837373231633439333437616536306466656530646332386164373963386431653532666262 +62386234393432633430636439663331386235366630333630383336363664663333383331386362 +31376362643263396634623662373134343039323663343433383836663261376463306337656236 +32633536336238326337336666313263613433353739333530316630653735636133303635313436 +66383761613932653132663939353734623663333736666462363235333336333733323963623663 +33616265636664373638656363636363656639373634353732366664363565383737313863666139 +35303364343063663764393261333336373864373839666664356166626238363035343163653131 +63316436373331393461626164346362366530636335613966353335376334643433333963396563 +39373364356439663363333566656565616130643037613332363937313964363433613436666230 +6232 diff --git a/ansible/infra/400_host_monitoring.yml b/ansible/infra/400_host_monitoring.yml new file mode 100644 index 0000000..ac81289 --- /dev/null +++ b/ansible/infra/400_host_monitoring.yml @@ -0,0 +1,175 @@ +--- +# Host-level monitoring for the whole estate, reported to Gatus. +# +# Every check here PUSHES. Gatus never reaches out, which matters because nodito +# and its VMs sit behind NAT, and because four of the five checks are internal +# state with no pollable surface at all - disk usage, CPU temperature, ZFS pool +# health and UPS mains status cannot be observed from outside the machine. +# +# Liveness is a push too, and that is a choice rather than a limitation. A +# heartbeat proves the host is running AND can reach the internet; an ICMP probe +# from one vantage point only proves it answers pings from there. And because +# Gatus alerts when a heartbeat window expires, a check that stops running +# raises the alarm by itself - a dead timer looks exactly like a dead host, +# which is the correct reading. +# +# Each host has ONE bearer token, shared across its own checks: a token can only +# write results for that host's endpoints, so a compromised host can lie about +# itself, which it could do anyway. +# +# The push URL must use Gatus's own key format (config/key/key.go): +# key = sanitize(group) + "_" + sanitize(name) +# where sanitize lowercases and replaces / _ . , space # + & with "-". So +# knots_box_local becomes knots-box-local in the URL but stays readable in the +# name. host_key below is the Jinja equivalent; do not hand-write these. + +# ───────────────────────────────────────────────────────────────────────────── +# Register everything with Gatus. +# +# This play runs FIRST on purpose. Gatus reloads its config within 30s, and the +# host plays below take minutes, so every endpoint exists before its first push +# arrives. Registering afterwards would 404 every first report. +# +# Heartbeat windows are several times the check interval, so one missed run - a +# slow apt run, a reboot - does not raise an alarm, but a check that has +# genuinely stopped does. +# ───────────────────────────────────────────────────────────────────────────── +- name: Register the host checks with Gatus + hosts: observability + become: yes + vars: + monitored: "{{ groups['managed'] | sort }}" + + tasks: + - name: Build the liveness endpoint list + ansible.builtin.set_fact: + liveness_endpoints: "{{ liveness_endpoints | default([]) + [{ + 'name': item, + 'group': 'liveness', + 'token': gatus_push_tokens[item], + 'heartbeat': '16m'}] }}" + loop: "{{ monitored }}" + + - name: Build the disk endpoint list + ansible.builtin.set_fact: + disk_endpoints: "{{ disk_endpoints | default([]) + [{ + 'name': item, + 'group': 'disk', + 'token': gatus_push_tokens[item], + 'heartbeat': '30h'}] }}" + loop: "{{ monitored }}" + + - name: Register liveness endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: liveness + gatus_endpoint_external: "{{ liveness_endpoints }}" + + - name: Register disk endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: disk + gatus_endpoint_external: "{{ disk_endpoints }}" + + - name: Register the hypervisor endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: hypervisor + gatus_endpoint_external: + - name: cpu + group: hypervisor + token: "{{ gatus_push_tokens['nodito'] }}" + heartbeat: "16m" + - name: zfs + group: hypervisor + token: "{{ gatus_push_tokens['nodito'] }}" + heartbeat: "30h" + - name: ups + group: hypervisor + token: "{{ gatus_push_tokens['nodito'] }}" + heartbeat: "16m" + +- name: Deploy host liveness and disk checks + hosts: managed + become: yes + vars: + gatus_api: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints" + host_key: "{{ inventory_hostname | lower | regex_replace('[/_.,# +&]', '-') }}" + host_token: "{{ gatus_push_tokens[inventory_hostname] }}" + + tasks: + - name: Is the host up? + ansible.builtin.include_role: + name: healthcheck + vars: + healthcheck_name: liveness + healthcheck_description: "Liveness heartbeat for {{ inventory_hostname }}" + healthcheck_check: liveness + healthcheck_interval: "5min" + healthcheck_boot_delay: "1min" + healthcheck_push_url: "{{ gatus_api }}/liveness_{{ host_key }}/external" + healthcheck_push_token: "{{ host_token }}" + + - name: Is the disk packed? + ansible.builtin.include_role: + name: healthcheck + vars: + healthcheck_name: disk-usage + healthcheck_description: "Disk usage for {{ inventory_hostname }}" + healthcheck_check: disk-usage + # Daily. RandomizedDelaySec spreads twelve hosts across the hour rather + # than having them all report in the same second. + healthcheck_on_calendar: "*-*-* 07:00:00" + healthcheck_randomized_delay: "3600" + healthcheck_boot_delay: "5min" + healthcheck_push_url: "{{ gatus_api }}/disk_{{ host_key }}/external" + healthcheck_push_token: "{{ host_token }}" + +- name: Deploy the hypervisor-only checks + hosts: hypervisor + become: yes + vars: + gatus_api: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints" + host_token: "{{ gatus_push_tokens[inventory_hostname] }}" + + tasks: + - name: Is the CPU hot? + ansible.builtin.include_role: + name: healthcheck + vars: + healthcheck_name: cpu-temp + healthcheck_description: "CPU temperature for {{ inventory_hostname }}" + healthcheck_check: cpu-temp + healthcheck_packages: [curl, lm-sensors] + healthcheck_interval: "5min" + healthcheck_push_url: "{{ gatus_api }}/hypervisor_cpu/external" + healthcheck_push_token: "{{ host_token }}" + + - name: Is ZFS broken? + ansible.builtin.include_role: + name: healthcheck + vars: + healthcheck_name: zfs-health + healthcheck_description: "ZFS pool health for {{ zfs_pool_name }}" + healthcheck_check: zfs-health + healthcheck_packages: [curl, jq] + healthcheck_zfs_pool: "{{ zfs_pool_name }}" + healthcheck_on_calendar: "*-*-* 07:20:00" + healthcheck_boot_delay: "10min" + healthcheck_push_url: "{{ gatus_api }}/hypervisor_zfs/external" + healthcheck_push_token: "{{ host_token }}" + + - name: Is the UPS online? + ansible.builtin.include_role: + name: healthcheck + vars: + healthcheck_name: ups-status + healthcheck_description: "UPS mains status for {{ ups_name }}" + healthcheck_check: ups-status + healthcheck_ups_name: "{{ ups_name }}" + healthcheck_interval: "5min" + healthcheck_push_url: "{{ gatus_api }}/hypervisor_ups/external" + healthcheck_push_token: "{{ host_token }}" diff --git a/ansible/roles/gatus/defaults/main.yml b/ansible/roles/gatus/defaults/main.yml index 95c41ae..9c7cb12 100644 --- a/ansible/roles/gatus/defaults/main.yml +++ b/ansible/roles/gatus/defaults/main.yml @@ -57,6 +57,21 @@ gatus_storage_type: sqlite gatus_storage_path: "/data/gatus.db" gatus_storage_caching: true +# Per-endpoint row caps. Gatus bounds the database by COUNT, not by time, and +# trims inline on insert (storage/store/sql/sql.go, InsertEndpointResult) - so +# there is no retention job to write and no way for this to fill a disk. +# +# History depth is therefore a function of check frequency, not of days: +# 900 results is ~3 days of a 5-minute liveness check, and ~2.5 years of a daily +# disk check. Upstream's default is 100, which would have been 8 hours of +# liveness - not enough to still see a weekend incident on Monday. +# +# Note the uptime table is separate and its 30-day retention is hard-coded +# upstream (uptimeRetention), so uptime percentages top out at 30 days whatever +# this is set to. +gatus_storage_max_results: 900 +gatus_storage_max_events: 50 + # ── Alerting ───────────────────────────────────────────────────────────────── # Pass-through: rendered verbatim under `alerting:`, so any provider Gatus # supports works without touching this role. Empty means "check and record, @@ -69,6 +84,10 @@ gatus_default_alerts: [] gatus_basic_auth: {} gatus_maintenance: {} + +# Keep serving the previous config if a contributed endpoint file is malformed, +# instead of panicking. See the note in config.yaml.j2. +gatus_skip_invalid_config_update: true gatus_log_level: INFO # ── Self-check ─────────────────────────────────────────────────────────────── diff --git a/ansible/roles/gatus/templates/config.yaml.j2 b/ansible/roles/gatus/templates/config.yaml.j2 index 67d6244..6a2a72e 100644 --- a/ansible/roles/gatus/templates/config.yaml.j2 +++ b/ansible/roles/gatus/templates/config.yaml.j2 @@ -23,12 +23,21 @@ web: address: 0.0.0.0 port: {{ gatus_port }} +# Gatus reloads when its config changes. If the NEW config fails to parse it +# calls panic() - unless this is set, in which case it logs the error and keeps +# running on the old config. Endpoint files are contributed by other playbooks, +# so one malformed file would otherwise take the monitor down, which is the +# worst possible time to lose it. +skip-invalid-config-update: {{ gatus_skip_invalid_config_update | bool | lower }} + storage: type: {{ gatus_storage_type }} {% if gatus_storage_type != 'memory' %} path: {{ gatus_storage_path }} {% endif %} caching: {{ gatus_storage_caching | bool | lower }} + maximum-number-of-results: {{ gatus_storage_max_results }} + maximum-number-of-events: {{ gatus_storage_max_events }} ui: title: {{ gatus_ui_title }} diff --git a/ansible/roles/gatus_endpoint/defaults/main.yml b/ansible/roles/gatus_endpoint/defaults/main.yml new file mode 100644 index 0000000..7d49d7c --- /dev/null +++ b/ansible/roles/gatus_endpoint/defaults/main.yml @@ -0,0 +1,28 @@ +--- +# One invocation writes ONE file into Gatus's endpoints directory. Gatus merges +# every *.yaml under GATUS_CONFIG_PATH and appends lists, so each caller owns +# its own file and they compose without coordinating - the same shape as +# caddy_site, where each service contributes its own vhost. + +# Filename stem: .yaml +gatus_endpoint_name: "" + +# PULLED endpoints - Gatus makes the request and evaluates conditions. +# - {name, group, url, interval, conditions: [...], alerts: [...]} +gatus_endpoint_pulled: [] + +# EXTERNAL endpoints - the host pushes its own result. Gatus never reaches out, +# which is what makes this work for machines behind NAT and for state that has +# no pollable surface at all (disk usage, ZFS health, UPS mains). +# +# - {name, group, token, heartbeat, alerts: [...]} +# +# `heartbeat` is the important one: if nothing reports within that window Gatus +# alerts. That is what makes a push check detect its own failure - a dead timer +# looks exactly like a dead host, which is the correct reading. +gatus_endpoint_external: [] + +# Where the files live. Matches roles/gatus. +gatus_config_dir: /opt/gatus/config +gatus_endpoints_dir: "{{ gatus_config_dir }}/endpoints" +gatus_gid: 10001 diff --git a/ansible/roles/gatus_endpoint/tasks/main.yml b/ansible/roles/gatus_endpoint/tasks/main.yml new file mode 100644 index 0000000..40f3448 --- /dev/null +++ b/ansible/roles/gatus_endpoint/tasks/main.yml @@ -0,0 +1,22 @@ +--- +- name: Assert gatus_endpoint parameters are sane + ansible.builtin.assert: + that: + - gatus_endpoint_name | length > 0 + - gatus_endpoint_pulled | length > 0 or gatus_endpoint_external | length > 0 + fail_msg: >- + gatus_endpoint needs a name and at least one of gatus_endpoint_pulled or + gatus_endpoint_external. An empty file would contribute nothing, and a + config with no endpoints at all makes Gatus refuse to start. + +- name: "Write Gatus endpoints '{{ gatus_endpoint_name }}'" + ansible.builtin.template: + src: endpoints.yaml.j2 + dest: "{{ gatus_endpoints_dir }}/{{ gatus_endpoint_name }}.yaml" + owner: root + group: "{{ gatus_gid }}" + mode: "0640" + # No handler. Gatus polls its own config every 30s + # (main.listenToConfigurationFileChanges) and reloads itself, so writing the + # file IS the deploy. A handler here would also fail whenever this role is + # used without roles/gatus loaded. diff --git a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 new file mode 100644 index 0000000..1c84996 --- /dev/null +++ b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 @@ -0,0 +1,40 @@ +# {{ gatus_endpoint_name }}.yaml — managed by Ansible (roles/gatus_endpoint) +# +# Contributed by a playbook, not hand-edited. Gatus appends the lists in every +# *.yaml under its config directory, so this file adds to whatever else is +# registered without knowing about it. +{% if gatus_endpoint_external %} + +external-endpoints: +{% for e in gatus_endpoint_external %} + - name: {{ e.name }} + group: {{ e.group }} + token: "{{ e.token }}" +{% if e.heartbeat is defined %} + heartbeat: + interval: {{ e.heartbeat }} +{% endif %} +{% if e.alerts | default([]) %} + alerts: +{{ e.alerts | to_nice_yaml(indent=2) | indent(6, true) }} +{% endif %} +{% endfor %} +{% endif %} +{% if gatus_endpoint_pulled %} + +endpoints: +{% for e in gatus_endpoint_pulled %} + - name: {{ e.name }} + group: {{ e.group }} + url: "{{ e.url }}" + interval: {{ e.interval | default('60s') }} + conditions: +{% for c in e.conditions %} + - "{{ c }}" +{% endfor %} +{% if e.alerts | default([]) %} + alerts: +{{ e.alerts | to_nice_yaml(indent=2) | indent(6, true) }} +{% endif %} +{% endfor %} +{% endif %} diff --git a/ansible/roles/healthcheck/defaults/main.yml b/ansible/roles/healthcheck/defaults/main.yml new file mode 100644 index 0000000..d29cbd1 --- /dev/null +++ b/ansible/roles/healthcheck/defaults/main.yml @@ -0,0 +1,38 @@ +--- +# One health check: a script, a systemd service, a timer, and an optional push. +# +# The exit code is the answer and systemd keeps it: +# systemctl is-failed -healthcheck.service +# Reporting anywhere else is optional and generic. Point healthcheck_push_url at +# Gatus, or at whatever replaces it, or at nothing. + +healthcheck_name: "" # e.g. disk-usage -> disk-usage-healthcheck +healthcheck_description: "" + +# The check itself. Pick ONE: +# healthcheck_check: a template under templates/checks/ (without .sh.j2) +# healthcheck_command: a shell one-liner that exits 0 for healthy +healthcheck_check: "" +healthcheck_command: "" + +# systemd timer. OnUnitActiveSec unless healthcheck_on_calendar is set. +healthcheck_interval: "5min" +healthcheck_on_calendar: "" +healthcheck_boot_delay: "2min" + +# ── Reporting ──────────────────────────────────────────────────────────────── +# Gatus external endpoints: +# POST {url}?success=true|false&error=... +# Authorization: Bearer {token} +# Empty url = check and log only, which is a valid state and not an error. +healthcheck_push_url: "" +healthcheck_push_token: "" + +healthcheck_script_dir: /usr/local/bin +healthcheck_log_dir: /var/log/healthchecks + +# Per-check knobs, consumed by the templates under checks/ +healthcheck_disk_threshold: 85 # percent +healthcheck_cpu_temp_threshold: 80 # celsius +healthcheck_zfs_pool: "" +healthcheck_ups_name: "" diff --git a/ansible/roles/healthcheck/tasks/main.yml b/ansible/roles/healthcheck/tasks/main.yml new file mode 100644 index 0000000..fb0257b --- /dev/null +++ b/ansible/roles/healthcheck/tasks/main.yml @@ -0,0 +1,78 @@ +--- +- name: "Assert healthcheck '{{ healthcheck_name }}' is fully specified" + ansible.builtin.assert: + that: + - healthcheck_name | length > 0 + - healthcheck_description | length > 0 + - (healthcheck_check | length > 0) != (healthcheck_command | length > 0) + - not (healthcheck_push_url | length > 0) or (healthcheck_push_token | length > 0) + fail_msg: >- + healthcheck needs a name, a description, exactly one of healthcheck_check + or healthcheck_command, and a token whenever a push URL is set. A push URL + with no token would report to Gatus and be rejected 401 on every run. + +- name: Install healthcheck dependencies + ansible.builtin.package: + name: "{{ healthcheck_packages | default(['curl']) }}" + state: present + +- name: Create the healthcheck log directory + ansible.builtin.file: + path: "{{ healthcheck_log_dir }}" + state: directory + owner: root + group: root + mode: "0750" + +- name: "Install the {{ healthcheck_name }} check script" + ansible.builtin.template: + src: healthcheck.sh.j2 + dest: "{{ healthcheck_script_dir }}/{{ healthcheck_name }}-healthcheck.sh" + owner: root + group: root + mode: "0755" + +# The token is in this unit file, so it must not be world-readable. +- name: "Install the {{ healthcheck_name }} systemd service" + ansible.builtin.template: + src: healthcheck.service.j2 + dest: "/etc/systemd/system/{{ healthcheck_name }}-healthcheck.service" + owner: root + group: root + mode: "0600" + +- name: "Install the {{ healthcheck_name }} systemd timer" + ansible.builtin.template: + src: healthcheck.timer.j2 + dest: "/etc/systemd/system/{{ healthcheck_name }}-healthcheck.timer" + owner: root + group: root + mode: "0644" + +- name: Reload systemd + ansible.builtin.systemd: + daemon_reload: yes + +# `restarted`, not `started`: started is a no-op on an already-active timer, so +# a changed interval or a stuck timer would never be picked up. +- name: "Enable and start the {{ healthcheck_name }} timer" + ansible.builtin.systemd: + name: "{{ healthcheck_name }}-healthcheck.timer" + enabled: yes + state: restarted + daemon_reload: yes + +- name: "Run the {{ healthcheck_name }} check once now" + ansible.builtin.command: "{{ healthcheck_script_dir }}/{{ healthcheck_name }}-healthcheck.sh" + environment: + HEALTHCHECK_PUSH_URL: "{{ healthcheck_push_url }}" + HEALTHCHECK_PUSH_TOKEN: "{{ healthcheck_push_token }}" + register: healthcheck_first_run + changed_when: false + failed_when: false + +- name: "Report the first {{ healthcheck_name }} result" + ansible.builtin.debug: + msg: >- + {{ healthcheck_name }}: {{ 'HEALTHY' if healthcheck_first_run.rc == 0 + else 'UNHEALTHY (rc=' ~ healthcheck_first_run.rc ~ ')' }} diff --git a/ansible/roles/healthcheck/templates/checks/cpu-temp.sh.j2 b/ansible/roles/healthcheck/templates/checks/cpu-temp.sh.j2 new file mode 100644 index 0000000..cf57509 --- /dev/null +++ b/ansible/roles/healthcheck/templates/checks/cpu-temp.sh.j2 @@ -0,0 +1,28 @@ + # Hottest core across every thermal zone and hwmon sensor lm-sensors knows + # about. Reading the hottest rather than an average is deliberate: one core + # throttling is a real problem that an average hides. + local threshold={{ healthcheck_cpu_temp_threshold }} + local hottest=0 label="" + + while read -r t; do + [ -z "$t" ] && continue + t=${t%.*} + if [ "$t" -gt "$hottest" ]; then hottest=$t; fi + done < <(sensors -u 2>/dev/null | awk '/_input:/ && /temp/ {print $2}') + + # Fall back to the kernel thermal zones if lm-sensors reports nothing. + if [ "$hottest" -eq 0 ]; then + for z in /sys/class/thermal/thermal_zone*/temp; do + [ -r "$z" ] || continue + local milli; milli=$(cat "$z" 2>/dev/null) || continue + local c=$((milli / 1000)) + if [ "$c" -gt "$hottest" ]; then hottest=$c; label=$(cat "${z%/temp}/type" 2>/dev/null); fi + done + fi + + if [ "$hottest" -eq 0 ]; then + MESSAGE="no temperature sensors readable" + return 1 + fi + MESSAGE="${hottest}C${label:+ (${label})}" + [ "$hottest" -lt "$threshold" ] diff --git a/ansible/roles/healthcheck/templates/checks/disk-usage.sh.j2 b/ansible/roles/healthcheck/templates/checks/disk-usage.sh.j2 new file mode 100644 index 0000000..b0e76ce --- /dev/null +++ b/ansible/roles/healthcheck/templates/checks/disk-usage.sh.j2 @@ -0,0 +1,19 @@ + # Every real filesystem must be under the threshold. tmpfs, devtmpfs, + # squashfs and overlay are excluded: they are either RAM, read-only, or + # container layers, and none of them fills up in a way an operator can act on. + local threshold={{ healthcheck_disk_threshold }} + local worst=0 worst_mount="" over="" + + while read -r pct mount; do + pct=${pct%\%} + [ -z "$pct" ] && continue + if [ "$pct" -gt "$worst" ]; then worst=$pct; worst_mount=$mount; fi + if [ "$pct" -ge "$threshold" ]; then over="${over}${over:+, }${mount} ${pct}%"; fi + done < <(df -P -x tmpfs -x devtmpfs -x squashfs -x overlay --output=pcent,target 2>/dev/null | tail -n +2) + + if [ -n "$over" ]; then + MESSAGE="over ${threshold}%: ${over}" + return 1 + fi + MESSAGE="max ${worst}% on ${worst_mount:-/}" + return 0 diff --git a/ansible/roles/healthcheck/templates/checks/liveness.sh.j2 b/ansible/roles/healthcheck/templates/checks/liveness.sh.j2 new file mode 100644 index 0000000..141e478 --- /dev/null +++ b/ansible/roles/healthcheck/templates/checks/liveness.sh.j2 @@ -0,0 +1,6 @@ + # Liveness has no test to run: the fact that this script executed at all is + # the signal. What proves the host is alive is the PUSH arriving at Gatus, + # and what detects the host being dead is the heartbeat window expiring with + # no push. So this always succeeds - the reporting is the check. + MESSAGE="up since $(uptime -p 2>/dev/null || echo unknown)" + return 0 diff --git a/ansible/roles/healthcheck/templates/checks/ups-status.sh.j2 b/ansible/roles/healthcheck/templates/checks/ups-status.sh.j2 new file mode 100644 index 0000000..508c860 --- /dev/null +++ b/ansible/roles/healthcheck/templates/checks/ups-status.sh.j2 @@ -0,0 +1,18 @@ + # OL means on line power. Anything else - OB (on battery), LB (low battery), + # or no answer at all - is a failure worth waking up for, because the + # hypervisor has a finite number of minutes left. + local ups="{{ healthcheck_ups_name }}" + local status charge runtime load + + status=$(upsc "${ups}@localhost" ups.status 2>/dev/null) + if [ -z "$status" ]; then + MESSAGE="cannot reach UPS ${ups} via upsd" + return 1 + fi + + charge=$(upsc "${ups}@localhost" battery.charge 2>/dev/null) + runtime=$(upsc "${ups}@localhost" battery.runtime 2>/dev/null) + load=$(upsc "${ups}@localhost" ups.load 2>/dev/null) + + MESSAGE="status=${status} charge=${charge}% runtime=${runtime}s load=${load}%" + [[ "$status" == *"OL"* ]] diff --git a/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 b/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 new file mode 100644 index 0000000..b6c7dae --- /dev/null +++ b/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 @@ -0,0 +1,52 @@ + # Five conditions, all of which have to hold. Ported from the check that + # infra/nodito/32_zfs_pool_setup_playbook.yml deployed, which was correct - + # only its reporting was tied to Uptime Kuma. + local pool="{{ healthcheck_zfs_pool }}" + local json issues="" + + json=$(zpool status -j "$pool" 2>&1) || { MESSAGE="zpool status failed: ${json}"; return 1; } + + # 1. pool state + local state + state=$(echo "$json" | jq -r --arg p "$pool" '.pools[$p].state') + [ "$state" = "ONLINE" ] || issues="${issues}${issues:+; }pool ${state}" + + # 2. every vdev and device ONLINE + local bad + bad=$(echo "$json" | jq -r --arg p "$pool" ' + .pools[$p].vdevs[] | .. | objects + | select(.state? and .state != "ONLINE") + | "\(.name // "unknown"):\(.state)"' 2>/dev/null | paste -sd, -) + [ -z "$bad" ] || issues="${issues}${issues:+; }devices ${bad}" + + # 3. resilver in progress + local fn st + fn=$(echo "$json" | jq -r --arg p "$pool" '.pools[$p].scan_stats.function // "NONE"') + st=$(echo "$json" | jq -r --arg p "$pool" '.pools[$p].scan_stats.state // "NONE"') + if [ "$fn" = "RESILVER" ] && [ "$st" = "SCANNING" ]; then + issues="${issues}${issues:+; }resilvering" + fi + + # 4. read/write/checksum errors. ZFS reports these as strings. + local errs + errs=$(echo "$json" | jq -r --arg p "$pool" ' + .pools[$p].vdevs[] | .. | objects + | select(.name? and ((.read_errors // "0" | tonumber) > 0 + or (.write_errors // "0" | tonumber) > 0 + or (.checksum_errors // "0" | tonumber) > 0)) + | "\(.name) r=\(.read_errors) w=\(.write_errors) c=\(.checksum_errors)"' 2>/dev/null | paste -sd, -) + [ -z "$errs" ] || issues="${issues}${issues:+; }errors ${errs}" + + # 5. errors from the last scrub + local scan_err + scan_err=$(echo "$json" | jq -r --arg p "$pool" '.pools[$p].scan_stats.errors // "0"') + if [ -n "$scan_err" ] && [ "$scan_err" != "0" ] && [ "$scan_err" != "null" ]; then + issues="${issues}${issues:+; }scan errors ${scan_err}" + fi + + if [ -n "$issues" ]; then MESSAGE="$issues"; return 1; fi + + local scrub + scrub=$(echo "$json" | jq -r --arg p "$pool" '.pools[$p].scan_stats.start_time // "never"') + MESSAGE="${pool} ONLINE, last scrub ${scrub}" + return 0 diff --git a/ansible/roles/healthcheck/templates/healthcheck.service.j2 b/ansible/roles/healthcheck/templates/healthcheck.service.j2 new file mode 100644 index 0000000..574e37c --- /dev/null +++ b/ansible/roles/healthcheck/templates/healthcheck.service.j2 @@ -0,0 +1,16 @@ +[Unit] +Description={{ healthcheck_description }} +After=network-online.target +Wants=network-online.target + +[Service] +Type=oneshot +User=root +ExecStart={{ healthcheck_script_dir }}/{{ healthcheck_name }}-healthcheck.sh +Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/healthcheck/templates/healthcheck.sh.j2 b/ansible/roles/healthcheck/templates/healthcheck.sh.j2 new file mode 100644 index 0000000..92c9d27 --- /dev/null +++ b/ansible/roles/healthcheck/templates/healthcheck.sh.j2 @@ -0,0 +1,68 @@ +#!/bin/bash +# {{ healthcheck_name }} — {{ healthcheck_description }} +# Managed by Ansible (roles/healthcheck). Do not edit on the host. +# +# The exit code is the real answer; systemd stores it: +# systemctl is-failed {{ healthcheck_name }}-healthcheck.service +# The push below is an optional extra, and having no URL is normal. + +set -uo pipefail + +LOG_FILE="{{ healthcheck_log_dir }}/{{ healthcheck_name }}.log" +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" + +log() { echo "$(date '+%Y-%m-%d %H:%M:%S') - $*" >> "$LOG_FILE"; } + +# Report to Gatus as an external endpoint. Note this is a POST with a bearer +# token, not a GET with a query string - it is not the shape Uptime Kuma used. +report() { + local success="$1" message="$2" + + # No push URL is normal, not an error: the exit code below is a complete + # answer for anything reading unit state. + [ -n "$PUSH_URL" ] || return 0 + + local encoded + encoded=$(printf '%s' "$message" | sed 's/%/%25/g; s/ /%20/g; s/&/%26/g; s/+/%2B/g; s/#/%23/g') + + local code + code=$(curl -s -o /dev/null -w '%{http_code}' -X POST \ + --max-time 15 --retry 2 --retry-delay 3 \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${success}&error=${encoded}" 2>/dev/null) + + if [ "$code" = "200" ]; then + log "reported success=${success}" + else + log "ERROR: report failed (HTTP ${code})" + return 1 + fi +} + +# ── the check ──────────────────────────────────────────────────────────────── +# +# The blank line before the closing brace below is load-bearing. Jinja strips an +# included template's trailing newline, and trim_blocks (on by default in +# Ansible) then eats the newline after {% raw %}{% endif %}{% endraw %} - so without it the brace lands +# on the same line as the check body's last statement, producing `return 0}` and +# a script that dies with "syntax error: unexpected end of file". +check() { +{% if healthcheck_check %} +{% include 'checks/' ~ healthcheck_check ~ '.sh.j2' %} +{% else %} + {{ healthcheck_command }} +{% endif %} + +} + +MESSAGE="" +if check; then + log "OK${MESSAGE:+ - $MESSAGE}" + report "true" "${MESSAGE:-ok}" + exit 0 +else + log "FAILED${MESSAGE:+ - $MESSAGE}" + report "false" "${MESSAGE:-check failed}" + exit 1 +fi diff --git a/ansible/roles/healthcheck/templates/healthcheck.timer.j2 b/ansible/roles/healthcheck/templates/healthcheck.timer.j2 new file mode 100644 index 0000000..0271735 --- /dev/null +++ b/ansible/roles/healthcheck/templates/healthcheck.timer.j2 @@ -0,0 +1,18 @@ +[Unit] +Description=Run {{ healthcheck_description }} +Requires={{ healthcheck_name }}-healthcheck.service + +[Timer] +OnBootSec={{ healthcheck_boot_delay }} +{% if healthcheck_on_calendar %} +OnCalendar={{ healthcheck_on_calendar }} +{% else %} +OnUnitActiveSec={{ healthcheck_interval }} +{% endif %} +# Run a missed occurrence on the next boot rather than silently skipping it. +Persistent=true +# Spread the pushes out so twelve hosts do not all report in the same second. +RandomizedDelaySec={{ healthcheck_randomized_delay | default('30') }} + +[Install] +WantedBy=timers.target From c2de6dbbd939409fd2395ebe4997420761d6bff3 Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 09:22:12 +0200 Subject: [PATCH 60/67] backups: monitor both the dump and the pull, per source Twelve endpoints in two groups, because they answer different questions and fail for different reasons: backup-dump_ pushed by the SOURCE right after its dump runs backup-store_ pushed by the BOX at 05:30, per source backup-store_pull-job pushed by the BOX, about the box itself The store alone could catch almost everything, because the artefact filename carries the source's dump timestamp - a source whose timer died still pulls "ok" forever, but the timestamp gives it away. What the source side adds is LATENCY and DIAGNOSIS: the store only learns at the next 04:00 pull, and it cannot tell you whether the dump broke or the pull did. pull-job is separate from the per-source checks because it is a LEADING indicator where those are lagging ones. A disabled pull timer, a failed pull job, or a filling disk are all visible immediately, while the per-source checks only fire once an artefact is >26h stale - about a day later. Disable the timer at 10:00 and every source stays green until tomorrow; pull-job goes red this morning and names the cause instead of showing six stale sources. Frequencies: dumps 02:00-02:30 staggered, pull 04:00, verification 05:30, all daily and Persistent. 26h staleness decides red; a 30h Gatus heartbeat catches the verification itself having stopped, so a dead check-backups.timer cannot hide a stale backup. arbret has no dump endpoint: prd-arbret is in [arbret], which `managed` deliberately excludes, so nothing of ours runs there. Store-checked only. check-backups.sh was manual-only; it now runs on a timer and reports per source rather than only printing. The human-readable report is unchanged. A SERIOUS bug introduced and fixed in this change, recorded because the shape is easy to repeat: the reporting hook was added to backup.sh as a second `trap ... EXIT`. Bash REPLACES the EXIT handler rather than adding to it, so that silently deleted the trap which restarts the stopped service - the one the script's own comment calls "the point", and the bug the role was written to eliminate. Every backup then stopped its service and left it stopped. It took forgejo, lnbits, headscale and memos down for several minutes each, and nothing caught it: the dumps exit 0, the artefacts are correct, the deploy reports failed=0, and liveness only proves the HOST is up. There is now ONE EXIT handler doing both jobs, armed BEFORE the stop so a failure during the stop still restarts. Verified by rendering both variants, asserting exactly one EXIT trap in each, and simulating a mid-way failure to confirm the restart fires. Verified: all 12 endpoints UP; six sources pulled cleanly (arbret 31M, headscale 198K, memos 7.8M, vaultwarden 2.1M, lnbits 30M, forgejo 2.6G), store 16% full, "RESULT: all checks passed". Known gaps, deliberately not closed here: * `yell` warnings - disk 75-90%, an artefact under half the previous size, retention not pruning - never reach Gatus, because a push is binary. * Nothing verifies a backed-up service came back UP. That is the gap that let the trap bug run unnoticed, and it is what the next change addresses. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/playbooks/backups.yml | 67 +++++++++++++++++++ ansible/roles/backup_source/defaults/main.yml | 13 ++++ ansible/roles/backup_source/tasks/main.yml | 19 ++++-- .../backup_source/templates/backup.service.j2 | 2 + .../backup_source/templates/backup.sh.j2 | 56 +++++++++++++++- ansible/roles/backup_store/defaults/main.yml | 15 +++++ ansible/roles/backup_store/tasks/main.yml | 29 +++++++- .../templates/check-backups.service.j2 | 16 +++++ .../templates/check-backups.sh.j2 | 53 ++++++++++++++- .../templates/check-backups.timer.j2 | 11 +++ .../services/forgejo/setup_backup_forgejo.yml | 4 ++ .../headscale/setup_backup_headscale.yml | 4 ++ .../services/lnbits/setup_backup_lnbits.yml | 4 ++ ansible/services/memos/setup_backup_memos.yml | 4 ++ .../vaultwarden/setup_backup_vaultwarden.yml | 4 ++ 15 files changed, 289 insertions(+), 12 deletions(-) create mode 100644 ansible/roles/backup_store/templates/check-backups.service.j2 create mode 100644 ansible/roles/backup_store/templates/check-backups.timer.j2 diff --git a/ansible/playbooks/backups.yml b/ansible/playbooks/backups.yml index 0686ab6..61c005d 100644 --- a/ansible/playbooks/backups.yml +++ b/ansible/playbooks/backups.yml @@ -7,6 +7,10 @@ ansible.builtin.include_role: name: backup_store vars: + # check-backups.sh reports one result per source plus one for the store + # itself, so it needs the collection URL and appends each key. + backup_store_check_push_base: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints" + backup_store_check_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" backup_store_sources: - name: arbret source: "arbret@prd-arbret:/opt/arbret/backups/" @@ -26,3 +30,66 @@ - name: forgejo source: "backup-pull@prd-vipy:/opt/backups/forgejo/" retention_days: 14 + +# ───────────────────────────────────────────────────────────────────────────── +# Register the backup checks with Gatus. +# +# Two groups on purpose, because they answer different questions and fail for +# different reasons: +# +# backup-dump did the SOURCE produce an artefact? Pushed by each dump right +# after it runs, so a broken dump is visible within minutes. +# backup-store did it ARRIVE, is it fresh, non-zero, plausibly sized, and is +# retention pruning? Pushed by check-backups.sh at 05:30. +# +# The store alone could catch almost everything, because the artefact filename +# carries the source's dump timestamp - a source whose timer died still pulls +# "ok" forever, but the timestamp gives it away. What the source side adds is +# LATENCY and DIAGNOSIS: the store only learns at the next 04:00 pull, and it +# cannot tell you whether the dump broke or the pull did. +# +# arbret has no dump endpoint: prd-arbret lives in [arbret], which `managed` +# deliberately excludes, so nothing of ours runs there. It is store-checked only. +# ───────────────────────────────────────────────────────────────────────────── +- name: Register the backup checks with Gatus + hosts: observability + become: yes + vars: + # Sources we deploy the dump for, and the host each one runs on. + dump_sources: + - {name: headscale, host: spacey} + - {name: memos, host: memos_box_local} + - {name: vaultwarden, host: vipy} + - {name: lnbits, host: vipy} + - {name: forgejo, host: vipy} + store_sources: [arbret, headscale, memos, vaultwarden, lnbits, forgejo] + + tasks: + - name: Build the dump endpoint list + ansible.builtin.set_fact: + dump_endpoints: "{{ dump_endpoints | default([]) + [{ + 'name': item.name, + 'group': 'backup-dump', + 'token': gatus_push_tokens[item.host], + 'heartbeat': '30h'}] }}" + loop: "{{ dump_sources }}" + + - name: Build the store endpoint list + ansible.builtin.set_fact: + store_endpoints: "{{ store_endpoints | default([]) + [{ + 'name': item, + 'group': 'backup-store', + 'token': gatus_push_tokens['small_backups_local'], + 'heartbeat': '30h'}] }}" + loop: "{{ store_sources }}" + + - name: Register the backup endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: backups + gatus_endpoint_external: "{{ dump_endpoints + store_endpoints + [{ + 'name': 'pull job', + 'group': 'backup-store', + 'token': gatus_push_tokens['small_backups_local'], + 'heartbeat': '30h'}] }}" diff --git a/ansible/roles/backup_source/defaults/main.yml b/ansible/roles/backup_source/defaults/main.yml index 7de6006..b195190 100644 --- a/ansible/roles/backup_source/defaults/main.yml +++ b/ansible/roles/backup_source/defaults/main.yml @@ -29,3 +29,16 @@ backup_source_retention_days: 7 # Schedule. The box pulls at 04:00, so dumps must land before that. backup_source_on_calendar: "*-*-* 02:00:00" + +# ── Reporting ──────────────────────────────────────────────────────────────── +# Where to report that this dump ran and produced a plausible artefact. +# Gatus external endpoint: +# POST {url}?success=true|false&error=... +# Authorization: Bearer {token} +# Empty is valid and is not an error: the unit's exit code is still the answer, +# and the STORE will independently notice a stale dump within ~26h because the +# artefact filename carries this dump's timestamp. Reporting here only buys +# earlier detection and tells you it was the DUMP that broke rather than the +# pull. +backup_source_push_url: "" +backup_source_push_token: "" diff --git a/ansible/roles/backup_source/tasks/main.yml b/ansible/roles/backup_source/tasks/main.yml index ec2d6cf..a964c5d 100644 --- a/ansible/roles/backup_source/tasks/main.yml +++ b/ansible/roles/backup_source/tasks/main.yml @@ -30,7 +30,12 @@ - name: Ensure age is installed ansible.builtin.apt: - name: age + name: + - age + # curl is needed only when backup_source_push_url is set, but installing it + # unconditionally keeps the task idempotent and it is present on every + # Debian host here anyway. + - curl state: present # The pull account: unprivileged, no sudo, exists only so small-backups-box can @@ -83,14 +88,18 @@ mode: '0750' validate: "bash -n %s" +# The .service carries the push token in an Environment= line, so it is 0600. +# The .timer holds nothing secret and stays world-readable. - name: "Install the {{ backup_source_name }}-backup systemd units" ansible.builtin.template: - src: "backup.{{ item }}.j2" - dest: "/etc/systemd/system/{{ backup_source_name }}-backup.{{ item }}" + src: "backup.{{ item.unit }}.j2" + dest: "/etc/systemd/system/{{ backup_source_name }}-backup.{{ item.unit }}" owner: root group: root - mode: '0644' - loop: [service, timer] + mode: "{{ item.mode }}" + loop: + - {unit: service, mode: "0600"} + - {unit: timer, mode: "0644"} notify: Reload systemd for backup units - name: "Enable the {{ backup_source_name }}-backup timer" diff --git a/ansible/roles/backup_source/templates/backup.service.j2 b/ansible/roles/backup_source/templates/backup.service.j2 index 91ad063..7a1df21 100644 --- a/ansible/roles/backup_source/templates/backup.service.j2 +++ b/ansible/roles/backup_source/templates/backup.service.j2 @@ -9,6 +9,8 @@ After={{ backup_source_stop_service if '.' in backup_source_stop_service else ba [Service] Type=oneshot ExecStart=/usr/local/bin/{{ backup_source_name }}-backup.sh +Environment=BACKUP_PUSH_URL={{ backup_source_push_url }} +Environment=BACKUP_PUSH_TOKEN={{ backup_source_push_token }} StandardOutput=journal StandardError=journal SyslogIdentifier={{ backup_source_name }}-backup diff --git a/ansible/roles/backup_source/templates/backup.sh.j2 b/ansible/roles/backup_source/templates/backup.sh.j2 index abdadfe..018c313 100644 --- a/ansible/roles/backup_source/templates/backup.sh.j2 +++ b/ansible/roles/backup_source/templates/backup.sh.j2 @@ -41,13 +41,65 @@ chmod 700 "$BACKUP_DIR" # here or they accumulate forever. rm -f "${BACKUP_DIR}/${NAME}_"*.partial +# --- Reporting ------------------------------------------------------------- +# A dump that exits non-zero, or that produces a zero-byte artefact, is a failed +# backup even though the script "finished". Both are reported as failures. +PUSH_URL="${BACKUP_PUSH_URL:-}" +PUSH_TOKEN="${BACKUP_PUSH_TOKEN:-}" + +report() { + local success="$1" message="$2" + [ -n "$PUSH_URL" ] || return 0 + local encoded + encoded=$(printf '%s' "$message" | sed 's/%/%25/g; s/ /%20/g; s/&/%26/g; s/+/%2B/g; s/#/%23/g') + curl -s -o /dev/null --max-time 15 --retry 2 --retry-delay 3 -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${success}&error=${encoded}" 2>/dev/null || true +} + +# Reports on ANY exit path, so a dump that dies halfway still reports rather +# than going quiet. The size of the FINISHED artefact decides success, not +# merely reaching the end of the script. +# +# This is called FROM the single EXIT trap below - it must never register an +# EXIT trap of its own. `trap ... EXIT` REPLACES the existing handler rather +# than adding to it, so a second trap here silently discards the one that +# restarts the service, and a backup run leaves the service stopped. That is +# precisely the failure the restart trap exists to prevent. +report_outcome() { + local rc="$1" + if [ "$rc" -ne 0 ]; then + report "false" "${NAME} dump exited ${rc}" + elif [ ! -s "$ARTIFACT" ]; then + report "false" "${NAME} produced no artefact at ${ARTIFACT}" + else + report "true" "${NAME} $(du -h "$ARTIFACT" | cut -f1)" + fi +} + +# --- One EXIT handler, doing both jobs ------------------------------------- +# bash keeps exactly ONE EXIT trap: `trap ... EXIT` REPLACES the previous +# handler rather than adding to it. Registering a second one here would +# silently discard the service restart and leave the service stopped after +# every backup - which is the exact bug the restart exists to prevent, and it +# is invisible until someone notices the service is down. +on_exit() { + local rc=$? {% if backup_source_stop_service or backup_source_stop_command %} -# --- Stop the service, and guarantee it comes back --- + log "Restarting ${SERVICE}..." + eval "$START_CMD" || true +{% endif %} + report_outcome "$rc" +} +trap on_exit EXIT + +{% if backup_source_stop_service or backup_source_stop_command %} +# --- Stop the service; the trap above guarantees it comes back ------------- # The trap is the point: without it a failed dump leaves the service down until # the next timer fires. Every hand-written script this replaced had that bug. +# It is armed BEFORE the stop, so even a failure during the stop restarts. log "Stopping ${SERVICE}..." eval "$STOP_CMD" -trap 'log "Restarting ${SERVICE}..."; eval "$START_CMD" || true' EXIT {% endif %} # --- Dump straight into age; plaintext never touches the disk --- diff --git a/ansible/roles/backup_store/defaults/main.yml b/ansible/roles/backup_store/defaults/main.yml index c0c128b..3a0754c 100644 --- a/ansible/roles/backup_store/defaults/main.yml +++ b/ansible/roles/backup_store/defaults/main.yml @@ -9,3 +9,18 @@ backup_store_on_calendar: "*-*-* 04:00:00" # source: "backup-pull@headscale.contrapeso.xyz:/opt/backups/headscale/" # retention_days: 90 backup_store_sources: [] + +# ── Reporting ──────────────────────────────────────────────────────────────── +# check-backups.sh reports one result PER SOURCE plus one for the store itself, +# so the base URL is the endpoints collection and the script appends each key. +# Empty is valid: the script still prints its report and exits 0/1. +backup_store_check_push_base: "" +backup_store_check_push_token: "" + +# Runs after the 04:00 pull. Late enough that a slow pull has finished, early +# enough that a failure is visible before the working day. +backup_store_check_on_calendar: "*-*-* 05:30:00" + +# An artefact older than this is stale. Sources dump daily at 02:00-02:30 and the +# pull is at 04:00, so 26h tolerates exactly one missed night before alarming. +backup_store_check_max_age_hours: 26 diff --git a/ansible/roles/backup_store/tasks/main.yml b/ansible/roles/backup_store/tasks/main.yml index 2596dbc..55500c6 100644 --- a/ansible/roles/backup_store/tasks/main.yml +++ b/ansible/roles/backup_store/tasks/main.yml @@ -33,9 +33,9 @@ validate: "bash -n %s" become: yes -# A human-run assertion that last night actually worked. Generated from the same -# source list as the puller, so it can never drift out of sync with what is -# supposed to be arriving. +# An assertion that last night actually worked. Generated from the same source +# list as the puller, so it can never drift out of sync with what is supposed to +# be arriving. Runs on a timer AND is useful by hand. - name: Install the backup check script ansible.builtin.template: src: check-backups.sh.j2 @@ -46,6 +46,29 @@ validate: "bash -n %s" become: yes +# The .service carries the push token, so it is 0600; the .timer is not secret. +- name: Install the check-backups systemd units + ansible.builtin.template: + src: "check-backups.{{ item.unit }}.j2" + dest: "/etc/systemd/system/check-backups.{{ item.unit }}" + owner: root + group: root + mode: "{{ item.mode }}" + loop: + - {unit: service, mode: "0600"} + - {unit: timer, mode: "0644"} + become: yes + +# restarted, not started: `started` is a no-op on an already-active timer, so a +# changed schedule would never be picked up. +- name: Enable the check-backups timer + ansible.builtin.systemd: + name: check-backups.timer + enabled: yes + state: restarted + daemon_reload: yes + become: yes + - name: Install the pull-backups systemd units ansible.builtin.template: src: "pull-backups.{{ item }}.j2" diff --git a/ansible/roles/backup_store/templates/check-backups.service.j2 b/ansible/roles/backup_store/templates/check-backups.service.j2 new file mode 100644 index 0000000..3e50b6a --- /dev/null +++ b/ansible/roles/backup_store/templates/check-backups.service.j2 @@ -0,0 +1,16 @@ +[Unit] +Description=Verify the nightly backup pull actually worked +After=network-online.target +Wants=network-online.target + +[Service] +Type=oneshot +User={{ ansible_user_id }} +ExecStart=/usr/local/bin/check-backups.sh {{ backup_store_check_max_age_hours }} +Environment=BACKUP_CHECK_PUSH_BASE={{ backup_store_check_push_base }} +Environment=BACKUP_CHECK_PUSH_TOKEN={{ backup_store_check_push_token }} +StandardOutput=journal +StandardError=journal + +[Install] +WantedBy=multi-user.target diff --git a/ansible/roles/backup_store/templates/check-backups.sh.j2 b/ansible/roles/backup_store/templates/check-backups.sh.j2 index f7a8830..37c9916 100644 --- a/ansible/roles/backup_store/templates/check-backups.sh.j2 +++ b/ansible/roles/backup_store/templates/check-backups.sh.j2 @@ -22,9 +22,34 @@ fails=0; warns=0 if [ -t 1 ]; then R=$'\033[31m'; Y=$'\033[33m'; G=$'\033[32m'; N=$'\033[0m' else R=''; Y=''; G=''; N=''; fi -red() { printf ' %sFAIL%s %s\n' "$R" "$N" "$*"; fails=$((fails+1)); } +# Per-source verdicts, so each source can be reported independently. A single +# aggregate red light tells you backups are broken; it does not tell you which +# one, which is the thing you need at 3am. +declare -A SRC_FAIL SRC_MSG +CURRENT="" + +red() { printf ' %sFAIL%s %s\n' "$R" "$N" "$*"; fails=$((fails+1)); + [ -n "$CURRENT" ] && { SRC_FAIL[$CURRENT]=1; SRC_MSG[$CURRENT]="${SRC_MSG[$CURRENT]:-}${SRC_MSG[$CURRENT]:+; }$*"; }; } yell() { printf ' %sWARN%s %s\n' "$Y" "$N" "$*"; warns=$((warns+1)); } -ok() { printf ' %sok%s %s\n' "$G" "$N" "$*"; } +ok() { printf ' %sok%s %s\n' "$G" "$N" "$*"; + [ -n "$CURRENT" ] && SRC_MSG[$CURRENT]="${SRC_MSG[$CURRENT]:-}${SRC_MSG[$CURRENT]:+; }$*"; } + +# --- Reporting ------------------------------------------------------------- +# Each source gets its own Gatus external endpoint, plus one for the store +# itself (the pull unit, the timer, and disk capacity). PUSH_BASE empty means +# report nowhere, which is valid: the exit code is still the whole answer. +PUSH_BASE="${BACKUP_CHECK_PUSH_BASE:-}" +PUSH_TOKEN="${BACKUP_CHECK_PUSH_TOKEN:-}" + +report() { + local key="$1" success="$2" message="$3" + [ -n "$PUSH_BASE" ] || return 0 + local encoded + encoded=$(printf '%s' "$message" | sed 's/%/%25/g; s/ /%20/g; s/&/%26/g; s/+/%2B/g; s/#/%23/g') + curl -s -o /dev/null --max-time 15 --retry 2 --retry-delay 3 -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_BASE}/${key}/external?success=${success}&error=${encoded}" 2>/dev/null || true +} hours_since() { echo $(( (NOW - $1) / 3600 )); } @@ -43,6 +68,9 @@ dump_epoch() { check_source() { local name="$1" keep="$2" dir="$STORE/$1" printf '\n%s\n' "== $name" + CURRENT="$name" + SRC_FAIL[$name]=0 + SRC_MSG[$name]="" [ -d "$dir" ] || { red "$name: no directory $dir"; return; } @@ -108,6 +136,15 @@ echo "Backup check on $(hostname) at $(date '+%Y-%m-%d %H:%M:%S %Z')" echo "Artefacts older than ${MAX_AGE_H}h are treated as stale." # --- the pull job itself --- +# Reserved key, reported as backup-store_pull-job. The store's own machinery is +# a different alarm from any one source being stale, and it is a LEADING +# indicator where the per-source checks are lagging ones: those only fire once +# an artefact is >26h stale, i.e. about a day after the fault. A disabled timer, +# a failed pull job or a filling disk are all visible here immediately, and they +# name the cause instead of showing six stale sources with no explanation. +CURRENT="__store" +SRC_FAIL[__store]=0 +SRC_MSG[__store]="" printf '\n%s\n' "== pull-backups.service" result=$(systemctl show pull-backups.service -p Result --value 2>/dev/null) status=$(systemctl show pull-backups.service -p ExecMainStatus --value 2>/dev/null) @@ -124,9 +161,16 @@ systemctl is-enabled pull-backups.timer >/dev/null 2>&1 \ # --- each source --- {% for src in backup_store_sources %} check_source "{{ src.name }}" {{ src.retention_days }} +CURRENT="" +# The store key is what Gatus computes from group+name: sanitize("backup-store") +# + "_" + sanitize("{{ src.name }}"). +report "backup-store_{{ src.name }}" \ + "$([ "${SRC_FAIL[{{ src.name }}]:-1}" -eq 0 ] && echo true || echo false)" \ + "${SRC_MSG[{{ src.name }}]:-no result}" {% endfor %} # --- capacity --- +CURRENT="__store" printf '\n%s\n' "== disk" use=$(df --output=pcent "$STORE" | tail -1 | tr -dc '0-9') avail=$(df -h --output=avail "$STORE" | tail -1 | tr -d ' ') @@ -134,6 +178,11 @@ if [ "$use" -ge 90 ]; then red "store is ${use}% full, ${avail} free" elif [ "$use" -ge 75 ]; then yell "store is ${use}% full, ${avail} free" else ok "store is ${use}% full, ${avail} free"; fi +CURRENT="" +report "backup-store_pull-job" \ + "$([ "${SRC_FAIL[__store]:-1}" -eq 0 ] && echo true || echo false)" \ + "${SRC_MSG[__store]:-no result}" + printf '\n%s\n' "-----" if [ "$fails" -gt 0 ]; then echo "RESULT: $fails failure(s), $warns warning(s)" diff --git a/ansible/roles/backup_store/templates/check-backups.timer.j2 b/ansible/roles/backup_store/templates/check-backups.timer.j2 new file mode 100644 index 0000000..ba5c5a8 --- /dev/null +++ b/ansible/roles/backup_store/templates/check-backups.timer.j2 @@ -0,0 +1,11 @@ +[Unit] +Description=Run the backup verification after the nightly pull +Requires=check-backups.service + +[Timer] +OnCalendar={{ backup_store_check_on_calendar }} +# Run a missed occurrence on the next boot rather than skipping the day. +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/ansible/services/forgejo/setup_backup_forgejo.yml b/ansible/services/forgejo/setup_backup_forgejo.yml index ee1769b..5329e2a 100644 --- a/ansible/services/forgejo/setup_backup_forgejo.yml +++ b/ansible/services/forgejo/setup_backup_forgejo.yml @@ -22,3 +22,7 @@ backup_source_stop_service: forgejo backup_source_retention_days: 2 backup_source_on_calendar: "*-*-* 02:30:00" + # Reported to Gatus as backup-dump_forgejo. The token is this HOST's token, + # shared with its other checks - see infra/400_host_monitoring.yml. + backup_source_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/backup-dump_forgejo/external" + backup_source_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" diff --git a/ansible/services/headscale/setup_backup_headscale.yml b/ansible/services/headscale/setup_backup_headscale.yml index 5ae3ad1..15ba5b7 100644 --- a/ansible/services/headscale/setup_backup_headscale.yml +++ b/ansible/services/headscale/setup_backup_headscale.yml @@ -18,3 +18,7 @@ backup_source_stop_service: headscale backup_source_retention_days: 7 backup_source_on_calendar: "*-*-* 02:00:00" + # Reported to Gatus as backup-dump_headscale. The token is this HOST's token, + # shared with its other checks - see infra/400_host_monitoring.yml. + backup_source_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/backup-dump_headscale/external" + backup_source_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" diff --git a/ansible/services/lnbits/setup_backup_lnbits.yml b/ansible/services/lnbits/setup_backup_lnbits.yml index 0c45b29..2da0ac0 100644 --- a/ansible/services/lnbits/setup_backup_lnbits.yml +++ b/ansible/services/lnbits/setup_backup_lnbits.yml @@ -22,3 +22,7 @@ backup_source_stop_service: lnbits backup_source_retention_days: 7 backup_source_on_calendar: "*-*-* 02:20:00" + # Reported to Gatus as backup-dump_lnbits. The token is this HOST's token, + # shared with its other checks - see infra/400_host_monitoring.yml. + backup_source_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/backup-dump_lnbits/external" + backup_source_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" diff --git a/ansible/services/memos/setup_backup_memos.yml b/ansible/services/memos/setup_backup_memos.yml index 36b8cf6..53ab85d 100644 --- a/ansible/services/memos/setup_backup_memos.yml +++ b/ansible/services/memos/setup_backup_memos.yml @@ -23,3 +23,7 @@ backup_source_stop_service: memos backup_source_retention_days: 7 backup_source_on_calendar: "*-*-* 02:00:00" + # Reported to Gatus as backup-dump_memos. The token is this HOST's token, + # shared with its other checks - see infra/400_host_monitoring.yml. + backup_source_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/backup-dump_memos/external" + backup_source_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" diff --git a/ansible/services/vaultwarden/setup_backup_vaultwarden.yml b/ansible/services/vaultwarden/setup_backup_vaultwarden.yml index 54fe147..d0dfdfa 100644 --- a/ansible/services/vaultwarden/setup_backup_vaultwarden.yml +++ b/ansible/services/vaultwarden/setup_backup_vaultwarden.yml @@ -22,3 +22,7 @@ backup_source_start_command: "docker compose -f /opt/vaultwarden/docker-compose.yml start" backup_source_retention_days: 7 backup_source_on_calendar: "*-*-* 02:10:00" + # Reported to Gatus as backup-dump_vaultwarden. The token is this HOST's token, + # shared with its other checks - see infra/400_host_monitoring.yml. + backup_source_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/backup-dump_vaultwarden/external" + backup_source_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" From efa9eb55ca42adc2ad00cf5c2ec09c2d2beec066 Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 09:33:57 +0200 Subject: [PATCH 61/67] monitoring: systemd services, domain expiry, DNS correctness, public endpoints MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four more check types, 42 endpoints, taking the estate from 39 to 81. ── systemd services (infra/401) ──────────────────────────────────────────── Every unit we deploy, checked every 5 minutes with a 16-minute heartbeat. This closes the gap that let a real bug run unnoticed earlier today: a backup script left forgejo, lnbits, headscale and memos stopped, and NOTHING caught it. The dumps exited 0, the artefacts were correct, the deploy said failed=0, and liveness only proves the HOST is up - not that anything on it serves. One endpoint PER UNIT but only ONE timer per host: the check iterates that host's units and pushes a result for each, the way check-backups.sh reports per source. Four units on vipy would otherwise mean four scripts, services and timers. A single host-level red light would also say "something on vipy is down" without saying which, which is not the question anyone has. Keys are host-qualified because unit names collide - caddy runs on four machines. Which units a host runs lives in host_vars//monitored_services, because "what runs here" is a property of the machine, the same reasoning as the cross-host ports. nut-driver-enumerator is deliberately excluded: it is a oneshot generator that is `enabled` but always `inactive`, so it would report down forever. Checked live before excluding it. ── domain, DNS and public endpoints (infra/402) ──────────────────────────── The first checks that PULL rather than push, and that is the right way round: all three are about how the outside world sees us, so they must be measured from outside. Nothing is installed anywhere - no script, no timer, no token. They also have no heartbeat, because a heartbeat answers "did the reporter report"; when Gatus does the checking itself, failure is immediate. domain 1 endpoint, 24h, [DOMAIN_EXPIRATION] > 336h (two weeks) dns 11 endpoints, 24h, [DNS_RCODE] == NOERROR and [BODY] == the IP public 14 endpoints, 5m, 11 HTTPS + 3 TCP Expected A records are derived from inventory (hostvars[host].ansible_host), not written down again. The estate's recurring bug is an address recorded in a second place and left behind when the machine moved; asserting against inventory means a renumbered box is one edit, not two. Expected HTTP status was checked live per site rather than assumed. Two return 401 - the Gatus dashboard and the DATUM dashboard, both behind basic auth - and that is what is asserted: expecting 200 there would go green precisely when the auth broke. headscale asserts /health rather than /, which is a 404 by design. Every HTTPS check also carries [CERTIFICATE_EXPIRATION] > 168h, which is free on an endpoint already being polled and catches a renewal that silently stops. Two things learned the hard way, both now in comments: * A domain-expiry endpoint needs a URL SCHEME. Gatus derives the endpoint type from the prefix (endpoint.Type()), so a bare "contrapeso.xyz" is UNKNOWN and the whole config is rejected. It is https:// plus a DOMAIN_EXPIRATION condition and no status assertion, so the registrar's parking page at the apex is irrelevant. Upstream also enforces a 5m minimum interval for that placeholder, because it uses a free whois service. * That rejection proved skip-invalid-config-update was worth adding. Gatus logged "the configuration file was updated, but it is not valid, the old configuration will continue being used" and kept running. Without it the reload path calls panic() and one malformed contributed file takes the monitor down. Verified: 15/15 service endpoints UP; 26/27 public-facing UP. The single DOWN is public_memos, correctly - memos-box is powered off, and the condition result reads [STATUS] (502) == 200. memos-box and arbret-staging-box were shut down deliberately from the Proxmox UI to test the liveness endpoints; their endpoints stay registered and will go green when the VMs return. Co-Authored-By: Claude Opus 5 (1M context) --- .../host_vars/forgejo_runner_local/main.yml | 12 ++ ansible/host_vars/fulcrum_box_local/main.yml | 11 ++ ansible/host_vars/knots_box_local/main.yml | 12 ++ ansible/host_vars/memos_box_local/main.yml | 12 ++ ansible/host_vars/monitoring/main.yml | 12 ++ ansible/host_vars/nodito/main.yml | 12 ++ ansible/host_vars/spacey/main.yml | 13 ++ ansible/host_vars/vipy/main.yml | 15 +++ ansible/host_vars/watchtower/main.yml | 13 ++ ansible/infra/401_service_monitoring.yml | 83 ++++++++++++ ansible/infra/402_public_monitoring.yml | 125 ++++++++++++++++++ .../roles/gatus_endpoint/defaults/main.yml | 5 + .../templates/endpoints.yaml.j2 | 12 ++ ansible/roles/healthcheck/defaults/main.yml | 11 ++ .../templates/checks/systemd-units.sh.j2 | 33 +++++ .../templates/healthcheck.service.j2 | 1 + .../healthcheck/templates/healthcheck.sh.j2 | 14 ++ 17 files changed, 396 insertions(+) create mode 100644 ansible/host_vars/forgejo_runner_local/main.yml create mode 100644 ansible/host_vars/memos_box_local/main.yml create mode 100644 ansible/host_vars/monitoring/main.yml create mode 100644 ansible/host_vars/spacey/main.yml create mode 100644 ansible/host_vars/vipy/main.yml create mode 100644 ansible/host_vars/watchtower/main.yml create mode 100644 ansible/infra/401_service_monitoring.yml create mode 100644 ansible/infra/402_public_monitoring.yml create mode 100644 ansible/roles/healthcheck/templates/checks/systemd-units.sh.j2 diff --git a/ansible/host_vars/forgejo_runner_local/main.yml b/ansible/host_vars/forgejo_runner_local/main.yml new file mode 100644 index 0000000..6e852ce --- /dev/null +++ b/ansible/host_vars/forgejo_runner_local/main.yml @@ -0,0 +1,12 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - forgejo-runner diff --git a/ansible/host_vars/fulcrum_box_local/main.yml b/ansible/host_vars/fulcrum_box_local/main.yml index 2cf3c65..aa189b0 100644 --- a/ansible/host_vars/fulcrum_box_local/main.yml +++ b/ansible/host_vars/fulcrum_box_local/main.yml @@ -4,3 +4,14 @@ # which publishes the port. See host_vars/knots_box_local/main.yml for why this # lives in host_vars rather than in the role's defaults. fulcrum_ssl_port: 50002 + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - fulcrum diff --git a/ansible/host_vars/knots_box_local/main.yml b/ansible/host_vars/knots_box_local/main.yml index fb15905..3866541 100644 --- a/ansible/host_vars/knots_box_local/main.yml +++ b/ansible/host_vars/knots_box_local/main.yml @@ -12,3 +12,15 @@ bitcoin_p2p_port: 8333 datum_gateway_api_port: 7152 datum_gateway_stratum_port: 23334 + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - bitcoind + - datum-gateway diff --git a/ansible/host_vars/memos_box_local/main.yml b/ansible/host_vars/memos_box_local/main.yml new file mode 100644 index 0000000..f1fc19b --- /dev/null +++ b/ansible/host_vars/memos_box_local/main.yml @@ -0,0 +1,12 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - memos diff --git a/ansible/host_vars/monitoring/main.yml b/ansible/host_vars/monitoring/main.yml new file mode 100644 index 0000000..2ef805d --- /dev/null +++ b/ansible/host_vars/monitoring/main.yml @@ -0,0 +1,12 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - caddy diff --git a/ansible/host_vars/nodito/main.yml b/ansible/host_vars/nodito/main.yml index a6f6a58..c8176af 100644 --- a/ansible/host_vars/nodito/main.yml +++ b/ansible/host_vars/nodito/main.yml @@ -31,3 +31,15 @@ ups_port: auto ups_user: counterweight ups_offdelay: 120 # Seconds after shutdown before UPS cuts outlet power ups_ondelay: 30 # Seconds after mains returns before UPS restores outlet power + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - nut-server + - nut-monitor diff --git a/ansible/host_vars/spacey/main.yml b/ansible/host_vars/spacey/main.yml new file mode 100644 index 0000000..a50d116 --- /dev/null +++ b/ansible/host_vars/spacey/main.yml @@ -0,0 +1,13 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - headscale + - caddy diff --git a/ansible/host_vars/vipy/main.yml b/ansible/host_vars/vipy/main.yml new file mode 100644 index 0000000..3b077df --- /dev/null +++ b/ansible/host_vars/vipy/main.yml @@ -0,0 +1,15 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - forgejo + - lnbits + - caddy + - phoenixd diff --git a/ansible/host_vars/watchtower/main.yml b/ansible/host_vars/watchtower/main.yml new file mode 100644 index 0000000..efdf7d6 --- /dev/null +++ b/ansible/host_vars/watchtower/main.yml @@ -0,0 +1,13 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - caddy + - ntfy diff --git a/ansible/infra/401_service_monitoring.yml b/ansible/infra/401_service_monitoring.yml new file mode 100644 index 0000000..f07a906 --- /dev/null +++ b/ansible/infra/401_service_monitoring.yml @@ -0,0 +1,83 @@ +--- +# Is each systemd-deployed service actually running? +# +# Every 5 minutes, with a 16-minute Gatus heartbeat - three missed runs before +# it alarms, so a reboot or a slow check does not page anyone, but a host that +# stops reporting does. +# +# This closes the gap that let a real bug run unnoticed: a backup script left +# forgejo, lnbits, headscale and memos stopped, and NOTHING caught it. The dumps +# exited 0, the artefacts were correct, the deploy said failed=0, and liveness +# only proves the HOST is up - not that anything on it is serving. +# +# One endpoint PER UNIT, not per host. A host running four services needs four +# endpoints, or a single red light says "something on vipy is down" without +# saying which - and that is the question you actually have at 3am. But only ONE +# timer per host: the check iterates that host's units and pushes a result for +# each, the same way check-backups.sh reports per source. Four units on vipy +# would otherwise mean four scripts, four services and four timers. +# +# Which units each host runs is in host_vars//main.yml as +# monitored_services, because "what runs here" is a property of the machine. +# +# Keys are host-qualified because unit names collide - caddy runs on four +# machines. Gatus computes sanitize(group)_sanitize(name), so group "services" +# and name "vipy/caddy" give services_vipy-caddy. + +# ───────────────────────────────────────────────────────────────────────────── +# Register one endpoint per unit. Runs first: Gatus reloads within 30s, and the +# host play above takes minutes, so every endpoint exists before its first push. +# ───────────────────────────────────────────────────────────────────────────── +- name: Register the service checks with Gatus + hosts: observability + become: yes + + tasks: + # Two plain steps rather than one clever expression: first collect which + # units each host declares, then flatten that into endpoints. + - name: Collect the units each host declares + ansible.builtin.set_fact: + host_units: "{{ host_units | default([]) + [{'host': item, 'units': hostvars[item].monitored_services}] }}" + loop: "{{ groups['managed'] | sort }}" + when: hostvars[item].monitored_services | default([]) | length > 0 + + - name: Build one endpoint per unit + ansible.builtin.set_fact: + service_endpoints: "{{ service_endpoints | default([]) + [{ + 'name': (item.0.host | lower | regex_replace('[/_.,# +&]', '-')) ~ '/' ~ item.1, + 'group': 'services', + 'token': gatus_push_tokens[item.0.host], + 'heartbeat': '16m'}] }}" + loop: "{{ host_units | subelements('units') }}" + + - name: Register the service endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: services + gatus_endpoint_external: "{{ service_endpoints }}" + +- name: Monitor systemd services on every host that has them + hosts: managed + become: yes + vars: + gatus_api: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints" + host_key: "{{ inventory_hostname | lower | regex_replace('[/_.,# +&]', '-') }}" + + tasks: + - name: Is every deployed service running? + ansible.builtin.include_role: + name: healthcheck + vars: + healthcheck_name: service-health + healthcheck_description: "systemd services on {{ inventory_hostname }}" + healthcheck_check: systemd-units + healthcheck_units: "{{ monitored_services }}" + healthcheck_units_key_prefix: "services_{{ host_key }}" + healthcheck_interval: "5min" + healthcheck_boot_delay: "2min" + # The per-unit results go to keys under this collection; the role's own + # single-result push is unused here, so only the base is set. + healthcheck_push_base: "{{ gatus_api }}" + healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" + when: monitored_services | default([]) | length > 0 diff --git a/ansible/infra/402_public_monitoring.yml b/ansible/infra/402_public_monitoring.yml new file mode 100644 index 0000000..b2c329b --- /dev/null +++ b/ansible/infra/402_public_monitoring.yml @@ -0,0 +1,125 @@ +--- +# Domain expiry, DNS correctness, and public endpoint reachability. +# +# These are the first checks in the estate that PULL rather than push, and that +# is the right way round for them: all three are about how the outside world +# sees us, so they must be measured from outside. Gatus polls from the +# observability host and needs nothing installed anywhere else - there is no +# script, no timer and no token, because nothing is reporting in. +# +# That also means these have no heartbeat. A heartbeat answers "did the thing +# that was supposed to report in do so"; when Gatus does the checking itself, +# failure is immediate and self-evident. + +- name: Register the public-facing checks with Gatus + hosts: observability + become: yes + + vars: + # Expected A records, derived from inventory rather than written down again. + # The estate's recurring bug is an address recorded in a second place and + # then left behind when the machine moved, so the check asserts against + # ansible_host - if a box is renumbered, inventory is the one edit. + dns_records: + - {sub: "{{ subdomains.gatus }}", host: monitoring} + - {sub: "{{ subdomains.ntfy }}", host: watchtower} + - {sub: "{{ subdomains.headscale }}", host: spacey} + - {sub: "{{ subdomains.vaultwarden }}", host: vipy} + - {sub: "{{ subdomains.forgejo }}", host: vipy} + - {sub: "{{ subdomains.lnbits }}", host: vipy} + - {sub: "{{ subdomains.ntfy_emergency_app }}", host: vipy} + - {sub: "{{ subdomains.personal_blog }}", host: vipy} + - {sub: "{{ subdomains.memos }}", host: vipy} + - {sub: "{{ subdomains.mempool }}", host: vipy} + - {sub: "{{ subdomains.datum_gateway }}", host: vipy} + + # A public resolver on purpose: this must test what the internet sees, not + # what a local cache or the tailnet's MagicDNS happens to answer. + dns_resolver: "1.1.1.1" + + # Expected status per site, checked live before being written down. + # 401 is the CORRECT answer for the two behind basic auth - asserting 200 + # there would go green precisely when the auth broke. + public_sites: + - {name: gatus, sub: "{{ subdomains.gatus }}", path: "/", status: 401} + - {name: ntfy, sub: "{{ subdomains.ntfy }}", path: "/", status: 200} + - {name: headscale, sub: "{{ subdomains.headscale }}", path: "/health", status: 200} + - {name: vaultwarden, sub: "{{ subdomains.vaultwarden }}", path: "/", status: 200} + - {name: forgejo, sub: "{{ subdomains.forgejo }}", path: "/", status: 200} + - {name: lnbits, sub: "{{ subdomains.lnbits }}", path: "/", status: 200} + - {name: avisame, sub: "{{ subdomains.ntfy_emergency_app }}", path: "/", status: 200} + - {name: blog, sub: "{{ subdomains.personal_blog }}", path: "/", status: 200} + - {name: memos, sub: "{{ subdomains.memos }}", path: "/", status: 200} + - {name: mempool, sub: "{{ subdomains.mempool }}", path: "/", status: 200} + - {name: datum, sub: "{{ subdomains.datum_gateway }}", path: "/", status: 401} + + # Ports published from the edge host by socket_proxy. + public_tcp: + - {name: bitcoin-p2p, host: vipy, port: "{{ hostvars['knots_box_local'].bitcoin_p2p_port }}"} + - {name: fulcrum-ssl, host: vipy, port: "{{ hostvars['fulcrum_box_local'].fulcrum_ssl_port }}"} + - {name: datum-stratum, host: vipy, port: "{{ hostvars['knots_box_local'].datum_gateway_stratum_port }}"} + + tasks: + # ── Domain expiry ──────────────────────────────────────────────────────── + - name: Build the domain endpoint + ansible.builtin.set_fact: + domain_endpoints: + - name: "{{ root_domain }}" + group: domain + # Needs a scheme: Gatus derives the endpoint TYPE from the URL prefix + # (endpoint.Type()), and a bare domain is UNKNOWN and rejected. The + # apex points at the registrar's parking page, which is irrelevant - + # the only condition here is the WHOIS expiry, and no status check is + # asserted, so what the page serves does not matter. + url: "https://{{ root_domain }}" + # 24h, and upstream enforces a 5m minimum for DOMAIN_EXPIRATION + # anyway because it uses a free whois service that must not be + # hammered and whose data updates slowly. + interval: 24h + # 336h = 14 days. Renewal is manual at the registrar, so this needs + # enough runway to act on. + conditions: + - "[DOMAIN_EXPIRATION] > 336h" + + # ── DNS ────────────────────────────────────────────────────────────────── + - name: Build the DNS endpoints + ansible.builtin.set_fact: + dns_endpoints: "{{ dns_endpoints | default([]) + [{ + 'name': item.sub ~ '.' ~ root_domain, + 'group': 'dns', + 'url': dns_resolver, + 'interval': '24h', + 'dns': {'query-type': 'A', 'query-name': item.sub ~ '.' ~ root_domain}, + 'conditions': ['[DNS_RCODE] == NOERROR', + '[BODY] == ' ~ hostvars[item.host].ansible_host]}] }}" + loop: "{{ dns_records }}" + + # ── Public HTTP ────────────────────────────────────────────────────────── + - name: Build the public HTTP endpoints + ansible.builtin.set_fact: + http_endpoints: "{{ http_endpoints | default([]) + [{ + 'name': item.name, + 'group': 'public', + 'url': 'https://' ~ item.sub ~ '.' ~ root_domain ~ item.path, + 'interval': '5m', + 'conditions': ['[STATUS] == ' ~ item.status, + '[CERTIFICATE_EXPIRATION] > 168h']}] }}" + loop: "{{ public_sites }}" + + # ── Public TCP ─────────────────────────────────────────────────────────── + - name: Build the public TCP endpoints + ansible.builtin.set_fact: + tcp_endpoints: "{{ tcp_endpoints | default([]) + [{ + 'name': item.name, + 'group': 'public', + 'url': 'tcp://' ~ hostvars[item.host].ansible_host ~ ':' ~ item.port, + 'interval': '5m', + 'conditions': ['[CONNECTED] == true']}] }}" + loop: "{{ public_tcp }}" + + - name: Register the public-facing endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: public + gatus_endpoint_pulled: "{{ domain_endpoints + dns_endpoints + http_endpoints + tcp_endpoints }}" diff --git a/ansible/roles/gatus_endpoint/defaults/main.yml b/ansible/roles/gatus_endpoint/defaults/main.yml index 7d49d7c..89b669d 100644 --- a/ansible/roles/gatus_endpoint/defaults/main.yml +++ b/ansible/roles/gatus_endpoint/defaults/main.yml @@ -9,6 +9,11 @@ gatus_endpoint_name: "" # PULLED endpoints - Gatus makes the request and evaluates conditions. # - {name, group, url, interval, conditions: [...], alerts: [...]} +# +# A DNS check adds `dns: {query-type, query-name}` - and note that for those, +# `url` is the RESOLVER to ask, not the name being looked up. +# A domain-expiry check is just `url: ` with a [DOMAIN_EXPIRATION] +# condition; it uses WHOIS/RDAP and needs no scheme. gatus_endpoint_pulled: [] # EXTERNAL endpoints - the host pushes its own result. Gatus never reaches out, diff --git a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 index 1c84996..fa13d1f 100644 --- a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 +++ b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 @@ -28,6 +28,18 @@ endpoints: group: {{ e.group }} url: "{{ e.url }}" interval: {{ e.interval | default('60s') }} +{% if e.dns is defined %} + # A DNS endpoint: `url` is the RESOLVER to ask, not the thing being asked + # about. The name being queried lives in query-name, and [BODY] holds the + # resolved record for the conditions below. + dns: + query-type: {{ e.dns['query-type'] }} + query-name: {{ e.dns['query-name'] }} +{% endif %} +{% if e.client is defined %} + client: +{{ e.client | to_nice_yaml(indent=2) | indent(6, true) }} +{% endif %} conditions: {% for c in e.conditions %} - "{{ c }}" diff --git a/ansible/roles/healthcheck/defaults/main.yml b/ansible/roles/healthcheck/defaults/main.yml index d29cbd1..526347a 100644 --- a/ansible/roles/healthcheck/defaults/main.yml +++ b/ansible/roles/healthcheck/defaults/main.yml @@ -28,6 +28,17 @@ healthcheck_boot_delay: "2min" healthcheck_push_url: "" healthcheck_push_token: "" +# Some checks report MORE THAN ONE result - a host with four systemd services +# needs four endpoints, or a single red light cannot tell you which one died. +# Those set healthcheck_push_base to the endpoints COLLECTION and the check body +# appends each key itself, the same way check-backups.sh reports per source. +healthcheck_push_base: "" + +# Units for the systemd-units check. Each becomes its own Gatus endpoint. +healthcheck_units: [] +# Prefix for the per-unit endpoint keys, e.g. "services_vipy" -> services_vipy-caddy. +healthcheck_units_key_prefix: "" + healthcheck_script_dir: /usr/local/bin healthcheck_log_dir: /var/log/healthchecks diff --git a/ansible/roles/healthcheck/templates/checks/systemd-units.sh.j2 b/ansible/roles/healthcheck/templates/checks/systemd-units.sh.j2 new file mode 100644 index 0000000..90c6193 --- /dev/null +++ b/ansible/roles/healthcheck/templates/checks/systemd-units.sh.j2 @@ -0,0 +1,33 @@ + # One result PER UNIT, not one for the host. A host running four services + # needs four endpoints: a single red light would tell you "something on vipy + # is down" without saying which, and that is the question you actually have. + # + # Keys are host-qualified because unit names collide - caddy runs on four + # machines. Gatus builds the key as sanitize(group)_sanitize(name), so + # group "services" + name "vipy/caddy" becomes services_vipy-caddy. + local prefix="{{ healthcheck_units_key_prefix }}" + local down="" + + for unit in {{ healthcheck_units | join(' ') }}; do + local state substate + state=$(systemctl is-active "$unit" 2>/dev/null || true) + substate=$(systemctl show -p SubState --value "$unit" 2>/dev/null || true) + + if [ "$state" = "active" ]; then + report_key "${prefix}-${unit}" "true" "${unit} active (${substate:-running})" + else + # `failed` and `inactive` are different stories: one crashed, one was + # stopped. Both are down, but the message should say which. + local detail="${unit} is ${state:-unknown}" + [ "$state" = "failed" ] && detail="${detail} (${substate:-failed})" + report_key "${prefix}-${unit}" "false" "$detail" + down="${down}${down:+, }${unit}=${state:-unknown}" + fi + done + + if [ -n "$down" ]; then + MESSAGE="down: ${down}" + return 1 + fi + MESSAGE="all {{ healthcheck_units | length }} units active" + return 0 diff --git a/ansible/roles/healthcheck/templates/healthcheck.service.j2 b/ansible/roles/healthcheck/templates/healthcheck.service.j2 index 574e37c..3ddd8aa 100644 --- a/ansible/roles/healthcheck/templates/healthcheck.service.j2 +++ b/ansible/roles/healthcheck/templates/healthcheck.service.j2 @@ -9,6 +9,7 @@ User=root ExecStart={{ healthcheck_script_dir }}/{{ healthcheck_name }}-healthcheck.sh Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} +Environment=HEALTHCHECK_PUSH_BASE={{ healthcheck_push_base }} StandardOutput=journal StandardError=journal diff --git a/ansible/roles/healthcheck/templates/healthcheck.sh.j2 b/ansible/roles/healthcheck/templates/healthcheck.sh.j2 index 92c9d27..6667878 100644 --- a/ansible/roles/healthcheck/templates/healthcheck.sh.j2 +++ b/ansible/roles/healthcheck/templates/healthcheck.sh.j2 @@ -11,6 +11,8 @@ set -uo pipefail LOG_FILE="{{ healthcheck_log_dir }}/{{ healthcheck_name }}.log" PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" +# Set only by checks that report several results; see healthcheck_push_base. +PUSH_BASE="${HEALTHCHECK_PUSH_BASE:-}" log() { echo "$(date '+%Y-%m-%d %H:%M:%S') - $*" >> "$LOG_FILE"; } @@ -40,6 +42,18 @@ report() { fi } +# Report to an arbitrary endpoint key under PUSH_BASE. Used by checks that +# produce one result per item rather than a single verdict. +report_key() { + local key="$1" success="$2" message="$3" + [ -n "$PUSH_BASE" ] || return 0 + local encoded + encoded=$(printf '%s' "$message" | sed 's/%/%25/g; s/ /%20/g; s/&/%26/g; s/+/%2B/g; s/#/%23/g') + curl -s -o /dev/null --max-time 15 --retry 2 --retry-delay 3 -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_BASE}/${key}/external?success=${success}&error=${encoded}" 2>/dev/null || true +} + # ── the check ──────────────────────────────────────────────────────────────── # # The blank line before the closing brace below is load-bearing. Jinja strips an From 99760dfd46ab79bae4a6423961d06d4ecdd033b1 Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 09:53:15 +0200 Subject: [PATCH 62/67] monitoring: track arbret.com expiry too Domains are now a list in group_vars/all (monitored_domains) rather than the single root_domain hardcoded in the playbook, so adding one is a line of data instead of a change to the task. arbret.com resolves to prd-arbret (167.99.242.62) and its RDAP record exposes an expiry of 2027-02-18, so the check reads real data rather than silently passing on a missing field - verified against rdap.verisign.com before wiring it up. Both domains checked daily with two weeks of runway, because registration renewal is a manual act at the registrar and losing a domain is not recoverable the way losing a host is. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/main.yml | 14 ++++++++++ ansible/infra/402_public_monitoring.yml | 35 ++++++++++++------------- 2 files changed, 31 insertions(+), 18 deletions(-) diff --git a/ansible/group_vars/all/main.yml b/ansible/group_vars/all/main.yml index 9cf3143..0ff8ef8 100644 --- a/ansible/group_vars/all/main.yml +++ b/ansible/group_vars/all/main.yml @@ -58,3 +58,17 @@ subdomains: # group covers them, so these are global rather than group_vars/. ntfy_topic: alerts headscale_namespace: counter-net + +# ───────────────────────────────────────────────────────────────────────────── +# Domains whose registration expiry is monitored (infra/402_public_monitoring). +# +# Registration renewal is a manual act at the registrar, and losing a domain is +# not recoverable in the way losing a host is - so these are checked daily and +# alarm with two weeks of runway. +# +# root_domain is the estate's own domain; the rest are domains we own that are +# served from it or from a host in the inventory. +# ───────────────────────────────────────────────────────────────────────────── +monitored_domains: + - "{{ root_domain }}" + - arbret.com diff --git a/ansible/infra/402_public_monitoring.yml b/ansible/infra/402_public_monitoring.yml index b2c329b..7ff7cf8 100644 --- a/ansible/infra/402_public_monitoring.yml +++ b/ansible/infra/402_public_monitoring.yml @@ -61,25 +61,24 @@ tasks: # ── Domain expiry ──────────────────────────────────────────────────────── - - name: Build the domain endpoint + # Each domain needs a URL SCHEME: Gatus derives the endpoint type from the + # prefix (endpoint.Type()), so a bare "example.com" is UNKNOWN and the whole + # config is rejected. No status is asserted, only the WHOIS/RDAP expiry, so + # whatever the apex serves - a real site, or the registrar's parking page - + # is irrelevant. + # + # 24h, and upstream enforces a 5m minimum for DOMAIN_EXPIRATION anyway + # because it uses a free whois service that must not be hammered. + # 336h = 14 days of runway, because renewal is a manual act at the registrar. + - name: Build the domain endpoints ansible.builtin.set_fact: - domain_endpoints: - - name: "{{ root_domain }}" - group: domain - # Needs a scheme: Gatus derives the endpoint TYPE from the URL prefix - # (endpoint.Type()), and a bare domain is UNKNOWN and rejected. The - # apex points at the registrar's parking page, which is irrelevant - - # the only condition here is the WHOIS expiry, and no status check is - # asserted, so what the page serves does not matter. - url: "https://{{ root_domain }}" - # 24h, and upstream enforces a 5m minimum for DOMAIN_EXPIRATION - # anyway because it uses a free whois service that must not be - # hammered and whose data updates slowly. - interval: 24h - # 336h = 14 days. Renewal is manual at the registrar, so this needs - # enough runway to act on. - conditions: - - "[DOMAIN_EXPIRATION] > 336h" + domain_endpoints: "{{ domain_endpoints | default([]) + [{ + 'name': item, + 'group': 'domain', + 'url': 'https://' ~ item, + 'interval': '24h', + 'conditions': ['[DOMAIN_EXPIRATION] > 336h']}] }}" + loop: "{{ monitored_domains }}" # ── DNS ────────────────────────────────────────────────────────────────── - name: Build the DNS endpoints From 853e62a19cb79e0bca2be03156c6ccedfadac64f Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 10:18:57 +0200 Subject: [PATCH 63/67] monitoring: retire the Uptime-Kuma-era checks, add ZFS pool capacity Five things deprecated, each verified against the DEPLOYED script before being deleted rather than assumed superseded: infra/410_disk_usage_alerts.yml -> disk-usage check (infra/400) infra/420_system_healthcheck.yml -> liveness check (infra/400) infra/430_cpu_temp_alerts.yml -> cpu-temp check (infra/400) 32_zfs play 2 (monitoring half) -> zfs-health check (infra/400) 34_nut play 2 (entirely) -> ups-status check (infra/400) Nothing is lost by the swap. The old system_healthcheck.sh only computed uptime and pushed, which is exactly a liveness heartbeat. The old disk monitor was WEAKER than its replacement: it checked "/" alone at 80%, where the new one walks every real filesystem at 85%. Deleting the playbooks was not the hard part. The units they installed live on the hosts, enabled, and keep firing regardless of what the repo says - two of them were still pushing to uptime.contrapeso.xyz every 15 minutes across nine machines. A playbook deleted without a cleanup leaves its output running forever with nothing left to explain it. So infra/409_remove_legacy_monitoring stops, disables and removes the units, deletes /opt/{disk-monitoring, system-healthcheck,nodito-monitoring,zfs-monitoring}, and removes the orphaned hand-written ups-heartbeat.sh. It ends by grepping for any surviving Kuma reference and reporting it. Kept permanently and idempotent, so a rebuilt or restored host cannot quietly bring them back. A trap avoided: the monthly ZFS scrub lived INSIDE 32_zfs play 2. Deleting the play wholesale would have silently stopped scrubbing the pool - and an unscrubbed pool makes the health check meaningless, because it would have nothing true to report. That play is now scrub-only and check-runs ok=5 changed=0. ZFS pool capacity added as a sixth condition to the zfs-health check. `zpool status` reports a 95% full pool as perfectly ONLINE, so capacity has to be read separately with `zpool list` - and it is the failure you get warning of rather than the one you discover. Threshold 80%, because ZFS allocation degrades badly past roughly that and fragmentation is hard to undo. The pool is at 47%. Verified both directions: passes on the real pool, and a simulated 91% exits 1. Also fixed the waste recorded in c2de6db: the healthcheck role installed its dependencies once per CHECK rather than per HOST - 29 apt transactions estate-wide for a curl already present, and the slowest part of every deploy. It now deduplicates within a play run, and the redundant standalone daemon_reload is gone (the systemd task already does one). site.yml updated, which exposed that services/gatus was never in it. It now runs before the three registration playbooks, since registering endpoints against a Gatus that is not yet serving would simply fail. Verified: no legacy timer remains on any host; 83 endpoints, 83 UP, 0 DOWN. Co-Authored-By: Claude Opus 5 (1M context) --- .../infra/409_remove_legacy_monitoring.yml | 109 ++++++ ansible/infra/410_disk_usage_alerts.yml | 338 ------------------ ansible/infra/420_system_healthcheck.yml | 320 ----------------- ansible/infra/430_cpu_temp_alerts.yml | 324 ----------------- .../nodito/32_zfs_pool_setup_playbook.yml | 130 ++----- .../nodito/34_nut_ups_setup_playbook.yml | 124 +------ .../nodito/templates/ups-heartbeat.service.j2 | 14 - .../nodito/templates/ups-heartbeat.timer.j2 | 11 - .../nodito/templates/ups_heartbeat.sh.j2 | 68 ---- .../templates/zfs-health-monitor.service.j2 | 14 - .../templates/zfs-health-monitor.timer.j2 | 11 - .../nodito/templates/zfs_health_monitor.sh.j2 | 181 ---------- ansible/roles/healthcheck/defaults/main.yml | 5 + ansible/roles/healthcheck/tasks/main.yml | 22 +- .../templates/checks/zfs-health.sh.j2 | 13 +- ansible/site.yml | 20 +- 16 files changed, 184 insertions(+), 1520 deletions(-) create mode 100644 ansible/infra/409_remove_legacy_monitoring.yml delete mode 100644 ansible/infra/410_disk_usage_alerts.yml delete mode 100644 ansible/infra/420_system_healthcheck.yml delete mode 100644 ansible/infra/430_cpu_temp_alerts.yml delete mode 100644 ansible/infra/nodito/templates/ups-heartbeat.service.j2 delete mode 100644 ansible/infra/nodito/templates/ups-heartbeat.timer.j2 delete mode 100644 ansible/infra/nodito/templates/ups_heartbeat.sh.j2 delete mode 100644 ansible/infra/nodito/templates/zfs-health-monitor.service.j2 delete mode 100644 ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 delete mode 100644 ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 diff --git a/ansible/infra/409_remove_legacy_monitoring.yml b/ansible/infra/409_remove_legacy_monitoring.yml new file mode 100644 index 0000000..8f57e46 --- /dev/null +++ b/ansible/infra/409_remove_legacy_monitoring.yml @@ -0,0 +1,109 @@ +--- +# Remove the Uptime-Kuma-era monitoring that 400/401/402 replaced. +# +# Deleting the playbooks that installed these is NOT enough: the units are on +# the hosts, enabled, and keep firing regardless of what the repo says. Two of +# them still push to https://uptime.contrapeso.xyz every 15 minutes. A playbook +# that is deleted without a cleanup leaves its output running forever, with +# nothing in the repo left to explain it. +# +# What replaced what, all verified against the deployed scripts before removal: +# +# disk-usage-monitor -> disk-usage-healthcheck (infra/400) +# The old one checked ONLY "/" at 80%. The replacement walks every real +# filesystem, excluding tmpfs/devtmpfs/squashfs/overlay, at 85%. Strictly +# more coverage, so nothing is lost. +# +# system-healthcheck -> liveness-healthcheck (infra/400) +# The old script computed uptime and pushed. That is exactly a liveness +# heartbeat and nothing more. +# +# nodito-cpu-temp-monitor -> cpu-temp-healthcheck (infra/400) +# zfs-health-monitor -> zfs-health-healthcheck (infra/400) +# The ZFS check logic was ported verbatim - same five conditions - so only +# the reporting transport changed. +# +# NOT removed, because they are not monitoring: +# zfs-monthly-scrub.{timer,service} the actual scrub (infra/nodito/32) +# pull-backups, check-backups the backup machinery (playbooks/backups) +# +# This play is idempotent and kept permanently rather than run once and deleted: +# on a host that never had these it does nothing, and it guarantees a rebuilt or +# restored machine cannot quietly bring them back. + +- name: Remove the legacy Uptime Kuma monitoring units + hosts: managed + become: yes + + vars: + legacy_units: + - disk-usage-monitor + - system-healthcheck + - nodito-cpu-temp-monitor + - zfs-health-monitor + legacy_dirs: + - /opt/disk-monitoring + - /opt/system-healthcheck + - /opt/nodito-monitoring + - /opt/zfs-monitoring + + tasks: + - name: Find which legacy units exist here + ansible.builtin.stat: + path: "/etc/systemd/system/{{ item.0 }}.{{ item.1 }}" + register: legacy_unit_files + loop: "{{ legacy_units | product(['timer', 'service']) | list }}" + + # Stop and disable BEFORE deleting the unit file: systemd cannot disable a + # unit whose file has already gone, which would leave a dangling symlink in + # multi-user.target.wants and a warning on every daemon-reload. + - name: Stop and disable the legacy units + ansible.builtin.systemd: + name: "{{ item.item.0 }}.{{ item.item.1 }}" + state: stopped + enabled: no + loop: "{{ legacy_unit_files.results }}" + loop_control: + label: "{{ item.item.0 }}.{{ item.item.1 }}" + when: item.stat.exists + failed_when: false + + - name: Remove the legacy unit files + ansible.builtin.file: + path: "/etc/systemd/system/{{ item.item.0 }}.{{ item.item.1 }}" + state: absent + loop: "{{ legacy_unit_files.results }}" + loop_control: + label: "{{ item.item.0 }}.{{ item.item.1 }}" + when: item.stat.exists + + - name: Reload systemd + ansible.builtin.systemd: + daemon_reload: yes + + - name: Remove the legacy monitoring scripts and their logs + ansible.builtin.file: + path: "{{ item }}" + state: absent + loop: "{{ legacy_dirs }}" + + # An orphan predating all of this: mode 0644, not executable, referenced by + # no unit and no cron entry, pushing to a Kuma monitor. Superseded by + # ups-status-healthcheck. + - name: Remove the orphaned hand-written UPS heartbeat + ansible.builtin.file: + path: /usr/local/bin/ups-heartbeat.sh + state: absent + + - name: Confirm nothing still pushes to Uptime Kuma + ansible.builtin.shell: >- + grep -rl "uptime.contrapeso.xyz" /etc/systemd/system /usr/local/bin /opt 2>/dev/null || true + register: kuma_refs + changed_when: false + + - name: Report any remaining references + ansible.builtin.debug: + msg: >- + {{ 'clean - nothing references Uptime Kuma' + if kuma_refs.stdout | trim | length == 0 + else 'STILL REFERENCING KUMA: ' ~ kuma_refs.stdout_lines | join(', ') }} diff --git a/ansible/infra/410_disk_usage_alerts.yml b/ansible/infra/410_disk_usage_alerts.yml deleted file mode 100644 index b286add..0000000 --- a/ansible/infra/410_disk_usage_alerts.yml +++ /dev/null @@ -1,338 +0,0 @@ -# ═════════════════════════════════════════════════════════════════════════════ -# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. -# -# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username -# and uptime_kuma_password were removed from the vault, so the "Validate Uptime -# Kuma configuration" assert stops it before anything is installed or changed. -# -# It is kept because the CHECK LOGIC is the durable part — what gets measured, -# the thresholds, and the systemd timer plumbing. When something replaces Uptime -# Kuma, only the push transport needs rewriting; the rest still applies. -# -# What was being monitored: archive/uptime_kuma/MONITORS.md -# ═════════════════════════════════════════════════════════════════════════════ -- name: Deploy Disk Usage Monitoring - hosts: managed - become: yes - - vars: - disk_usage_threshold_percent: 80 - disk_check_interval_minutes: 15 - monitored_mount_point: "/" - monitoring_script_dir: /opt/disk-monitoring - monitoring_script_path: "{{ monitoring_script_dir }}/disk_usage_monitor.sh" - log_file: "{{ monitoring_script_dir }}/disk_usage_monitor.log" - systemd_service_name: disk-usage-monitor - # Uptime Kuma configuration (auto-configured from group_vars/all/) - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - - name: Validate Uptime Kuma configuration - assert: - that: - - uptime_kuma_api_url is defined - - uptime_kuma_api_url != "" - - uptime_kuma_username is defined - - uptime_kuma_username != "" - - uptime_kuma_password is defined - - uptime_kuma_password != "" - fail_msg: "uptime_kuma_api_url, uptime_kuma_username and uptime_kuma_password must be set" - - - name: Get hostname for monitor identification - command: hostname - register: host_name - changed_when: false - - - name: Set monitor name and group based on hostname and mount point - set_fact: - monitor_name: "disk-usage-{{ host_name.stdout }}-{{ monitored_mount_point | replace('/', 'root') }}" - monitor_friendly_name: "Disk Usage: {{ host_name.stdout }} ({{ monitored_mount_point }})" - uptime_kuma_monitor_group: "{{ host_name.stdout }} - infra" - - - name: Create Uptime Kuma monitor setup script - copy: - dest: /tmp/setup_uptime_kuma_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - ntfy_topic = sys.argv[8] if len(sys.argv) > 8 else "alerts" - - api = UptimeKumaApi(api_url, timeout=60, wait_events=2.0) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Get all notifications and find ntfy notification - notifications = api.get_notifications() - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - # Find or create group - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name=group_name) - # Refresh to get the full group object with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - # Find or create/update push monitor - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': True, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - monitor = api.edit_monitor(existing_monitor['id'], **monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - else: - monitor_result = api.add_monitor(**monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - # Output result as JSON - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma monitor setup script - command: > - {{ ansible_playbook_python }} - /tmp/setup_uptime_kuma_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "{{ uptime_kuma_monitor_group }}" - "{{ monitor_name }}" - "{{ monitor_friendly_name }} - Alerts when usage exceeds {{ disk_usage_threshold_percent }}%" - "{{ (disk_check_interval_minutes * 60) + 60 }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL and monitor ID as facts - set_fact: - uptime_kuma_disk_usage_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - uptime_kuma_monitor_id: "{{ monitor_info_parsed.monitor_id }}" - - - name: Install required packages for disk monitoring - package: - name: - - curl - state: present - - - name: Create monitoring script directory - file: - path: "{{ monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create disk usage monitoring script - copy: - dest: "{{ monitoring_script_path }}" - content: | - #!/bin/bash - - # Disk Usage Monitoring Script - # Monitors disk usage and sends alerts to Uptime Kuma - # Mode: "No news is good news" - only sends alerts when disk usage is HIGH - - LOG_FILE="{{ log_file }}" - USAGE_THRESHOLD="{{ disk_usage_threshold_percent }}" - UPTIME_KUMA_URL="{{ uptime_kuma_disk_usage_push_url }}" - MOUNT_POINT="{{ monitored_mount_point }}" - - # Function to log messages - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - # Function to get disk usage percentage - get_disk_usage() { - local mount_point="$1" - local usage="" - - # Get disk usage percentage (without % sign) - usage=$(df -h "$mount_point" 2>/dev/null | awk 'NR==2 {gsub(/%/, "", $5); print $5}') - - if [ -z "$usage" ]; then - log_message "ERROR: Could not read disk usage for $mount_point" - return 1 - fi - - echo "$usage" - } - - # Function to get disk usage details - get_disk_details() { - local mount_point="$1" - df -h "$mount_point" 2>/dev/null | awk 'NR==2 {print "Used: "$3" / Total: "$2" ("$5" full)"}' - } - - # Function to send alert to Uptime Kuma when disk usage exceeds threshold - # With upside-down mode enabled, sending status=up will trigger an alert - send_uptime_kuma_alert() { - local usage="$1" - local details="$2" - local message="DISK FULL WARNING: ${MOUNT_POINT} is ${usage}% full (Threshold: ${USAGE_THRESHOLD}%) - ${details}" - - log_message "ALERT: $message" - - # Send push notification to Uptime Kuma with status=up - # In upside-down mode, status=up is treated as down/alert - response=$(curl -s -w "\n%{http_code}" -G \ - --data-urlencode "status=up" \ - --data-urlencode "msg=$message" \ - "$UPTIME_KUMA_URL" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Alert sent successfully to Uptime Kuma (HTTP $http_code)" - else - log_message "ERROR: Failed to send alert to Uptime Kuma (HTTP $http_code)" - fi - } - - # Main monitoring logic - main() { - log_message "Starting disk usage check for $MOUNT_POINT" - - # Get current disk usage - current_usage=$(get_disk_usage "$MOUNT_POINT") - - if [ $? -ne 0 ] || [ -z "$current_usage" ]; then - log_message "ERROR: Could not read disk usage" - exit 1 - fi - - # Get disk details - disk_details=$(get_disk_details "$MOUNT_POINT") - - log_message "Current disk usage: ${current_usage}% - $disk_details" - - # Check if usage exceeds threshold - if [ "$current_usage" -gt "$USAGE_THRESHOLD" ]; then - log_message "WARNING: Disk usage ${current_usage}% exceeds threshold ${USAGE_THRESHOLD}%" - send_uptime_kuma_alert "$current_usage" "$disk_details" - else - log_message "Disk usage is within normal range - no alert needed (no news is good news)" - fi - } - - # Run main function - main - owner: root - group: root - mode: '0755' - - - name: Create systemd service for disk usage monitoring - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.service" - content: | - [Unit] - Description=Disk Usage Monitor - After=network.target - - [Service] - Type=oneshot - ExecStart={{ monitoring_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for disk usage monitoring - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.timer" - content: | - [Unit] - Description=Run Disk Usage Monitor every {{ disk_check_interval_minutes }} minute(s) - Requires={{ systemd_service_name }}.service - - [Timer] - OnBootSec={{ disk_check_interval_minutes }}min - OnUnitActiveSec={{ disk_check_interval_minutes }}min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start disk usage monitoring timer - systemd: - name: "{{ systemd_service_name }}.timer" - enabled: yes - state: started - - - name: Test disk usage monitoring script - command: "{{ monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "Disk usage monitoring script failed to execute properly" - - - name: Clean up temporary Uptime Kuma setup script - file: - path: /tmp/setup_uptime_kuma_monitor.py - state: absent - delegate_to: localhost - become: no diff --git a/ansible/infra/420_system_healthcheck.yml b/ansible/infra/420_system_healthcheck.yml deleted file mode 100644 index 9813d0c..0000000 --- a/ansible/infra/420_system_healthcheck.yml +++ /dev/null @@ -1,320 +0,0 @@ -# ═════════════════════════════════════════════════════════════════════════════ -# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. -# -# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username -# and uptime_kuma_password were removed from the vault, so the "Validate Uptime -# Kuma configuration" assert stops it before anything is installed or changed. -# -# It is kept because the CHECK LOGIC is the durable part — what gets measured, -# the thresholds, and the systemd timer plumbing. When something replaces Uptime -# Kuma, only the push transport needs rewriting; the rest still applies. -# -# What was being monitored: archive/uptime_kuma/MONITORS.md -# ═════════════════════════════════════════════════════════════════════════════ -- name: Deploy System Healthcheck Monitoring - hosts: managed - become: yes - - vars: - healthcheck_interval_seconds: 60 # Send healthcheck every 60 seconds (1 minute) - healthcheck_timeout_seconds: 90 # Uptime Kuma should alert if no ping received within 90s - healthcheck_retries: 1 # Number of retries before alerting - monitoring_script_dir: /opt/system-healthcheck - monitoring_script_path: "{{ monitoring_script_dir }}/system_healthcheck.sh" - log_file: "{{ monitoring_script_dir }}/system_healthcheck.log" - systemd_service_name: system-healthcheck - # Uptime Kuma configuration (auto-configured from group_vars/all/) - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - - name: Validate Uptime Kuma configuration - assert: - that: - - uptime_kuma_api_url is defined - - uptime_kuma_api_url != "" - - uptime_kuma_username is defined - - uptime_kuma_username != "" - - uptime_kuma_password is defined - - uptime_kuma_password != "" - fail_msg: "uptime_kuma_api_url, uptime_kuma_username and uptime_kuma_password must be set" - - - name: Get hostname for monitor identification - command: hostname - register: host_name - changed_when: false - - - name: Set monitor name and group based on hostname - set_fact: - monitor_name: "system-healthcheck-{{ host_name.stdout }}" - monitor_friendly_name: "System Healthcheck: {{ host_name.stdout }}" - uptime_kuma_monitor_group: "{{ host_name.stdout }} - infra" - - - name: Create Uptime Kuma monitor setup script - copy: - dest: /tmp/setup_uptime_kuma_healthcheck_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - retries = int(sys.argv[8]) - ntfy_topic = sys.argv[9] if len(sys.argv) > 9 else "alerts" - - api = UptimeKumaApi(api_url, timeout=120, wait_events=2.0) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Get all notifications and find ntfy notification - notifications = api.get_notifications() - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - # Find or create group - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name=group_name) - # Refresh to get the full group object with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - # Find or create/update push monitor - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': False, # Normal mode: receiving pings = healthy - 'maxretries': retries, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - monitor = api.edit_monitor(existing_monitor['id'], **monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - else: - monitor_result = api.add_monitor(**monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - # Output result as JSON - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma monitor setup script - command: > - {{ ansible_playbook_python }} - /tmp/setup_uptime_kuma_healthcheck_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "{{ uptime_kuma_monitor_group }}" - "{{ monitor_name }}" - "{{ monitor_friendly_name }} - Regular healthcheck ping every {{ healthcheck_interval_seconds }}s" - "{{ healthcheck_timeout_seconds }}" - "{{ healthcheck_retries }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL and monitor ID as facts - set_fact: - uptime_kuma_healthcheck_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - uptime_kuma_monitor_id: "{{ monitor_info_parsed.monitor_id }}" - - - name: Install required packages for healthcheck monitoring - package: - name: - - curl - state: present - - - name: Create monitoring script directory - file: - path: "{{ monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create system healthcheck script - copy: - dest: "{{ monitoring_script_path }}" - content: | - #!/bin/bash - - # System Healthcheck Script - # Sends regular heartbeat pings to Uptime Kuma - # This ensures the system is running and able to communicate - - LOG_FILE="{{ log_file }}" - UPTIME_KUMA_URL="{{ uptime_kuma_healthcheck_push_url }}" - HOSTNAME=$(hostname) - - # Function to log messages - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - # Function to send healthcheck ping to Uptime Kuma - send_healthcheck() { - local uptime_seconds=$(awk '{print int($1)}' /proc/uptime) - local uptime_days=$((uptime_seconds / 86400)) - local uptime_hours=$(((uptime_seconds % 86400) / 3600)) - local uptime_minutes=$(((uptime_seconds % 3600) / 60)) - - local message="System healthy - Uptime: ${uptime_days}d ${uptime_hours}h ${uptime_minutes}m" - - log_message "Sending healthcheck ping: $message" - - # Send push notification to Uptime Kuma with status=up - encoded_message=$(printf '%s\n' "$message" | sed 's/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') - response=$(curl -s -w "\n%{http_code}" "$UPTIME_KUMA_URL?status=up&msg=$encoded_message" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Healthcheck ping sent successfully (HTTP $http_code)" - else - log_message "ERROR: Failed to send healthcheck ping (HTTP $http_code)" - return 1 - fi - } - - # Main healthcheck logic - main() { - log_message "Starting system healthcheck for $HOSTNAME" - - # Send healthcheck ping - if send_healthcheck; then - log_message "Healthcheck completed successfully" - else - log_message "ERROR: Healthcheck failed" - exit 1 - fi - } - - # Run main function - main - owner: root - group: root - mode: '0755' - - - name: Create systemd service for system healthcheck - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.service" - content: | - [Unit] - Description=System Healthcheck Monitor - After=network.target - - [Service] - Type=oneshot - ExecStart={{ monitoring_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for system healthcheck - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.timer" - content: | - [Unit] - Description=Run System Healthcheck every minute - Requires={{ systemd_service_name }}.service - - [Timer] - OnBootSec=30sec - OnUnitActiveSec={{ healthcheck_interval_seconds }}sec - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start system healthcheck timer - systemd: - name: "{{ systemd_service_name }}.timer" - enabled: yes - state: started - - - name: Test system healthcheck script - command: "{{ monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "System healthcheck script failed to execute properly" - - - name: Display monitor information - debug: - msg: | - ✓ System healthcheck monitoring deployed successfully! - - Monitor Name: {{ monitor_friendly_name }} - Monitor Group: {{ uptime_kuma_monitor_group }} - Healthcheck Interval: Every {{ healthcheck_interval_seconds }} seconds (1 minute) - Timeout: {{ healthcheck_timeout_seconds }} seconds (90s) - Retries: {{ healthcheck_retries }} - - The system will send a heartbeat ping every minute. - Uptime Kuma will alert if no ping is received within 90 seconds (with 1 retry). - - - name: Clean up temporary Uptime Kuma setup script - file: - path: /tmp/setup_uptime_kuma_healthcheck_monitor.py - state: absent - delegate_to: localhost - become: no - diff --git a/ansible/infra/430_cpu_temp_alerts.yml b/ansible/infra/430_cpu_temp_alerts.yml deleted file mode 100644 index 2e00bdb..0000000 --- a/ansible/infra/430_cpu_temp_alerts.yml +++ /dev/null @@ -1,324 +0,0 @@ -# ═════════════════════════════════════════════════════════════════════════════ -# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. -# -# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username -# and uptime_kuma_password were removed from the vault, so the "Validate Uptime -# Kuma configuration" assert stops it before anything is installed or changed. -# -# It is kept because the CHECK LOGIC is the durable part — what gets measured, -# the thresholds, and the systemd timer plumbing. When something replaces Uptime -# Kuma, only the push transport needs rewriting; the rest still applies. -# -# What was being monitored: archive/uptime_kuma/MONITORS.md -# ═════════════════════════════════════════════════════════════════════════════ -- name: Deploy CPU Temperature Monitoring - hosts: hypervisor - become: yes - - vars: - temp_threshold_celsius: 80 - temp_check_interval_minutes: 1 - monitoring_script_dir: /opt/nodito-monitoring - monitoring_script_path: "{{ monitoring_script_dir }}/cpu_temp_monitor.sh" - log_file: "{{ monitoring_script_dir }}/cpu_temp_monitor.log" - systemd_service_name: nodito-cpu-temp-monitor - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - - name: Validate Uptime Kuma configuration - assert: - that: - - uptime_kuma_api_url is defined - - uptime_kuma_api_url != "" - - uptime_kuma_username is defined - - uptime_kuma_username != "" - - uptime_kuma_password is defined - - uptime_kuma_password != "" - fail_msg: "uptime_kuma_api_url, uptime_kuma_username and uptime_kuma_password must be set" - - - name: Get hostname for monitor identification - command: hostname - register: host_name - changed_when: false - - - name: Set monitor name and group based on hostname - set_fact: - monitor_name: "cpu-temp-{{ host_name.stdout }}" - monitor_friendly_name: "CPU Temperature: {{ host_name.stdout }}" - uptime_kuma_monitor_group: "{{ host_name.stdout }} - infra" - - - name: Create Uptime Kuma CPU temperature monitor setup script - copy: - dest: /tmp/setup_uptime_kuma_cpu_temp_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - ntfy_topic = sys.argv[8] if len(sys.argv) > 8 else "alerts" - - api = UptimeKumaApi(api_url, timeout=60, wait_events=2.0) - api.login(username, password) - - monitors = api.get_monitors() - notifications = api.get_notifications() - - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - api.add_monitor(type='group', name=group_name) - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': True, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - api.edit_monitor(existing_monitor['id'], **monitor_data) - else: - api.add_monitor(**monitor_data) - - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma monitor setup script - command: > - {{ ansible_playbook_python }} - /tmp/setup_uptime_kuma_cpu_temp_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "{{ uptime_kuma_monitor_group }}" - "{{ monitor_name }}" - "{{ monitor_friendly_name }} - Alerts when temperature exceeds {{ temp_threshold_celsius }}°C" - "{{ (temp_check_interval_minutes * 60) + 60 }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL and monitor ID as facts - set_fact: - uptime_kuma_cpu_temp_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - uptime_kuma_monitor_id: "{{ monitor_info_parsed.monitor_id }}" - - - name: Install required packages for temperature monitoring - package: - name: - - lm-sensors - - curl - - jq - - bc - state: present - - - name: Create monitoring script directory - file: - path: "{{ monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create CPU temperature monitoring script - copy: - dest: "{{ monitoring_script_path }}" - content: | - #!/bin/bash - - # CPU Temperature Monitoring Script - # Monitors CPU temperature and sends alerts to Uptime Kuma - - LOG_FILE="{{ log_file }}" - TEMP_THRESHOLD="{{ temp_threshold_celsius }}" - UPTIME_KUMA_URL="{{ uptime_kuma_cpu_temp_push_url }}" - - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - get_cpu_temp() { - local temp="" - - if command -v sensors >/dev/null 2>&1; then - temp=$(sensors 2>/dev/null | grep -E "Core 0|Package id 0|Tdie|Tctl" | head -1 | grep -oE '[0-9]+\.[0-9]+°C' | grep -oE '[0-9]+\.[0-9]+') - fi - - if [ -z "$temp" ] && [ -f /sys/class/thermal/thermal_zone0/temp ]; then - temp=$(cat /sys/class/thermal/thermal_zone0/temp) - temp=$(echo "scale=1; $temp/1000" | bc -l 2>/dev/null || echo "$temp") - fi - - if [ -z "$temp" ] && command -v acpi >/dev/null 2>&1; then - temp=$(acpi -t 2>/dev/null | grep -oE '[0-9]+\.[0-9]+' | head -1) - fi - - echo "$temp" - } - - send_uptime_kuma_alert() { - local temp="$1" - local message="CPU Temperature Alert: ${temp}°C (Threshold: ${TEMP_THRESHOLD}°C)" - - log_message "ALERT: $message" - - encoded_message=$(printf '%s\n' "$message" | sed 's/ /%20/g; s/°/%C2%B0/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g') - response=$(curl -s -w "\n%{http_code}" "$UPTIME_KUMA_URL?status=up&msg=$encoded_message" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Alert sent successfully to Uptime Kuma (HTTP $http_code)" - else - log_message "ERROR: Failed to send alert to Uptime Kuma (HTTP $http_code)" - fi - } - - main() { - log_message "Starting CPU temperature check" - - current_temp=$(get_cpu_temp) - - if [ -z "$current_temp" ]; then - log_message "ERROR: Could not read CPU temperature" - exit 1 - fi - - log_message "Current CPU temperature: ${current_temp}°C" - - if (( $(echo "$current_temp > $TEMP_THRESHOLD" | bc -l) )); then - log_message "WARNING: CPU temperature ${current_temp}°C exceeds threshold ${TEMP_THRESHOLD}°C" - send_uptime_kuma_alert "$current_temp" - else - log_message "CPU temperature is within normal range" - fi - } - - main - owner: root - group: root - mode: '0755' - - - name: Create systemd service for CPU temperature monitoring - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.service" - content: | - [Unit] - Description=CPU Temperature Monitor - After=network.target - - [Service] - Type=oneshot - ExecStart={{ monitoring_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for CPU temperature monitoring - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.timer" - content: | - [Unit] - Description=Run CPU Temperature Monitor every {{ temp_check_interval_minutes }} minute(s) - Requires={{ systemd_service_name }}.service - - [Timer] - OnBootSec={{ temp_check_interval_minutes }}min - OnUnitActiveSec={{ temp_check_interval_minutes }}min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start CPU temperature monitoring timer - systemd: - name: "{{ systemd_service_name }}.timer" - enabled: yes - state: started - - - name: Test CPU temperature monitoring script - command: "{{ monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "CPU temperature monitoring script failed to execute properly" - - - name: Display monitoring configuration - debug: - msg: - - "CPU Temperature Monitoring configured successfully" - - "Temperature threshold: {{ temp_threshold_celsius }}°C" - - "Check interval: {{ temp_check_interval_minutes }} minute(s)" - - "Monitor Name: {{ monitor_friendly_name }}" - - "Monitor Group: {{ uptime_kuma_monitor_group }}" - - "Uptime Kuma Push URL: {{ uptime_kuma_cpu_temp_push_url }}" - - "Monitoring script: {{ monitoring_script_path }}" - - "Systemd Service: {{ systemd_service_name }}.service" - - "Systemd Timer: {{ systemd_service_name }}.timer" - - - name: Clean up temporary Uptime Kuma setup script - file: - path: /tmp/setup_uptime_kuma_cpu_temp_monitor.py - state: absent - delegate_to: localhost - become: no - diff --git a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml index 1aad2e7..2efbf8c 100644 --- a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml +++ b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml @@ -170,141 +170,55 @@ when: "'ONLINE' not in final_zfs_status.stdout" # ───────────────────────────────────────────────────────────────────────────── -# ZFS health monitoring and monthly scrub. +# The monthly scrub. # -# The check script decides healthy/unhealthy and says so in its exit code, which -# systemd keeps: systemctl is-failed zfs-health-monitor.service +# The ZFS HEALTH CHECK that used to share this play is gone: it is now the +# zfs-health check in infra/400_host_monitoring.yml, which carries the same five +# conditions - pool state, device states, resilver in progress, read/write/ +# checksum errors, and errors from the last scan - but reports to Gatus like +# every other host check instead of owning its own push plumbing. # -# Reporting anywhere else is optional. Set `healthcheck_push_url` and the script -# will also GET it with ?status=up|down; leave it empty and the exit code is -# still the whole answer. Today that URL points at Uptime Kuma, which is still -# running on watchtower but is no longer deployed by Ansible - the credentials -# were retired, the service was not. If it is ever replaced, `healthcheck_push_url` -# is the only thing that needs to change here. +# The scrub itself stays here, because it is not monitoring: it is the +# maintenance that gives the health check something true to report. A pool that +# is never scrubbed has no idea whether it is healthy. # ───────────────────────────────────────────────────────────────────────────── -- name: Setup ZFS Pool Health Monitoring and Monthly Scrubs +- name: Schedule the monthly ZFS scrub hosts: hypervisor become: true + vars_files: + - ../../infra_vars.yml vars: - zfs_check_interval_seconds: 86400 # 24 hours - zfs_check_timeout_seconds: 90000 # ~25 hours (interval + buffer) - zfs_check_retries: 1 - zfs_monitoring_script_dir: /opt/zfs-monitoring - zfs_monitoring_script_path: "{{ zfs_monitoring_script_dir }}/zfs_health_monitor.sh" - zfs_log_file: "{{ zfs_monitoring_script_dir }}/zfs_health_monitor.log" - zfs_systemd_health_service_name: zfs-health-monitor zfs_systemd_scrub_service_name: zfs-monthly-scrub - # Optional. Empty is fine and is not an error - see the banner above. - healthcheck_push_url: "{{ healthcheck_push_urls.zfs_health | default('') }}" tasks: - - name: Install required packages for ZFS monitoring - package: - name: - - curl - - jq - state: present - - - name: Create monitoring script directory - file: - path: "{{ zfs_monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create ZFS health monitoring script - template: - dest: "{{ zfs_monitoring_script_path }}" - src: templates/zfs_health_monitor.sh.j2 - owner: root - group: root - mode: '0755' - - - name: Create systemd service for ZFS health monitoring - template: - dest: "/etc/systemd/system/{{ zfs_systemd_health_service_name }}.service" - src: templates/zfs-health-monitor.service.j2 - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for daily ZFS health monitoring - template: - dest: "/etc/systemd/system/{{ zfs_systemd_health_service_name }}.timer" - src: templates/zfs-health-monitor.timer.j2 - owner: root - group: root - mode: '0644' - - name: Create systemd service for ZFS monthly scrub template: - dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.service" src: templates/zfs-monthly-scrub.service.j2 + dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.service" owner: root group: root mode: '0644' - name: Create systemd timer for monthly ZFS scrub template: - dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.timer" src: templates/zfs-monthly-scrub.timer.j2 + dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.timer" owner: root group: root mode: '0644' - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start ZFS health monitoring timer - systemd: - name: "{{ zfs_systemd_health_service_name }}.timer" - enabled: yes - state: started - - - name: Enable and start ZFS monthly scrub timer + - name: Enable and start the monthly scrub timer systemd: name: "{{ zfs_systemd_scrub_service_name }}.timer" enabled: yes state: started + daemon_reload: yes - - name: Test ZFS health monitoring script - command: "{{ zfs_monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "ZFS health monitoring script failed - check pool health" - - - name: Display monitoring configuration + - name: Report the scrub schedule debug: - msg: | - ✓ ZFS Pool Health Monitoring deployed successfully! - - Pool Name: {{ zfs_pool_name }} - Push URL: {{ healthcheck_push_url | default('', true) | ternary('set', 'not set - exit code only') }} - - Health Check: - - Frequency: Every {{ zfs_check_interval_seconds }} seconds (24 hours) - - Timeout: {{ zfs_check_timeout_seconds }} seconds (~25 hours) - - Script: {{ zfs_monitoring_script_path }} - - Log: {{ zfs_log_file }} - - Service: {{ zfs_systemd_health_service_name }}.service - - Timer: {{ zfs_systemd_health_service_name }}.timer - - Monthly Scrub: - - Schedule: Last day of month at 4:00 AM - - Service: {{ zfs_systemd_scrub_service_name }}.service - - Timer: {{ zfs_systemd_scrub_service_name }}.timer - - Conditions monitored: - - Pool state (must be ONLINE) - - Device states (no DEGRADED/FAULTED/OFFLINE/UNAVAIL) - - Resilver status (alerts if resilvering) - - Read/Write/Checksum errors - - Scrub errors + msg: >- + Monthly scrub of {{ zfs_pool_name }}: + last day of each month at 04:00. + Health is reported separately by the zfs-health check + (infra/400_host_monitoring.yml). diff --git a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml index 8e188b6..12fe720 100644 --- a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml +++ b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml @@ -225,121 +225,11 @@ - nut-server - nut-monitor - -# ───────────────────────────────────────────────────────────────────────────── -# UPS heartbeat monitoring. +# The UPS heartbeat play that used to live here is gone. What it deployed - +# /opt/ups-monitoring plus a ups-heartbeat timer - is now the ups-status check +# in infra/400_host_monitoring.yml, which reports to Gatus like every other +# host check instead of carrying its own push plumbing. # -# The script decides on-mains/on-battery and says so in its exit code, which -# systemd keeps: systemctl is-failed ups-heartbeat.service -# -# Reporting anywhere else is optional. Set `healthcheck_push_url` and the script -# will also GET it with ?status=up|down; leave it empty and the exit code is -# still the whole answer. Today that URL points at Uptime Kuma, which is still -# running on watchtower but is no longer deployed by Ansible - the credentials -# were retired, the service was not. If it is ever replaced, `healthcheck_push_url` -# is the only thing that needs to change here. -# -# NOTE: this play has never actually been applied to nodito. /opt/ups-monitoring -# does not exist and there is no ups-heartbeat timer. What is on the box is a -# hand-written /usr/local/bin/ups-heartbeat.sh - mode 0644, not executable, and -# referenced by no unit and no cron entry, so nothing has ever run it. Its push -# token (uLmCPkLLO4) belongs to a monitor that DOES still exist and answer, so -# that monitor has had no heartbeat since January 2026. It is now -# healthcheck_push_urls.ups in the vault, and running this play is what will -# finally start feeding it. -# ───────────────────────────────────────────────────────────────────────────── -- name: Setup UPS Heartbeat Monitoring - hosts: hypervisor - become: true - - vars: - ups_heartbeat_interval_seconds: 60 - ups_heartbeat_timeout_seconds: 120 - ups_heartbeat_retries: 1 - ups_monitoring_script_dir: /opt/ups-monitoring - ups_monitoring_script_path: "{{ ups_monitoring_script_dir }}/ups_heartbeat.sh" - ups_log_file: "{{ ups_monitoring_script_dir }}/ups_heartbeat.log" - ups_systemd_service_name: ups-heartbeat - # Optional. Empty is fine and is not an error - see the banner above. - healthcheck_push_url: "{{ healthcheck_push_urls.ups | default('') }}" - - tasks: - - name: Install required packages for UPS monitoring - package: - name: - - curl - state: present - - - name: Create monitoring script directory - file: - path: "{{ ups_monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create UPS heartbeat monitoring script - template: - dest: "{{ ups_monitoring_script_path }}" - src: templates/ups_heartbeat.sh.j2 - owner: root - group: root - mode: '0755' - - - name: Create systemd service for UPS heartbeat - template: - dest: "/etc/systemd/system/{{ ups_systemd_service_name }}.service" - src: templates/ups-heartbeat.service.j2 - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for UPS heartbeat - template: - dest: "/etc/systemd/system/{{ ups_systemd_service_name }}.timer" - src: templates/ups-heartbeat.timer.j2 - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start UPS heartbeat timer - systemd: - name: "{{ ups_systemd_service_name }}.timer" - enabled: yes - state: started - - - name: Test UPS heartbeat script - command: "{{ ups_monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "UPS heartbeat script failed - check UPS status and communication" - - - name: Display monitoring configuration - debug: - msg: - - "UPS Monitoring configured successfully" - - "" - - "NUT Configuration:" - - " UPS Name: {{ ups_name }}" - - " UPS Description: {{ ups_desc }}" - - " Off Delay: {{ ups_offdelay }}s (time after shutdown before UPS cuts power)" - - " On Delay: {{ ups_ondelay }}s (time after mains returns before UPS restores power)" - - "" - - "Health reporting:" - - " Check interval: {{ ups_heartbeat_interval_seconds }}s" - - " Push URL: {{ healthcheck_push_url | default('', true) | ternary('set', 'not set - exit code only') }}" - - "" - - "Scripts and Services:" - - " Script: {{ ups_monitoring_script_path }}" - - " Log: {{ ups_log_file }}" - - " Service: {{ ups_systemd_service_name }}.service" - - " Timer: {{ ups_systemd_service_name }}.timer" +# This playbook is now purely NUT setup: the driver, upsd, upsmon and the +# shutdown behaviour. Monitoring whether the UPS is on mains is a separate +# concern and belongs with the other host checks. diff --git a/ansible/infra/nodito/templates/ups-heartbeat.service.j2 b/ansible/infra/nodito/templates/ups-heartbeat.service.j2 deleted file mode 100644 index 71484c1..0000000 --- a/ansible/infra/nodito/templates/ups-heartbeat.service.j2 +++ /dev/null @@ -1,14 +0,0 @@ -[Unit] -Description=UPS Heartbeat Monitor -After=network.target nut-monitor.service - -[Service] -Type=oneshot -ExecStart={{ ups_monitoring_script_path }} -User=root -Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} -StandardOutput=journal -StandardError=journal - -[Install] -WantedBy=multi-user.target diff --git a/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 b/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 deleted file mode 100644 index 925998e..0000000 --- a/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 +++ /dev/null @@ -1,11 +0,0 @@ -[Unit] -Description=Run UPS Heartbeat Monitor every {{ ups_heartbeat_interval_seconds }} seconds -Requires={{ ups_systemd_service_name }}.service - -[Timer] -OnBootSec=1min -OnUnitActiveSec={{ ups_heartbeat_interval_seconds }}sec -Persistent=true - -[Install] -WantedBy=timers.target diff --git a/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 b/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 deleted file mode 100644 index c8d5d0d..0000000 --- a/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 +++ /dev/null @@ -1,68 +0,0 @@ -#!/bin/bash - -# UPS heartbeat check - managed by Ansible (infra/nodito/34_nut_ups_setup_playbook.yml) -# -# The exit code is the answer and systemd keeps it: -# systemctl is-failed {{ ups_systemd_service_name }}.service -# Reporting anywhere else is optional and generic. - -LOG_FILE="{{ ups_log_file }}" -UPS_NAME="{{ ups_name }}" -PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" - -log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" -} - -report() { - local status="$1" - local message="$2" - - # No push URL is normal, not an error: the exit code below is still a - # complete answer for anything reading unit state. - [ -n "$PUSH_URL" ] || return 0 - - local encoded_message - encoded_message=$(printf '%s\n' "$message" | sed 's/%/%25/g; s/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') - - local response http_code - response=$(curl -s --max-time 10 --retry 2 -w "\n%{http_code}" "${PUSH_URL}?status=${status}&msg=${encoded_message}&ping=" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Reported ${status}: $message (HTTP $http_code)" - return 0 - else - log_message "ERROR: Failed to report ${status} (HTTP $http_code)" - return 1 - fi -} - -main() { - local status charge runtime load - - status=$(upsc ${UPS_NAME}@localhost ups.status 2>/dev/null) - - if [ -z "$status" ]; then - log_message "ERROR: Cannot communicate with UPS" - report "down" "cannot communicate with UPS ${UPS_NAME}" - exit 1 - fi - - charge=$(upsc ${UPS_NAME}@localhost battery.charge 2>/dev/null) - runtime=$(upsc ${UPS_NAME}@localhost battery.runtime 2>/dev/null) - load=$(upsc ${UPS_NAME}@localhost ups.load 2>/dev/null) - - if [[ "$status" == *"OL"* ]]; then - local message="UPS on mains (charge=${charge}% runtime=${runtime}s load=${load}%)" - log_message "$message" - report "up" "$message" - exit 0 - else - log_message "UPS not on mains power (status=$status)" - report "down" "UPS not on mains (status=${status} charge=${charge}%)" - exit 1 - fi -} - -main diff --git a/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 b/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 deleted file mode 100644 index 104d974..0000000 --- a/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 +++ /dev/null @@ -1,14 +0,0 @@ -[Unit] -Description=ZFS Pool Health Monitor -After=zfs.target network.target - -[Service] -Type=oneshot -ExecStart={{ zfs_monitoring_script_path }} -User=root -Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} -StandardOutput=journal -StandardError=journal - -[Install] -WantedBy=multi-user.target diff --git a/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 b/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 deleted file mode 100644 index 7878ee3..0000000 --- a/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 +++ /dev/null @@ -1,11 +0,0 @@ -[Unit] -Description=Run ZFS Pool Health Monitor daily -Requires={{ zfs_systemd_health_service_name }}.service - -[Timer] -OnBootSec=5min -OnUnitActiveSec={{ zfs_check_interval_seconds }}sec -Persistent=true - -[Install] -WantedBy=timers.target diff --git a/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 b/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 deleted file mode 100644 index 3ff4c41..0000000 --- a/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 +++ /dev/null @@ -1,181 +0,0 @@ -#!/bin/bash - -# ZFS pool health check - managed by Ansible (infra/nodito/32_zfs_pool_setup_playbook.yml) -# -# The exit code is the answer and systemd keeps it: -# systemctl is-failed {{ zfs_systemd_health_service_name }}.service -# Reporting anywhere else is optional and generic. - -LOG_FILE="{{ zfs_log_file }}" -POOL_NAME="{{ zfs_pool_name }}" -PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" -HOSTNAME=$(hostname) - -# Function to log messages -log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" -} - -# Function to check pool health using JSON output -check_pool_health() { - local pool="$1" - local issues_found=0 - - # Get pool status as JSON - local pool_json - pool_json=$(zpool status -j "$pool" 2>&1) - - if [ $? -ne 0 ]; then - log_message "ERROR: Failed to get pool status for $pool" - log_message " -> $pool_json" - return 1 - fi - - # Check 1: Pool state must be ONLINE - local pool_state - pool_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].state') - - if [ "$pool_state" != "ONLINE" ]; then - log_message "ISSUE: Pool state is $pool_state (expected ONLINE)" - issues_found=1 - else - log_message "OK: Pool state is ONLINE" - fi - - # Check 2: Check all vdevs and devices for non-ONLINE states - local bad_states - bad_states=$(echo "$pool_json" | jq -r --arg pool "$pool" ' - .pools[$pool].vdevs[] | - .. | objects | - select(.state? and .state != "ONLINE") | - "\(.name // "unknown"): \(.state)" - ' 2>/dev/null) - - if [ -n "$bad_states" ]; then - log_message "ISSUE: Found devices not in ONLINE state:" - echo "$bad_states" | while read -r line; do - log_message " -> $line" - done - issues_found=1 - else - log_message "OK: All devices are ONLINE" - fi - - # Check 3: Check for resilvering in progress - local scan_function scan_state - scan_function=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.function // "NONE"') - scan_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.state // "NONE"') - - if [ "$scan_function" = "RESILVER" ] && [ "$scan_state" = "SCANNING" ]; then - local resilver_progress - resilver_progress=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.issued // "unknown"') - log_message "ISSUE: Pool is currently resilvering (disk reconstruction in progress) - ${resilver_progress} processed" - issues_found=1 - fi - - # Check 4: Check for read/write/checksum errors on all devices - # Note: ZFS JSON output has error counts as strings, so convert to numbers for comparison - local devices_with_errors - devices_with_errors=$(echo "$pool_json" | jq -r --arg pool "$pool" ' - .pools[$pool].vdevs[] | - .. | objects | - select(.name? and ((.read_errors // "0" | tonumber) > 0 or (.write_errors // "0" | tonumber) > 0 or (.checksum_errors // "0" | tonumber) > 0)) | - "\(.name): read=\(.read_errors // 0) write=\(.write_errors // 0) cksum=\(.checksum_errors // 0)" - ' 2>/dev/null) - - if [ -n "$devices_with_errors" ]; then - log_message "ISSUE: Found devices with I/O errors:" - echo "$devices_with_errors" | while read -r line; do - log_message " -> $line" - done - issues_found=1 - else - log_message "OK: No read/write/checksum errors detected" - fi - - # Check 5: Check for scan errors (from last scrub/resilver) - local scan_errors - scan_errors=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.errors // "0"') - - if [ "$scan_errors" != "0" ] && [ "$scan_errors" != "null" ] && [ -n "$scan_errors" ]; then - log_message "ISSUE: Last scan reported $scan_errors errors" - issues_found=1 - else - log_message "OK: No scan errors" - fi - - return $issues_found -} - -# Function to get last scrub info for status message -get_scrub_info() { - local pool="$1" - local pool_json - pool_json=$(zpool status -j "$pool" 2>/dev/null) - - local scan_func scan_state scan_start - scan_func=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.function // "NONE"') - scan_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.state // "NONE"') - scan_start=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.start_time // ""') - - if [ "$scan_func" = "SCRUB" ] && [ "$scan_state" = "SCANNING" ]; then - echo "scrub in progress (started $scan_start)" - elif [ "$scan_func" = "SCRUB" ] && [ -n "$scan_start" ]; then - echo "last scrub: $scan_start" - else - echo "no scrub history" - fi -} - -# Optional reporting to whatever is watching. No push URL is normal, not an -# error: the script's exit code is still a complete answer for anything reading -# unit state. -report() { - local status="$1" - local message="$2" - - [ -n "$PUSH_URL" ] || return 0 - - log_message "Reporting ${status}: $message" - - # URL encode the message - local encoded_message - encoded_message=$(printf '%s\n' "$message" | sed 's/%/%25/g; s/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') - - local response http_code - response=$(curl -s --max-time 10 --retry 2 -w "\n%{http_code}" "${PUSH_URL}?status=${status}&msg=${encoded_message}&ping=" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Report sent successfully (HTTP $http_code)" - return 0 - else - log_message "ERROR: Failed to report (HTTP $http_code)" - return 1 - fi -} - -# Main health check logic -main() { - log_message "==========================================" - log_message "Starting ZFS health check for pool: $POOL_NAME on $HOSTNAME" - - # Run all health checks - if check_pool_health "$POOL_NAME"; then - local scrub_info - scrub_info=$(get_scrub_info "$POOL_NAME") - - local message="Pool $POOL_NAME healthy ($scrub_info)" - report "up" "$message" - - log_message "Health check completed: ALL OK" - exit 0 - else - log_message "Health check completed: ISSUES DETECTED" - report "down" "Pool $POOL_NAME unhealthy - see $LOG_FILE" - exit 1 - fi -} - -# Run main function -main diff --git a/ansible/roles/healthcheck/defaults/main.yml b/ansible/roles/healthcheck/defaults/main.yml index 526347a..51534c2 100644 --- a/ansible/roles/healthcheck/defaults/main.yml +++ b/ansible/roles/healthcheck/defaults/main.yml @@ -46,4 +46,9 @@ healthcheck_log_dir: /var/log/healthchecks healthcheck_disk_threshold: 85 # percent healthcheck_cpu_temp_threshold: 80 # celsius healthcheck_zfs_pool: "" +# ZFS degrades badly once a pool passes roughly 80% - allocation gets slow and +# fragmentation becomes hard to undo, and unlike a normal filesystem you cannot +# simply delete your way back to good performance. So this alarms well before +# the pool is actually out of space. +healthcheck_zfs_capacity_threshold: 80 healthcheck_ups_name: "" diff --git a/ansible/roles/healthcheck/tasks/main.yml b/ansible/roles/healthcheck/tasks/main.yml index fb0257b..f37a0d5 100644 --- a/ansible/roles/healthcheck/tasks/main.yml +++ b/ansible/roles/healthcheck/tasks/main.yml @@ -11,10 +11,26 @@ or healthcheck_command, and a token whenever a push URL is set. A push URL with no token would report to Gatus and be rejected 401 on every run. +# Deduplicated across the whole play run. This role is included once PER CHECK, +# and a host with several checks was otherwise running apt several times to +# install a curl that was already there - 29 apt transactions estate-wide, and +# the slowest thing in the deploy by a wide margin. The fact below remembers +# what has already been ensured on this host. - name: Install healthcheck dependencies ansible.builtin.package: - name: "{{ healthcheck_packages | default(['curl']) }}" + name: "{{ healthcheck_wanted_packages }}" state: present + vars: + healthcheck_wanted_packages: >- + {{ (healthcheck_packages | default(['curl'])) + | difference(healthcheck_installed_packages | default([])) }} + when: healthcheck_wanted_packages | length > 0 + +- name: Remember which dependencies this host already has + ansible.builtin.set_fact: + healthcheck_installed_packages: >- + {{ (healthcheck_installed_packages | default([])) + | union(healthcheck_packages | default(['curl'])) }} - name: Create the healthcheck log directory ansible.builtin.file: @@ -49,10 +65,6 @@ group: root mode: "0644" -- name: Reload systemd - ansible.builtin.systemd: - daemon_reload: yes - # `restarted`, not `started`: started is a no-op on an already-active timer, so # a changed interval or a stuck timer would never be picked up. - name: "Enable and start the {{ healthcheck_name }} timer" diff --git a/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 b/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 index b6c7dae..ebe88b9 100644 --- a/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 +++ b/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 @@ -44,9 +44,20 @@ issues="${issues}${issues:+; }scan errors ${scan_err}" fi + # 6. Capacity. Not an error condition in `zpool status` - a 95% full pool is + # reported perfectly ONLINE - so it has to be read separately, and it is + # the failure you get warning of rather than the one you discover. + local capacity + capacity=$(zpool list -H -o capacity "$pool" 2>/dev/null | tr -dc '0-9') + if [ -z "$capacity" ]; then + issues="${issues}${issues:+; }cannot read capacity" + elif [ "$capacity" -ge {{ healthcheck_zfs_capacity_threshold }} ]; then + issues="${issues}${issues:+; }pool ${capacity}% full (>={{ healthcheck_zfs_capacity_threshold }}%)" + fi + if [ -n "$issues" ]; then MESSAGE="$issues"; return 1; fi local scrub scrub=$(echo "$json" | jq -r --arg p "$pool" '.pools[$p].scan_stats.start_time // "never"') - MESSAGE="${pool} ONLINE, last scrub ${scrub}" + MESSAGE="${pool} ONLINE, ${capacity}% full, last scrub ${scrub}" return 0 diff --git a/ansible/site.yml b/ansible/site.yml index fd042ce..a5f8576 100644 --- a/ansible/site.yml +++ b/ansible/site.yml @@ -18,6 +18,17 @@ - import_playbook: infra/02_firewall_and_fail2ban_playbook.yml - import_playbook: infra/900_install_rsync.yml - import_playbook: infra/920_join_headscale_mesh.yml +# Idempotent and kept permanently: guarantees a rebuilt or restored host cannot +# quietly bring the Uptime-Kuma-era monitoring back. +- import_playbook: infra/409_remove_legacy_monitoring.yml + +# ── Monitoring ────────────────────────────────────────────────────────────── +# Gatus first: the three plays below register endpoints with it, and registering +# against a host that is not serving yet would simply fail. +- import_playbook: services/gatus/deploy_gatus_playbook.yml +- import_playbook: infra/400_host_monitoring.yml +- import_playbook: infra/401_service_monitoring.yml +- import_playbook: infra/402_public_monitoring.yml # 910_docker says `hosts: managed`, but only 5 of 11 managed hosts have or need # Docker. Left out until it has a [docker] group — see the note in PLAN_7. @@ -56,13 +67,6 @@ # Deliberately not here. Every playbook in the repo is either imported above or # listed below, so this file accounts for all of them: # -# infra/410_disk_usage_alerts.yml assert on the Uptime Kuma credentials that -# infra/420_system_healthcheck.yml were removed from the vault, so they fail -# infra/430_cpu_temp_alerts.yml before doing anything. What they deploy IS -# running on the boxes - same state the two -# nodito playbooks were in before 0f03c50. -# Add them once they are de-Kuma'd. -# # infra/910_docker_playbook.yml says `hosts: managed`, but Docker is on 5 # of 11 managed hosts and those 5 are exactly # the ones that need it. Running it would @@ -74,7 +78,7 @@ # # services/ntfy/setup_ntfy_uptime_ creates a notification channel INSIDE # kuma_notification.yml Uptime Kuma. Kuma-specific tooling, not -# a deployment. +# a deployment, and Kuma is being retired. # # services/vaultwarden/disable_ deliberate manual actions, not convergence # vaultwarden_sign_ups_playbook.yml From bf3d21fef773b8b99698328122bbfdef7c4bd4a6 Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 10:30:28 +0200 Subject: [PATCH 64/67] uptime kuma: remove every live reference, repoint the probes to Gatus MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Nothing in the repo pushes to, authenticates against, or is gated by Uptime Kuma any more. ── The sixth instance of the banner bug ──────────────────────────────────── memos had `Restart memos` guarded by `uptime_kuma_enabled`, because the deprecation banner was placed immediately above it and swept it in. It is a HANDLER, so every memos config change since 2026-09-11 applied to disk and silently never restarted the service. Ungated. That is the same failure found in forgejo-runner's self-assert, phoenixd's timer enable, mempool's three timer enables, fulcrum's restart handler and bitcoind's restart handler. Every guard was read and asked "monitoring or deployment?" before being deleted, which is the only reason this was caught. ── What was removed ──────────────────────────────────────────────────────── 30 uptime_kuma_enabled guards across 7 unconverted service playbooks, and the 29 Kuma monitor-creation tasks they gated (embedded Python that drove the Kuma API, temp credential files, cleanup) 7 dead uptime_kuma_api_url definitions 7 stale DEPRECATED banners uptime_kuma_enabled and subdomains.uptime_kuma from group_vars/all healthcheck_push_urls from the vault - 30 push tokens services/ntfy/setup_ntfy_uptime_kuma_notification.yml -> archive/ The explanatory comments in the six converted roles are KEPT on purpose. They record why a handler is ungated, and deleting the explanation invites someone to helpfully re-add the guard. ── The probes moved rather than died ─────────────────────────────────────── Eight per-service health checks were still pushing to Kuma. They are not superseded by infra/401: that answers "is the unit running", these answer "does the service actually respond" - an RPC call to bitcoind, a TCP connect to Fulcrum's Electrum port, an HTTP fetch from Mempool's backend. A process can be perfectly `active` and useless. So they were repointed, not deleted. Gatus external endpoints take a POST with a bearer token and success=true|false where Kuma took a GET with ?status=up, so report() now maps up/down to true/false internally and no call site changed. Registered by infra/403 as the `probe` group, one token per host. Two bugs fixed while in there: * forgejo-runner's check only ever reported SUCCESS - it exited before pushing when the runner was down, so a failure was invisible until the heartbeat window expired. Reporting the failure is the entire point of a check. * All six healthcheck .service units were mode 0644 and now carry a bearer token. They are 0600. Verified: 91 endpoints, 91 UP, 0 DOWN. Every probe triggered by hand and confirmed arriving. Zero Kuma URLs left in the vault, zero live references in any playbook or role. Still standing, deliberately: the Kuma container on watchtower, its Caddy vhost, and the uptime.contrapeso.xyz DNS record. Turning the service off is a separate decision from removing the code that talked to it. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/main.yml | 5 - ansible/group_vars/all/vault.yml | 409 +++++++----------- .../infra/403_service_probe_registration.yml | 47 ++ ansible/roles/bitcoin_knots/defaults/main.yml | 3 + .../roles/bitcoin_knots/tasks/healthcheck.yml | 2 +- .../templates/healthcheck.service.j2 | 1 + .../bitcoin_knots/templates/healthcheck.sh.j2 | 11 +- ansible/roles/datum_gateway/defaults/main.yml | 3 + .../roles/datum_gateway/tasks/healthcheck.yml | 2 +- .../templates/healthcheck.service.j2 | 1 + .../datum_gateway/templates/healthcheck.sh.j2 | 11 +- .../roles/forgejo_runner/defaults/main.yml | 3 + .../forgejo_runner/tasks/healthcheck.yml | 2 +- .../templates/healthcheck.service.j2 | 2 + .../templates/healthcheck.sh.j2 | 38 +- ansible/roles/fulcrum/defaults/main.yml | 3 + ansible/roles/fulcrum/tasks/healthcheck.yml | 2 +- .../fulcrum/templates/healthcheck.service.j2 | 1 + .../roles/fulcrum/templates/healthcheck.sh.j2 | 11 +- ansible/roles/mempool/defaults/main.yml | 2 + ansible/roles/mempool/tasks/healthcheck.yml | 2 +- .../templates/healthcheck-backend.sh.j2 | 11 +- .../templates/healthcheck-frontend.sh.j2 | 11 +- .../templates/healthcheck-mariadb.sh.j2 | 11 +- .../mempool/templates/healthcheck.service.j2 | 1 + ansible/roles/phoenixd/defaults/main.yml | 3 + ansible/roles/phoenixd/tasks/healthcheck.yml | 2 +- .../phoenixd/templates/healthcheck.service.j2 | 1 + .../phoenixd/templates/healthcheck.sh.j2 | 11 +- .../deploy_bitcoin_knots_playbook.yml | 3 +- .../deploy_datum_gateway_playbook.yml | 3 +- .../deploy_forgejo_runner_playbook.yml | 3 +- .../forgejo/deploy_forgejo_playbook.yml | 121 ------ .../fulcrum/deploy_fulcrum_playbook.yml | 3 +- .../headscale/deploy_headscale_playbook.yml | 120 ----- .../lnbits/deploy_lnbits_playbook.yml | 121 ------ .../services/memos/deploy_memos_playbook.yml | 134 ------ .../mempool/deploy_mempool_playbook.yml | 9 +- .../deploy_ntfy_emergency_app_playbook.yml | 125 ------ .../deploy_personal_blog_playbook.yml | 121 ------ .../phoenixd/deploy_phoenixd_playbook.yml | 3 +- .../deploy_vaultwarden_playbook.yml | 126 ------ ansible/site.yml | 4 + .../setup_ntfy_uptime_kuma_notification.yml | 0 44 files changed, 337 insertions(+), 1171 deletions(-) create mode 100644 ansible/infra/403_service_probe_registration.yml rename {ansible/services/ntfy => archive/uptime_kuma}/setup_ntfy_uptime_kuma_notification.yml (100%) diff --git a/ansible/group_vars/all/main.yml b/ansible/group_vars/all/main.yml index 0ff8ef8..316b178 100644 --- a/ansible/group_vars/all/main.yml +++ b/ansible/group_vars/all/main.yml @@ -6,7 +6,6 @@ root_domain: contrapeso.xyz # Uptime Kuma was decommissioned on 2026-09-11. The monitoring blocks in the # playbooks are kept deliberately — the check logic is meant to be rewired to # whatever replaces it. This flag keeps them inert until then. See archive/uptime_kuma/. -uptime_kuma_enabled: false # age recipient for all backup artefacts age_backup_recipient: "age192wwdaseqej2ggwyp884gtm05c396anp7chr0vr8m47g50fahpyqr9fsza" @@ -28,10 +27,6 @@ subdomains: # Monitoring gatus: status ntfy: ntfy - # Uptime Kuma IS still running and this subdomain DOES resolve - # (164.92.239.72, HTTP 302). Only the Ansible code and the vault credentials - # were retired. A comment here previously claimed the opposite. - uptime_kuma: uptime # VPN infrastructure (spacey) headscale: headscale diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index d68ae8b..eafc633 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,254 +1,157 @@ $ANSIBLE_VAULT;1.1;AES256 -63323431376238353966386463626539656230326233323861656165386335383832316631353236 -3731396264313366313166653861313736666435346537630a303363346131366433313264626135 -30633836613636393833333239666364623962383763343434353463343739633033383433306664 -6339616335366562310a393463663239333530633034373462313537376266393937373033346233 -62333331613063356535353134613538636663383166353731616633336366653864656333356330 -37323139616436353065363866633139323764336432656663326236636466356432333132656530 -39396566656161373365653738316162653134393234613130356561663464623564316161383566 -32356161646631623134313733343030353064666635346134653361366635316362316564323562 -38623236383137383665623934336631636537343361656539393538346665613234333638346466 -61653434653430636534636137613236326330356234616137303130666363323861623230626662 -37363839333334373561323863353563303861636137613338303736343064663232663064303466 -62396466363966303339643938653930313163323561633835396562636363633633646163363461 -61636331316161386630393766326431316239326563633033376538323330303863646162303332 -62306439666232626533306238343338343938626461363939326363333137306430643033373363 -65393164626166366663656637373330633939326361653261336339393135363934376164303537 -33326631396161393936656563336636643132666232343035633465383632613661633135343165 -33656162656334353636303130613231393835626166633666316561343165333138376439356539 -35356461353832643166613437343633346563393636323631353034366233323566613039343030 -65356438656664366434343638643963396563663434623961663432653334646639653435343262 -31366465366337633638393061643139643138316536396333653035613132623230646561373465 -65366332386565623161303563386237666538633433386438623535386564633937386434343264 -39653335343235386439383964356664313339356531323732363362643566363964653634393039 -37643338316264633733383531633934373132643034653433316438303962356139376364306466 -37613232636365623732313233303766373566623162303965353163373131363763346135313230 -39316466613136323039373765613336333835323465323737323535393736366433343664616539 -32303839376538393238613763366433346433643436613662383062306366643561626164363430 -38353738653734393237663765613733633230343965363732643336333337646438623562303561 -34373031623466343539663738613561346665316163623631643236633236633765616361623764 -35343465346435656533336464393633616663343239343162343664363763373736656431663433 -62306565643862613065386265613036326336353130633530373166343966343033346665346235 -37333761613437393236393935646239373930613239323639616564383336376664386532366236 -31313862363232656537613733666239373433343835633964333164633335656436373766353838 -36303234353333333233313531396464336262663864633638306236343633336432333737376539 -36623439336133356362626632663966336162646636393932353537356638336337346663303230 -66623964656133333534313231656337336463346363363635396135393062643736336133393134 -66376233376161373561386531633639376137643732613535333066646666616133623963323137 -63356565383430656535633639643363653232323737363566663832633931616230343132663563 -37633435626266393937373831366632643535623634386131343065353036653163326464666364 -39613732653739643436303665363132336235333339653630353335646432306432643235376139 -33643935366365383232646234313436383534353130633039376562363939363033643830303936 -64376131346631306263363137316435343661393562386638363636336261623831616232313934 -30373133633034626430313936373537323366626562323239623538306130623135333562333333 -33303761623232666165323438643364313530316563396466653331636531366166633964353033 -61633861363833613264353036383766343233646439393764636664353938656335623330323062 -64306565626630326632383963353266383464393530623639333836663739306132346336373936 -36636532623032666433303339643062383663646536636663383662646237323336386366386138 -31643930356330343138323132613238333837356137643061346364653134346165383661366663 -31643630353631656131633430323838383233323936623661613466663761313264383338373636 -63323062646334666665633662366364383434393561313863653862633533356166396336303133 -61383637326535656539656538353238326337353138616533313534643131346163356533396331 -30303031656133623735366163323664626362356365663730306438653132396461393539386138 -65343865633765306337663830343931376265666362653361616662666665386138313462633432 -32623436643239363437373161353162626663623332373163656133613465333139356564383130 -36653433653535326433313337636635336132663636383764653237356432313362393535306630 -32663738663835336131353737303262313966663264303264313864353663323733653334313263 -32333866613335666436303236393536633835653837363133666437303736303630633939616232 -65316337646665343062353864623839613263356335353938386136666166363565303062633331 -39633761316630656662636136393834626564616139396336663363373931666132383062626536 -61376462333131633634383130393765353037653536653837373032663636623031316530613961 -63633861613366373235653735616333303262333765636531353734633664383766623261636238 -32623132336137666366643035306562643833333537633666343637366230623366666230366638 -37333238336435636235333063636538666635666438383239336164613536666262343132646634 -38633132343966663564326138346561313731623966356435653937383431396261313138656261 -65326265323333366538656335306431396666643663313561663964626436356464356336323661 -66663763396166313831623534353131633566346634613738316634653935373038303730336662 -34656634343834623930623833386632306136323738646535396539643463393861343832333065 -35333464316163626130366233663439633435333462353534616530316464666537613436386637 -37363939333834626638396337636234343561313565613636643463393132626466333632636463 -31326261656537616130613634323736313132633361653162666631303965613036653434653236 -33313037313034356139353261666138386636663637353831666330616563323536303834623863 -30346539376536396337313561396561363237613865623533633063613936316162353138306164 -64646634396236646535633135613036376538343964363663616234666532386165643938396436 -36333832313136363965383334323430313630663336343562626365633461366133653931376139 -38613364383236393634386436643733333433336563306138356337636337623239646164306539 -61336433383339353861313131343166353265393132356631366230343438646239323730613865 -35333438326236353264386436373338303532646336333161636232393235653830303237323137 -36323464303634663839326230613539376238323931663137616363663434373662636535373266 -38363261653937633363316665613130386135626135333662356261323462303939383962363462 -37663333386336626261663464393561666135633537616365376665393664346466633932656635 -65343762656162613831643265633562373865616662313631313034363964626538383633656363 -61363639343663643638353935643635643663636139313963303739343830613336343661663733 -32323463626264663639396130343930383162373133386232613635633934393334626631303561 -64346636636330623538343534663037363635323138343566303664613337346638353637306336 -36386432326661316633303764323761303363646337303539633831666665373833666663333766 -62363461616462383663316434373561653063323062356131643930633838336364643065613838 -66386565373635396138623832646137636236646335386463326265326565353566346136306564 -36323666333331396636353762363866616239343362313431313765373334386637633366623936 -31333432663331623232373330306430626264333761373362303365656262303164656439623637 -36663065376137336435646232323262646439623266333534623766353035376663353535333137 -38373335313833633239313963613439316433643832653938386434323666333263373437663337 -32613661666664366636616562396232366237333237316430346565653066623561636263343265 -34313137643933366239353836383265633030373636393232663534393036343130633438643331 -36643236636232323532363036646563646436613630323638356534373831373737386139393664 -62616133323861333265626665616331333665356666643734666537356565393334373036663330 -66346361363361303538663936616664663864316530303338316634616636353235353161366135 -33363630626566336363336630656331343437393666353262396137386163633134343238623361 -61306163333831626164326461616630636266633666383765356539656536363133393537333263 -66646635343437643762373432333164626334333730373762336565636533333132363965636334 -64316232623437353263656131633335396166616133653234623432386234666531356431373333 -63316462643464626362646463323935303765643061343535393065333961303931663564353735 -33336565383332303637383538316237343430626432393533376166323565303263643435313136 -30626165623765346430373165323030386635376338666235306534643730656564653531633765 -36663864396638666165303837646236386434326166663365356164353036646464363132356261 -66373861343834643161343665616139373466346130353233623135656663373230333630306132 -66653233623632303463343261653764636230333038623936353138356565623061643432346265 -36326263613833386237333566333534636238336539303638643233376331616331626636666635 -66303535383838336135306439306239323531343331343832376636363931626663333337616464 -61656532623437363634343039313534356565383361303562336666383561333235303939303339 -65326336633631313133396263326139636535346166373333343934316435363435353236363331 -30396231383336326461326462393633353739346463653636393331313534613466363738376263 -62663264643833333032343839373165353262663637376635373430333631316532336335653038 -32303835333461363635306664653331613164316264613632326131633639666263633234306336 -30656562383439333639623534303164323231626337396362373235383530323731626139333335 -38386238396565643533303030393364373735376564373765373632386335613432313735336132 -63306231373038366131353934393735323932633233646439383666366333653535373634386333 -31643130376266626236353937633235623765323462396130383831376366313337643939363366 -62326439623135393539366133386337353964306637343238633730363639373831633662653565 -64333134336666643465633565333765613835373765663664653331353935666437633566386165 -61656239653264646530306165396234626365326238616566303831353935626362366338613339 -30303364356137323935626662636663383761383935343531666461656537643333346637666365 -37663836386466303433313339663531373432643732333461613739636233336566356139323934 -36663965626330333436373764613730393365616166653866306336633939393765633564626331 -31353131663038323235396564623234396138386237663030316530353337373934323232633433 -39363931656466313639346265363466646631363032353363383662306436346162363431353833 -36303934323364363465343236633064336239643565393034353934373239376139386535373061 -34643365393866636430356338626438386136646161376538363762336265653632643334336331 -38646131386463666135376162613864636534366264336630356135316637393135646630333733 -64343731633836306565623238313936656164653038393236356130306162346331396231393436 -38303435383963626434323230643832303838646239646263353737323166613965623734313933 -38313164313062373263326138656239366135383361616436333539356331383064363562376366 -31333838313664653432373937306239363631626234336136396364396166656234623365616237 -36363434393537396639633062626639353738386232393066333034343132303831366362623031 -36643862376637643739626662656239633361653933313130646661656332316535396230386162 -39373738366138373130643636663339613732626532383465316365363638386361623838656630 -66636262666536303739616361343763366135313835353938323330363635343135633138306361 -61346433336366643430646334336136346136646166363962613336366239653236373135376138 -61316464646264316532333839626634623165336334323836643130323137303632353232616631 -32653036313133333237363233653932366334623133646565343461373132306365313331373335 -33623561623163383562366137663733343833333136353738386535313439386164346164383565 -32393331373130633065323266623465613732353431623234653435383133373537363436346463 -63323666666136613034643237623463383462326334343334303731363961303438646330646562 -64336132663561353565376361306331363138306638353834613231323331356238343730616164 -35666432626436376539373633383466343732616430333661613865653530366439626131303233 -63353633613232636238316532633231633539633363623734316337633764333430366466626430 -34336534616537656531613538356231343061326462333663653662343762666131393765306462 -34353435626337656362663637613763623537623534666336643833643963626265303266623434 -32323836663432313438363530663839653637346562626566316163373137363063633564323731 -64336638616661366437656431396435303439373132383330646635306435646138633531336461 -39663366643961303832323838643530653632383230373366396531646639336138373434396431 -32383064313533393735336235356432646162353236343365313435313666363935333835336331 -39316330306336396637613634306531346531393036326536653961313438623664643362623937 -36363535383562663535343332363064326530636335343963366166333365613234643035343764 -62366334623332653931393836303934333139636635363638303734663230363537663338383461 -34366661313931316639303762373131636530363232386263313361666561313034343033626333 -63386630396136316338653134306632323664396466343361306636623164616562393430353732 -32646537653963383639616530646561653065386631386138616130343936333935393861643161 -36613166633561646138336539373731373939633234633861636634396563646137316134636463 -30363133633131303632316634376361623637373133393234393362386338626536653438666539 -31633837353935346438316133333136326237303430313533643265363966396537373161346236 -33656163636338363565653834636461333837653433633639393361636565613337653766353765 -36613834633631346136366232663539636566666339343939383732306537396465373932646434 -31663736323535353763623633366535333262636234316363366537393531666430633631373364 -36393766343863353135333864656536393739636563363631306362336665616636303832666330 -66623336663566316330333337343064626366323563613463646338613433623637363064656639 -63626465613963333932346131656639653239353034613862323565666436316338666563623836 -37356465633635363834316564313839366137656232346637663231373261393035623035643065 -30316663613732616237376265363561326164313631366466653139323534623531323537623164 -35383365336438356538343934356134646337626635326266336533386262323262323734663534 -35313239393232393435346531643138376362336437323963613933623739343463613831326539 -61643939393461386534656664366361363062643939393962376664353563643635373336333266 -39383132363964656438623031366336306362353736633634353033656539376233343131653030 -62653736663637653166373930313664333930346262396339386432353066646465373937393964 -35656538373064643936323731393436386665353838313563613832323834316539366163373638 -30623365333266393331396135613663393139303136663636383766613731636435386635666132 -30633664303830353137393438303666656235616332393132613364333238646566303363363365 -61636132663366666139363965663239623764366334333432633830316636303263373331313931 -30323337383931353363343231656632663534323435353338646664393635323733623166343830 -39633865336666363635346462306261613935383666396261653531396164383961383830333434 -39303063383365306435303430633834386566316332313864373664383766646463393734343961 -34656339343761613863363831393933396438636332333339393433636435316330393634353032 -66303139333933393164363735303534346332336564333366343032353631396633366337333831 -63616637313262316335386230343034323038613530663432353764656337343638383135316361 -32376339333235316335656638613434306633316631376230383434303234666532303662333630 -32626562656363633837316335643162363232623265396362653264313934336566366130376634 -63633132626539353138373263313135633265623761393063383136376130646132643664343431 -32616136306335336636343434366236376536663730643638636234623136383262643766613137 -34666431333265633063356566623139666266643138653365613961613532343337346161643337 -65316332393762386133633030366534353763323738303437636537326137306365626632383630 -34363566643436376535386134383638313237643934303931653839616637643036373134363830 -66656463303331626430663063356239373434386361323139343838333763343565396637356561 -63386436353934646336616537323233336339326261653466653135653735316463653666343231 -36343261616338303864343162646338323339633634663032306433313138376539643236393462 -35333734306662393836386231306466616133383762353464343965376132643634393237396164 -34373831316634626133353735376661633464373238383561316565616331666431356165623039 -34386130363761663365666238323534393933313866333030643132346563646463393063333764 -64646432616231343032633965643337363234393435346430303931633665623362306465383962 -39383437623761613339353733643862376462623166636264303833666437393231656437663764 -65313065323134623265333466373461306639326465326362646532643037333230373837653230 -62383033363561343735623639626565333666636563306639643139666134383762353261343864 -34386334306237363762313465333163643037626534613830663463653563663537396231356431 -36646139333534643765653732303330343532373933656537633465326632663539373638323865 -33386363333065373565336566613333313738313437613764663664633032393865313331633433 -65336664386437353034366638643533366365393864363033366363626630336132313038653430 -30396539313865646333666462643061666566313033613634326537363132663462623861666137 -66386663633634396163666164316631393635623938333137653361363963303235363732623930 -30376632376264343036643461613438396531373865613162643734303963373539363464343337 -39653066626635626164613435643738616534623064306338303734653830633737386531393435 -61303461666666363230643638626230353839653962663439353166346635376438623832663637 -35663738356136363165653133323666383935363566616135376161356165316239373661643039 -35353666326336623561356635343232393137383536356565353762616639383932313464383063 -35393734306636313561383133393961646132633639363136326332366338633535666139643634 -39303735653564656232326339313238383137333630383539623139356539323561353366323131 -37393130356639623133653134623131633031393633386330653034353166376466633739303638 -66613534393331383438326230313936653532336134643632306634323530343630363236646235 -38656463343932333637613866386232376162623939626466303063306466623132643731623338 -63656662376132653238343537343265373236306134376565323961323264383830363065323738 -33366137323535633732316563386237646230613339666337386162633166353533346565393337 -34316233323734643262663532373964346331393031356134373266616265373564386236616237 -38373730623261346365353363383833633964383132623361343333373637386339313262396633 -37666437613361646337323039343938353466346538356138376130313337633533666137626266 -66313838323638626230306138623135363266346366316339313164626233656237376266353762 -65356461346239613066613038353735626233383939666330393131363064363337356330363435 -39656633616166356430343564366433323864333236623934356335336661346338653834363362 -39343435653934643132363931363334366631646463306261666537303938363633336464666537 -38336632643561633461383965333264376462306131666232623735313265373832343762393434 -61386534393034353363613230626330393234326234363837393738376634633561613562323137 -66613930306330623434323539366333663364396165663466303365653331363431656365353266 -63623430353734616439393735623564626638313336613636383438363531306234663939383066 -37623165333233643465363334663433663034636438613433633966306334376462316233343338 -61333530376236333134306164323263633266376666663030303438646335393661316137646237 -32383865636433366635323134663938383339393933656438633662333334313264643338636563 -64343533643432386164333630366531383434333231393762613136616435376534653530346364 -64396531383363376437613366613066633534616163396133323835323431353034373563306536 -35616335636463623565636534346435643463376330333962353261663062613034653863373834 -37653566383164373263616265643536343037346464633930303935393337336333616338633730 -35316166656164356535613364386366373666306131373063376465663935303530666432383435 -65393934396639333765313933643263306337623635623930656430343361653861653039323861 -62326165313038313137323539343934366134383630363632653939633331626566653663613666 -30336161616136613034353133663738646464306164663931373365343664373337303564643565 -31623132396431336130396236656136656335336236306364656164353431633136343732663631 -37316366633837323961643338343538653933306664356236373165333464643032623864633538 -32613030313563343930653863346638393662303030396537343264326234643735323532376337 -61313936623663323961333364306664613331626233393430626632373765333832616136333065 -65323837373231633439333437616536306466656530646332386164373963386431653532666262 -62386234393432633430636439663331386235366630333630383336363664663333383331386362 -31376362643263396634623662373134343039323663343433383836663261376463306337656236 -32633536336238326337336666313263613433353739333530316630653735636133303635313436 -66383761613932653132663939353734623663333736666462363235333336333733323963623663 -33616265636664373638656363636363656639373634353732366664363565383737313863666139 -35303364343063663764393261333336373864373839666664356166626238363035343163653131 -63316436373331393461626164346362366530636335613966353335376334643433333963396563 -39373364356439663363333566656565616130643037613332363937313964363433613436666230 -6232 +30373030336530356533636231303630303132666434393339663833366639366234313434333635 +3763366235343633303836383563333734643766346336390a383732313137623263343836616531 +35656262386162346434343462396166653435633330613366633263356638373061643833386336 +6438373137326563380a353962396135393365376133626538303862396236363330636564323538 +33616565363066633234633330386338316133613931613931653963356562323336396437313130 +33316331333166393161376635306237363033633535613761373762393761363165623462336466 +63663265376532663262323463643462336136373639316565396436653966336461656164316135 +64303235386236316136623063653133363961303733376166363462303565393761636632666230 +66333263626537343935383765303565626237346235356338373063633133326465653133366362 +62326563666531343166663563393932393938363663393132616561336363643037363735393233 +38336233346138356262616533643835396230663563623237616461626265666339613161616331 +34353263363065646336663538663561373536656364623061633863643137633230643931333662 +34386433326661336362636466386566636561623339383438626530333431313039623037616430 +62333832363531323363646238363733316237646133343337363434373465353463363935323233 +37326335343333636337626337643763323336623238636463623933646663356165363062383835 +30636630303534636436326530313030626565393439643238353433396638363133353865643135 +39666464323537383435636636303930396338333332363264306664383538666236616436353939 +61316564623964613839343533316438393337616363633033636231313937366139373031616236 +61373763303231376437313332343438376132633361353661336232333966313338373262383663 +36363931376132333763363332313534616133326437303637343139383861353138623465623736 +62643062396536633730663130353730356637353533616635343166643636616163343832363866 +65646266383062373635306239323565383039646334346535653962316434393365646461653366 +37396163396362313535393138663833653435353934623432616538663565353165363732316662 +35316332363030363963666136653437316639303766376431333738343061663835306339336337 +64346137313763656531306636333030323162646533633735646435613566616532363066353361 +32326439633432656633303732626361636238323735636631633732383635656436373435343931 +64356437326633326163306430306432326134373366616364656361383133656531653333633164 +37653934643061343237376664643331623633653437336534313261643765303134626536393938 +33393161363539313130376132313864666539363865393035626463313539376635393135633565 +35306435636363613735343136663939663435653135376337643034396634623135393039366265 +36326332386564303233623863363236356535323538616237303262613261363964643839363837 +63336536613139333939343964323835626433373464343730353864303532366535343636663362 +34333935316630323137366130356135643962373532346334333931366139356434316431393234 +61386632353434323432306130386239663931306538323335333231366234383132623936343465 +32393765316136663862663334616234326165336364316137396234343532643639366537343731 +33383938363530323337333838643835653238376532376431393439636266306634386233383937 +64623230353533643362633563663537343732646261353763366363343231376562336565333362 +65623665326136383764373761626461653238303335616463636333383235393037323530636630 +65333036653062386332653433366532383132376362356264373432653564316530656163396161 +34386133326133393638653961383032323830366232346134323663653436383034356334643137 +37663062396534616563363166333535336466373335613733323864363736633263646438336662 +39643333366565383130346430623637666132656330653833393836313566393461656638653938 +32623664363835376366353532646564643235356637363432363164353837643066643932383163 +34643232623734613738613638306664326333346439383830326662653934636536386166323239 +63626264613661336663316636353436643262353733303666663866346431353533353435346461 +34343763316364303061353030626431613434383062363737356131366234626137656331383433 +61653339616336376562326638343462343435613632656632353532376438623462656266326539 +35336465366133386534343466323936663934306363353462356530323531383961633336393234 +37366266343831386462626539306637623165363066616164393635373631393737613830363661 +34656530336634336531373830613039616231343238653637623532386538626365386364353930 +32396535643435306231386636363562623239393461393237303030353361633639626632653562 +31663261653663356266376264623762353261613163303430623738343939336438333532353436 +62343465353033343034373837376538323239653064623135613766343463373436663236616366 +30313062663861646331316362653230653139623937343538346230613832623965366666323365 +35623635646263396436346162333835343335623364393037353366343537336462636163303761 +37346663306339333264643034393630613833343163306430396333656637316662616439343136 +35336366306535333733633465343538383462616631326433666663303638633837313065656666 +34326136363433303264353331393133626639343166303364343065333266646439663463353833 +62346637363265643736616238663033666563326462633562643530383862616265306439376439 +63353432343534303433353138663461303165313433353866303838346338363761376333366138 +34353035356132383836313134643363636532353834323438313933346534333238646532326263 +30663964326564623164306664323265383134616165353634653261383931326663356137363636 +36383663666537373863623532376165343334633031626661306664663139616637636162373838 +32663863393765343431613536386133653731363266663566303665316435363830303035656230 +31366139636332336531326131346237653732313337343336393463373438373531653262363035 +39613739323832353332643030343861373836653064326139343965613563363166363236343733 +39653666333438613131396566623237643765393435313163333031383461303366396439353437 +31376562323231386133356231343366623363396437323866326234613034353664323163646631 +36386436653465306665646130616563646564343764346532363961663762303132336137356231 +61376565333036313033633732616430343266356435376434386266306432633633316361656538 +66393266383735336632653834373561636261663738333039653934636434343165353065353566 +34616431616430306234313032346235383065633734323632643065613634343834386664313336 +33333861306362333032346633636630326562323430323863373532613831313931303164303438 +36353831623262663030313564616634376235383666656434373261643464363264653831386534 +63663931643733313239323138393936373435303263663662613266306162373363353334336430 +62393835363263666136646631653738393432346331353538306666663864323233386665306332 +39353839313731313832333139613830396539306133363236653138643161323337356534643037 +38626636353465636166313965346364663238363032383537306532303332643839346230666564 +66613261396633346365633730633738306438326234386438373234393537353635626664616239 +35363737363537326132376638373839326139643135333234333764626162343766623531643265 +32656464383064626566653332323366306138373537663634333833613932666264306164666435 +39623335323163636135313361376231316231626335343765323261653134636439643561663463 +33376130356362316334386162306333333038316664636464653463313835356461386464653035 +38626230376339663361306161386332656230353737376133306464626466653038653266646139 +62313937383262633938633735393765616365646165396436653434303835666162373164333938 +61643432343237386133316433613030633638633731323936393562376139353033643562633835 +36393837383261663761393763613039623263643266303637613461626462313162613762383535 +33373064646533366136376563383663633331373161646534653330653566303332616262386564 +30313330346463643032363464346537613430633163306365313866373031393965383134323031 +63653238343338613033633336366235623332646239613235643637313666616263313833366362 +66666633393232303639363966306230663731353730333264663732386235653633303336316139 +33646263303164613437343735616362636364376261323364616131383433633864323132353063 +32646530323838633962613834663564326137663466623935323665343132653037386339613961 +61666363323763636361396330663537386630306637663638343565653437623966313738306366 +37633436373434396231396461656634316638393336653964396266363532343864346264323830 +32616461353630353730393639306231363662663034626434316661653936636262393761633934 +33393863383461316532373839663432336666356332393562393634386533346262306331643733 +62373162643763616637316233636163636363363639303662313630336233353162343437356533 +30303339666136343665313038643930383539313035373566336139643766353534386232343330 +32633836356435326332666266393237353162646537376564656236353138386531323032626636 +37373963666666346462343938656466343766323434373064363736386134656232636631653535 +38643739333230313265643132356430393035656662356465623836323038663237383439396137 +33383335356433383030393131666233663832643638646537363665396435323261613732303363 +63653634666164636632306462346166623237303936326435326431636630323562613233386462 +64376562316362653937646266366232653733663764633738303162616538656661326261643139 +30376233633835333637356134626266376566393962353639663039323439616639316465353038 +32373136643162633836663937373664306665643638366237653631313631393934643135393261 +35353338393761383431326631633166653932313335346165653364316431333831306135643339 +63316538343963336237653366343836343330656364373661343866656235303931633432313966 +35393631333263363533316537616132323038303230653838326266336434336532346439626537 +30656334316239356261386633393635643563336230326362303034633235323830343435646136 +63333237346164323763633665636234313263393366636233393538353335383933636331656261 +39313733383238613634363139326536616237353031666232663161653763326162386464626533 +39633931353035356561643564656637633163396561356430663636313231333962363765373536 +62613366616538306331623537613564353437366239616332626237623466613931653339333431 +39646239383438623063663063333861356536326637656337303561346432333539656431373733 +31316461343636313430303434336535353430353932373536366162333938646464653763356237 +39363938373065663732393862383031336438396164366664323130343734323130306662343138 +35383734663465303034393561306264316139623265386134323162373734393364393230333234 +31383336363832373962613535346136353037383065363066653935373435383037313764636266 +65383133333433326662626639356162303261393866373732366665353465356534333431633639 +39303539346266316132643537393564346562353533363438376363613036613739633939353564 +37313764316661663939396664376538626437366534383630613030323465396535366537346263 +31636338353233306333653962373763353439346337323839633238636336653439386362333533 +31653264343435633432646633386531386139366461646133396334326631333632393462666232 +35383735356232336335336435343932613433393530653633616466373965383935383861366136 +31303936616333396638326465633531373164626232356362613130376434303633366238336666 +37643035303932393265363037646430343866323163386234633739393938656530646533633163 +61653135633436653566303832383564663462353235353134313033616364636561643663383835 +30326561653934636164636363363736346630316263616664386233373535356339336239653333 +32336362363033616337663936623331663134656665643739626637353739666231643766346134 +63353631663631643436633935306261373939323530363639666366373531626231366138643766 +34623137643233333232303339323464333566633038333539356337306134383330353632383965 +34626461316261363830376664343336666132376365333635666131653933323562666635353763 +66376335663861623332623436333161353161336337313136613531646632363531383138343130 +39306330336239393965383939333765633532376539653661353965373030313038666161313265 +37653164343832333937356434646230623736646561623561643135626531343063626636663333 +30363937613834663434313038653739613933656534626532326136323336386164366339313830 +35396130386634666363656237373835313563343961633838383766663933376439326364616530 +64383136623739623137363562666431333565643130346166663531373738643761363337333536 +38313866363339393066643339653734316266393037396432356138303137343536313134623434 +62623362383863303539663264386430353436613261643865666534356538626538623630643766 +66646266336238396230613838626161336637313564303465663034653232306363306430633438 +35303364356531636565303965323539393233663566646562383863323461396166623032396338 +36346537343633643864323163623464623539616434376137313538646333666339626437656330 +31326463666335626361616566383065363264356663323435646337663566636436356336376531 +38636466386539663937326462636464613638653833666263333134396236313432643030366138 +66326364623435356633666463333066373238363864356230633634353764616331303163636266 +36306239323561326134393665643031663362636535656139376237383130616663316362326366 +61623236356562346233643332663261626333646638373262613664623935626266306230356164 +32336363383538333935333062386366653839303139366565343933346266623637363534393335 +31346431313631373535306138633930383536613636393034343434623664376237373066633036 +30323864613430333239326366613439613632633933656631356237333930303330326264646563 +38303066323563303262663530326637663964373039336431353339626436363335326466313262 +35386533633833363038636339663265626261373464633565666166633135333461363165353931 +63633332613966663133 diff --git a/ansible/infra/403_service_probe_registration.yml b/ansible/infra/403_service_probe_registration.yml new file mode 100644 index 0000000..a8c8b35 --- /dev/null +++ b/ansible/infra/403_service_probe_registration.yml @@ -0,0 +1,47 @@ +--- +# The per-service health probes. +# +# These are NOT the same thing as the systemd checks in infra/401. Those answer +# "is the unit running"; these answer "does the service actually respond" - an +# RPC call to bitcoind, a TCP connect to Fulcrum's Electrum port, an HTTP fetch +# from the Mempool backend. A process can be perfectly `active` and useless, +# which is precisely the gap these close. +# +# The checks themselves live in each service's own role, deployed by that +# service's playbook. This play only registers where they report, because the +# endpoints must exist in Gatus before the first push arrives. +# +# They used to push to Uptime Kuma. The scripts now POST with a bearer token +# instead of GETting ?status=up, and each host uses its own token. + +- name: Register the per-service probes with Gatus + hosts: observability + become: yes + + vars: + probes: + - {name: bitcoin-knots, host: knots_box_local} + - {name: datum-gateway, host: knots_box_local} + - {name: fulcrum, host: fulcrum_box_local} + - {name: phoenixd, host: vipy} + - {name: forgejo-runner, host: forgejo_runner_local} + - {name: mempool-mariadb, host: mempool_box_local} + - {name: mempool-backend, host: mempool_box_local} + - {name: mempool-frontend, host: mempool_box_local} + + tasks: + - name: Build the probe endpoint list + ansible.builtin.set_fact: + probe_endpoints: "{{ probe_endpoints | default([]) + [{ + 'name': item.name, + 'group': 'probe', + 'token': gatus_push_tokens[item.host], + 'heartbeat': '16m'}] }}" + loop: "{{ probes }}" + + - name: Register the probe endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: probes + gatus_endpoint_external: "{{ probe_endpoints }}" diff --git a/ansible/roles/bitcoin_knots/defaults/main.yml b/ansible/roles/bitcoin_knots/defaults/main.yml index 3a12ad1..1045e77 100644 --- a/ansible/roles/bitcoin_knots/defaults/main.yml +++ b/ansible/roles/bitcoin_knots/defaults/main.yml @@ -51,6 +51,9 @@ bitcoin_group: bitcoin # check, exit honestly, report nowhere. Any endpoint accepting an HTTP ping # works; nothing here is specific to a monitoring product. healthcheck_push_url: "" +# Bearer token for the Gatus external endpoint. Required whenever a push URL +# is set: Gatus rejects an unauthenticated push with 401. +healthcheck_push_token: "" # --- Logging ---------------------------------------------------------------- # The live node logs to a file. Set to "" to use printtoconsole=1 (journald). diff --git a/ansible/roles/bitcoin_knots/tasks/healthcheck.yml b/ansible/roles/bitcoin_knots/tasks/healthcheck.yml index 34b5529..c6c8db3 100644 --- a/ansible/roles/bitcoin_knots/tasks/healthcheck.yml +++ b/ansible/roles/bitcoin_knots/tasks/healthcheck.yml @@ -24,7 +24,7 @@ dest: /etc/systemd/system/bitcoin-knots-healthcheck.service owner: root group: root - mode: '0644' + mode: "0600" - name: Create systemd timer for Bitcoin Knots health check ansible.builtin.template: diff --git a/ansible/roles/bitcoin_knots/templates/healthcheck.service.j2 b/ansible/roles/bitcoin_knots/templates/healthcheck.service.j2 index a2ba83d..9df056e 100644 --- a/ansible/roles/bitcoin_knots/templates/healthcheck.service.j2 +++ b/ansible/roles/bitcoin_knots/templates/healthcheck.service.j2 @@ -7,6 +7,7 @@ Type=oneshot User=root ExecStart=/usr/local/bin/bitcoin-knots-healthcheck-push.sh Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} StandardOutput=journal StandardError=journal diff --git a/ansible/roles/bitcoin_knots/templates/healthcheck.sh.j2 b/ansible/roles/bitcoin_knots/templates/healthcheck.sh.j2 index 4e6ea9d..538b8dd 100644 --- a/ansible/roles/bitcoin_knots/templates/healthcheck.sh.j2 +++ b/ansible/roles/bitcoin_knots/templates/healthcheck.sh.j2 @@ -12,6 +12,7 @@ RPC_PORT={{ bitcoin_rpc_port }} RPC_USER="{{ bitcoin_rpc_user }}" RPC_PASSWORD="{{ bitcoin_rpc_password }}" PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" # Check if bitcoind RPC is responding check_bitcoind() { @@ -46,8 +47,14 @@ report() { # URL encode spaces in message local encoded_msg="${msg// /%20}" - if ! curl -s --max-time 10 --retry 2 -o /dev/null \ - "${PUSH_URL}?status=${status}&msg=${encoded_msg}&ping="; then + # Gatus external endpoint: a POST with a bearer token, NOT Uptime Kuma's + # GET with ?status=up. The callers still pass up/down, so the mapping is + # done here rather than at every call site. + local _ok=false + [ "${status}" = "up" ] && _ok=true + if ! curl -s --max-time 15 --retry 2 -o /dev/null -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${_ok}&error=${encoded_msg}"; then return 1 fi } diff --git a/ansible/roles/datum_gateway/defaults/main.yml b/ansible/roles/datum_gateway/defaults/main.yml index b9829f9..0c30aa7 100644 --- a/ansible/roles/datum_gateway/defaults/main.yml +++ b/ansible/roles/datum_gateway/defaults/main.yml @@ -56,3 +56,6 @@ datum_pooled_mining_only: true # WHERE TO REPORT HEALTH — the one place to plug in monitoring. Empty means # check, exit honestly, report nowhere. healthcheck_push_url: "" +# Bearer token for the Gatus external endpoint. Required whenever a push URL +# is set: Gatus rejects an unauthenticated push with 401. +healthcheck_push_token: "" diff --git a/ansible/roles/datum_gateway/tasks/healthcheck.yml b/ansible/roles/datum_gateway/tasks/healthcheck.yml index 211b85b..b5ccb90 100644 --- a/ansible/roles/datum_gateway/tasks/healthcheck.yml +++ b/ansible/roles/datum_gateway/tasks/healthcheck.yml @@ -19,7 +19,7 @@ dest: /etc/systemd/system/datum-gateway-healthcheck.service owner: root group: root - mode: '0644' + mode: "0600" notify: Restart datum-gateway health check timer - name: Create datum-gateway health check systemd timer diff --git a/ansible/roles/datum_gateway/templates/healthcheck.service.j2 b/ansible/roles/datum_gateway/templates/healthcheck.service.j2 index 5dcd3f5..e21672e 100644 --- a/ansible/roles/datum_gateway/templates/healthcheck.service.j2 +++ b/ansible/roles/datum_gateway/templates/healthcheck.service.j2 @@ -7,6 +7,7 @@ Type=oneshot User=root ExecStart=/usr/local/bin/datum-gateway-healthcheck-push.sh Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} StandardOutput=journal StandardError=journal diff --git a/ansible/roles/datum_gateway/templates/healthcheck.sh.j2 b/ansible/roles/datum_gateway/templates/healthcheck.sh.j2 index ba8d069..43ec6b2 100644 --- a/ansible/roles/datum_gateway/templates/healthcheck.sh.j2 +++ b/ansible/roles/datum_gateway/templates/healthcheck.sh.j2 @@ -5,6 +5,7 @@ # systemctl is-failed datum-gateway-healthcheck.service # Reporting anywhere else is optional and generic. PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" STRATUM_PORT={{ datum_gateway_stratum_port }} check_datum() { @@ -19,8 +20,14 @@ report() { # No push URL is normal, not an error: the exit code below is still a # complete answer for anything reading unit state. [ -n "$PUSH_URL" ] || return 0 - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true + # Gatus external endpoint: a POST with a bearer token, NOT Uptime Kuma's + # GET with ?status=up. The callers still pass up/down, so the mapping is + # done here rather than at every call site. + local _ok=false + [ "${status}" = "up" ] && _ok=true + curl -s --max-time 15 --retry 2 -o /dev/null -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${_ok}&error=${msg// /%20}" || true } if check_datum; then diff --git a/ansible/roles/forgejo_runner/defaults/main.yml b/ansible/roles/forgejo_runner/defaults/main.yml index 4a80d73..aba6b61 100644 --- a/ansible/roles/forgejo_runner/defaults/main.yml +++ b/ansible/roles/forgejo_runner/defaults/main.yml @@ -36,3 +36,6 @@ healthcheck_service_name: forgejo-runner-healthcheck # A pull-based monitor (Prometheus node_exporter textfile, say) needs this left # empty — it reads the systemd unit state instead. healthcheck_push_url: "" +# Bearer token for the Gatus external endpoint. Required whenever a push URL +# is set: Gatus rejects an unauthenticated push with 401. +healthcheck_push_token: "" diff --git a/ansible/roles/forgejo_runner/tasks/healthcheck.yml b/ansible/roles/forgejo_runner/tasks/healthcheck.yml index 0abdf1f..42fbaf7 100644 --- a/ansible/roles/forgejo_runner/tasks/healthcheck.yml +++ b/ansible/roles/forgejo_runner/tasks/healthcheck.yml @@ -27,7 +27,7 @@ dest: "/etc/systemd/system/{{ healthcheck_service_name }}.service" owner: root group: root - mode: '0644' + mode: "0600" - name: Create healthcheck systemd timer ansible.builtin.template: diff --git a/ansible/roles/forgejo_runner/templates/healthcheck.service.j2 b/ansible/roles/forgejo_runner/templates/healthcheck.service.j2 index aae9eb5..173f42b 100644 --- a/ansible/roles/forgejo_runner/templates/healthcheck.service.j2 +++ b/ansible/roles/forgejo_runner/templates/healthcheck.service.j2 @@ -5,6 +5,8 @@ After=network.target [Service] Type=oneshot ExecStart={{ healthcheck_script_path }} +Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} User=root StandardOutput=journal StandardError=journal diff --git a/ansible/roles/forgejo_runner/templates/healthcheck.sh.j2 b/ansible/roles/forgejo_runner/templates/healthcheck.sh.j2 index b9d43c3..9aa5ac1 100644 --- a/ansible/roles/forgejo_runner/templates/healthcheck.sh.j2 +++ b/ansible/roles/forgejo_runner/templates/healthcheck.sh.j2 @@ -10,34 +10,38 @@ # it. Nothing here knows or cares which monitoring product is on the other end. LOG_FILE="{{ healthcheck_log_file }}" -PUSH_URL="{{ healthcheck_push_url }}" +# Read from the environment rather than templated in, so the unit file is the +# only place the token lives and the script is not secret. +PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" log_message() { echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" } +# Gatus external endpoint: a POST with a bearer token and success=true|false. +# +# This used to report ONLY success - it exited before pushing when the runner +# was down - so a failure was invisible until the heartbeat window expired. +# Reporting the failure is the whole point of having a check. +report() { + local ok="$1" msg="$2" + [ -n "$PUSH_URL" ] || return 0 + curl -s --max-time 15 --retry 2 -o /dev/null -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${ok}&error=${msg// /%20}" || true +} + main() { if ! systemctl is-active --quiet forgejo-runner; then log_message "ERROR: forgejo-runner is not active" + report false "forgejo-runner is not active" exit 1 fi - if [ -z "$PUSH_URL" ]; then - # Healthy, and nothing to report to. Not an error: the exit code below - # is still a complete answer for anything reading unit state. - log_message "forgejo-runner is active (no push URL configured)" - exit 0 - fi - - log_message "forgejo-runner is active, sending ping" - response=$(curl -s -w "\n%{http_code}" "$PUSH_URL?status=up&msg=forgejo-runner%20is%20active" 2>&1) - http_code=$(echo "$response" | tail -n1) - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Ping sent successfully (HTTP $http_code)" - else - log_message "ERROR: Failed to send ping (HTTP $http_code)" - exit 1 - fi + log_message "forgejo-runner is active" + report true "active" + exit 0 } main diff --git a/ansible/roles/fulcrum/defaults/main.yml b/ansible/roles/fulcrum/defaults/main.yml index cc778d4..3e97fa0 100644 --- a/ansible/roles/fulcrum/defaults/main.yml +++ b/ansible/roles/fulcrum/defaults/main.yml @@ -71,6 +71,9 @@ fulcrum_group: fulcrum # check, exit honestly, report nowhere. Any endpoint accepting an HTTP ping # works; nothing here is specific to a monitoring product. healthcheck_push_url: "" +# Bearer token for the Gatus external endpoint. Required whenever a push URL +# is set: Gatus rejects an unauthenticated push with 401. +healthcheck_push_token: "" # Explicit db_mem in MB. When set it wins over fulcrum_db_mem_percent; empty # means compute from RAM. Set here because the live host had been hand-tuned to diff --git a/ansible/roles/fulcrum/tasks/healthcheck.yml b/ansible/roles/fulcrum/tasks/healthcheck.yml index e86726e..983bffb 100644 --- a/ansible/roles/fulcrum/tasks/healthcheck.yml +++ b/ansible/roles/fulcrum/tasks/healthcheck.yml @@ -19,7 +19,7 @@ dest: /etc/systemd/system/fulcrum-healthcheck.service owner: root group: root - mode: '0644' + mode: "0600" - name: Create systemd timer for Fulcrum health check ansible.builtin.template: diff --git a/ansible/roles/fulcrum/templates/healthcheck.service.j2 b/ansible/roles/fulcrum/templates/healthcheck.service.j2 index 27995f4..b519808 100644 --- a/ansible/roles/fulcrum/templates/healthcheck.service.j2 +++ b/ansible/roles/fulcrum/templates/healthcheck.service.j2 @@ -7,6 +7,7 @@ Type=oneshot User=root ExecStart=/usr/local/bin/fulcrum-healthcheck-push.sh Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} StandardOutput=journal StandardError=journal diff --git a/ansible/roles/fulcrum/templates/healthcheck.sh.j2 b/ansible/roles/fulcrum/templates/healthcheck.sh.j2 index ee1c7da..4449fed 100644 --- a/ansible/roles/fulcrum/templates/healthcheck.sh.j2 +++ b/ansible/roles/fulcrum/templates/healthcheck.sh.j2 @@ -10,6 +10,7 @@ FULCRUM_HOST="{{ fulcrum_tcp_bind }}" FULCRUM_PORT={{ fulcrum_tcp_port }} PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" check_fulcrum() { timeout 5 bash -c "echo > /dev/tcp/${FULCRUM_HOST}/${FULCRUM_PORT}" 2>/dev/null @@ -20,8 +21,14 @@ report() { # No push URL is normal, not an error: the exit code below is still a # complete answer for anything reading unit state. [ -n "$PUSH_URL" ] || return 0 - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true + # Gatus external endpoint: a POST with a bearer token, NOT Uptime Kuma's + # GET with ?status=up. The callers still pass up/down, so the mapping is + # done here rather than at every call site. + local _ok=false + [ "${status}" = "up" ] && _ok=true + curl -s --max-time 15 --retry 2 -o /dev/null -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${_ok}&error=${msg// /%20}" || true } if check_fulcrum; then diff --git a/ansible/roles/mempool/defaults/main.yml b/ansible/roles/mempool/defaults/main.yml index 2f983c6..9ce8f6a 100644 --- a/ansible/roles/mempool/defaults/main.yml +++ b/ansible/roles/mempool/defaults/main.yml @@ -45,6 +45,8 @@ mariadb_user: "mempool" # push_url is where to report, and is the single plug-in point for whatever # monitoring exists. Empty means check, exit honestly, report nowhere. # The URLs are credentials, so callers pass them from the vault. +healthcheck_push_token: "" + mempool_healthchecks: - name: mariadb label: MariaDB diff --git a/ansible/roles/mempool/tasks/healthcheck.yml b/ansible/roles/mempool/tasks/healthcheck.yml index fa53a9c..1e1ac28 100644 --- a/ansible/roles/mempool/tasks/healthcheck.yml +++ b/ansible/roles/mempool/tasks/healthcheck.yml @@ -22,7 +22,7 @@ dest: "/etc/systemd/system/mempool-{{ hc.name }}-healthcheck.service" owner: root group: root - mode: '0644' + mode: "0600" loop: "{{ mempool_healthchecks }}" loop_control: loop_var: hc diff --git a/ansible/roles/mempool/templates/healthcheck-backend.sh.j2 b/ansible/roles/mempool/templates/healthcheck-backend.sh.j2 index 3a6630a..fe0f57c 100644 --- a/ansible/roles/mempool/templates/healthcheck-backend.sh.j2 +++ b/ansible/roles/mempool/templates/healthcheck-backend.sh.j2 @@ -2,6 +2,7 @@ # Mempool backend health check — managed by Ansible (roles/mempool) # The exit code is the answer; systemd keeps it. Reporting is optional. PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" BACKEND_PORT="{{ mempool_backend_port }}" check() { @@ -10,8 +11,14 @@ check() { report() { [ -n "$PUSH_URL" ] || return 0 - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${PUSH_URL}?status=$1&msg=${2// /%20}&ping=" || true + # Gatus external endpoint: a POST with a bearer token, NOT Uptime Kuma's + # GET with ?status=up. The callers still pass up/down, so the mapping is + # done here rather than at every call site. + local _ok=false + [ "$1" = "up" ] && _ok=true + curl -s --max-time 15 --retry 2 -o /dev/null -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${_ok}&error=${2// /%20}" || true } if check; then report up "OK"; exit 0 diff --git a/ansible/roles/mempool/templates/healthcheck-frontend.sh.j2 b/ansible/roles/mempool/templates/healthcheck-frontend.sh.j2 index b2541d3..8452202 100644 --- a/ansible/roles/mempool/templates/healthcheck-frontend.sh.j2 +++ b/ansible/roles/mempool/templates/healthcheck-frontend.sh.j2 @@ -2,6 +2,7 @@ # Mempool frontend health check — managed by Ansible (roles/mempool) # The exit code is the answer; systemd keeps it. Reporting is optional. PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" FRONTEND_PORT="{{ mempool_frontend_port }}" check() { @@ -10,8 +11,14 @@ check() { report() { [ -n "$PUSH_URL" ] || return 0 - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${PUSH_URL}?status=$1&msg=${2// /%20}&ping=" || true + # Gatus external endpoint: a POST with a bearer token, NOT Uptime Kuma's + # GET with ?status=up. The callers still pass up/down, so the mapping is + # done here rather than at every call site. + local _ok=false + [ "$1" = "up" ] && _ok=true + curl -s --max-time 15 --retry 2 -o /dev/null -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${_ok}&error=${2// /%20}" || true } if check; then report up "OK"; exit 0 diff --git a/ansible/roles/mempool/templates/healthcheck-mariadb.sh.j2 b/ansible/roles/mempool/templates/healthcheck-mariadb.sh.j2 index cbc36f5..922adab 100644 --- a/ansible/roles/mempool/templates/healthcheck-mariadb.sh.j2 +++ b/ansible/roles/mempool/templates/healthcheck-mariadb.sh.j2 @@ -2,6 +2,7 @@ # Mempool MariaDB health check — managed by Ansible (roles/mempool) # The exit code is the answer; systemd keeps it. Reporting is optional. PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" check() { {% raw %} @@ -13,8 +14,14 @@ report() { # No push URL is normal, not an error. The previous version logged # "ERROR: UPTIME_KUMA_PUSH_URL not set" on every fire, once a minute. [ -n "$PUSH_URL" ] || return 0 - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${PUSH_URL}?status=$1&msg=${2// /%20}&ping=" || true + # Gatus external endpoint: a POST with a bearer token, NOT Uptime Kuma's + # GET with ?status=up. The callers still pass up/down, so the mapping is + # done here rather than at every call site. + local _ok=false + [ "$1" = "up" ] && _ok=true + curl -s --max-time 15 --retry 2 -o /dev/null -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${_ok}&error=${2// /%20}" || true } if check; then report up "OK"; exit 0 diff --git a/ansible/roles/mempool/templates/healthcheck.service.j2 b/ansible/roles/mempool/templates/healthcheck.service.j2 index 44548bd..76fa4e9 100644 --- a/ansible/roles/mempool/templates/healthcheck.service.j2 +++ b/ansible/roles/mempool/templates/healthcheck.service.j2 @@ -7,6 +7,7 @@ Type=oneshot User=root ExecStart=/usr/local/bin/mempool-{{ hc.name }}-healthcheck-push.sh Environment=HEALTHCHECK_PUSH_URL={{ hc.push_url }} +Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} StandardOutput=journal StandardError=journal diff --git a/ansible/roles/phoenixd/defaults/main.yml b/ansible/roles/phoenixd/defaults/main.yml index 9fea220..2fcf1ea 100644 --- a/ansible/roles/phoenixd/defaults/main.yml +++ b/ansible/roles/phoenixd/defaults/main.yml @@ -51,3 +51,6 @@ phoenixd_healthcheck_service_name: phoenixd-healthcheck # Empty means check, log, exit honestly, report nowhere. Any endpoint that # accepts an HTTP ping works; nothing here is specific to a monitoring product. healthcheck_push_url: "" +# Bearer token for the Gatus external endpoint. Required whenever a push URL +# is set: Gatus rejects an unauthenticated push with 401. +healthcheck_push_token: "" diff --git a/ansible/roles/phoenixd/tasks/healthcheck.yml b/ansible/roles/phoenixd/tasks/healthcheck.yml index ad5cc22..d439047 100644 --- a/ansible/roles/phoenixd/tasks/healthcheck.yml +++ b/ansible/roles/phoenixd/tasks/healthcheck.yml @@ -19,7 +19,7 @@ dest: "/etc/systemd/system/{{ phoenixd_healthcheck_service_name }}.service" owner: root group: root - mode: "0644" + mode: "0600" notify: Restart phoenixd health check timer - name: Create phoenixd health check systemd timer diff --git a/ansible/roles/phoenixd/templates/healthcheck.service.j2 b/ansible/roles/phoenixd/templates/healthcheck.service.j2 index 4060dbc..6ff21ab 100644 --- a/ansible/roles/phoenixd/templates/healthcheck.service.j2 +++ b/ansible/roles/phoenixd/templates/healthcheck.service.j2 @@ -7,6 +7,7 @@ Type=oneshot User=root ExecStart={{ phoenixd_healthcheck_script_path }} Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} +Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} StandardOutput=journal StandardError=journal diff --git a/ansible/roles/phoenixd/templates/healthcheck.sh.j2 b/ansible/roles/phoenixd/templates/healthcheck.sh.j2 index d2cec21..a1717c5 100644 --- a/ansible/roles/phoenixd/templates/healthcheck.sh.j2 +++ b/ansible/roles/phoenixd/templates/healthcheck.sh.j2 @@ -6,6 +6,7 @@ # systemctl is-failed {{ phoenixd_healthcheck_service_name }}.service # That is a complete answer on its own. Reporting anywhere else is optional. PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" +PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" export PHOENIX_DATADIR="{{ phoenixd_data_dir }}" check_phoenixd() { @@ -25,8 +26,14 @@ report() { # answers the question. The previous version logged ERROR here on every # single fire, once a minute, which is noise that trains you to ignore it. [ -n "$PUSH_URL" ] || return 0 - curl -s --max-time 10 --retry 2 -o /dev/null \ - "${PUSH_URL}?status=${status}&msg=${msg// /%20}&ping=" || true + # Gatus external endpoint: a POST with a bearer token, NOT Uptime Kuma's + # GET with ?status=up. The callers still pass up/down, so the mapping is + # done here rather than at every call site. + local _ok=false + [ "${status}" = "up" ] && _ok=true + curl -s --max-time 15 --retry 2 -o /dev/null -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_URL}?success=${_ok}&error=${msg// /%20}" || true } if check_phoenixd; then diff --git a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml index 0ca1e13..d67006b 100644 --- a/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml +++ b/ansible/services/bitcoin-knots/deploy_bitcoin_knots_playbook.yml @@ -12,7 +12,8 @@ vars: # Preserves the push URL this check has been reporting to. The role knows # nothing about Uptime Kuma — this is just "a URL that accepts a ping". - healthcheck_push_url: "{{ healthcheck_push_urls.bitcoin_knots | default('') }}" + healthcheck_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/probe_bitcoin-knots/external" + healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" roles: - bitcoin_knots diff --git a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml index 84de178..2e88a27 100644 --- a/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml +++ b/ansible/services/datum-gateway/deploy_datum_gateway_playbook.yml @@ -13,7 +13,8 @@ vars: # Preserves the push URL this check reports to. The role knows nothing about # Uptime Kuma — this is just "a URL that accepts a ping". - healthcheck_push_url: "{{ healthcheck_push_urls.datum_gateway | default('') }}" + healthcheck_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/probe_datum-gateway/external" + healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" roles: - datum_gateway diff --git a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml index 031f081..04081ff 100644 --- a/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml +++ b/ansible/services/forgejo-runner/deploy_forgejo_runner_playbook.yml @@ -7,6 +7,7 @@ # move to a role changes no behaviour. The role itself knows nothing about # Uptime Kuma — this is just "a URL that accepts a ping", and whatever # replaces it sets the same variable. - healthcheck_push_url: "{{ healthcheck_push_urls.forgejo_runner | default('') }}" + healthcheck_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/probe_forgejo-runner/external" + healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" roles: - forgejo_runner diff --git a/ansible/services/forgejo/deploy_forgejo_playbook.yml b/ansible/services/forgejo/deploy_forgejo_playbook.yml index 04d7041..f908586 100644 --- a/ansible/services/forgejo/deploy_forgejo_playbook.yml +++ b/ansible/services/forgejo/deploy_forgejo_playbook.yml @@ -6,7 +6,6 @@ vars: forgejo_subdomain: "{{ subdomains.forgejo }}" forgejo_domain: "{{ forgejo_subdomain }}.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - name: Ensure required packages are installed @@ -91,123 +90,3 @@ caddy_site_name: forgejo caddy_site_domain: "{{ forgejo_domain }}" caddy_site_upstream: "localhost:{{ forgejo_port }}" - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create Uptime Kuma monitor setup script for Forgejo - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_forgejo_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ forgejo_domain }}/api/healthz" - monitor_name: "Forgejo" - mode: '0644' - - - name: Run Uptime Kuma monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_forgejo_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_forgejo_monitor.py - - /tmp/ansible_config.yml - diff --git a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml index 8927aba..1b7b963 100644 --- a/ansible/services/fulcrum/deploy_fulcrum_playbook.yml +++ b/ansible/services/fulcrum/deploy_fulcrum_playbook.yml @@ -8,7 +8,8 @@ vars: # Preserves the push URL this check has been configured with. The role knows # nothing about Uptime Kuma — this is just "a URL that accepts a ping". - healthcheck_push_url: "{{ healthcheck_push_urls.fulcrum | default('') }}" + healthcheck_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/probe_fulcrum/external" + healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" roles: - fulcrum diff --git a/ansible/services/headscale/deploy_headscale_playbook.yml b/ansible/services/headscale/deploy_headscale_playbook.yml index 11c967e..e181f24 100644 --- a/ansible/services/headscale/deploy_headscale_playbook.yml +++ b/ansible/services/headscale/deploy_headscale_playbook.yml @@ -7,7 +7,6 @@ headscale_subdomain: "{{ subdomains.headscale }}" headscale_domain: "{{ headscale_subdomain }}.{{ root_domain }}" headscale_base_domain: "tailnet.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - name: Install required packages @@ -254,125 +253,6 @@ # All API operations require a valid Bearer token in the Authorization header reverse_proxy * http://localhost:{{ headscale_port }} - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create Uptime Kuma monitor setup script for Headscale - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_headscale_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ headscale_domain }}/health" - monitor_name: "Headscale" - mode: '0644' - - - name: Run Uptime Kuma monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_headscale_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_headscale_monitor.py - - /tmp/ansible_config.yml - handlers: - name: Restart headscale become: yes diff --git a/ansible/services/lnbits/deploy_lnbits_playbook.yml b/ansible/services/lnbits/deploy_lnbits_playbook.yml index 5d0c21d..7b89f33 100644 --- a/ansible/services/lnbits/deploy_lnbits_playbook.yml +++ b/ansible/services/lnbits/deploy_lnbits_playbook.yml @@ -6,7 +6,6 @@ vars: lnbits_subdomain: "{{ subdomains.lnbits }}" lnbits_domain: "{{ lnbits_subdomain }}.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - name: Create lnbits directory @@ -152,123 +151,3 @@ caddy_site_upstream: "localhost:{{ lnbits_port }}" caddy_site_headers_up: X-Forwarded-Host: "{{ lnbits_domain }}" - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create Uptime Kuma monitor setup script for LNBits - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_lnbits_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ lnbits_domain }}/api/v1/health" - monitor_name: "LNBits" - mode: '0644' - - - name: Run Uptime Kuma monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_lnbits_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_lnbits_monitor.py - - /tmp/ansible_config.yml - diff --git a/ansible/services/memos/deploy_memos_playbook.yml b/ansible/services/memos/deploy_memos_playbook.yml index 756ab4f..8b21b85 100644 --- a/ansible/services/memos/deploy_memos_playbook.yml +++ b/ansible/services/memos/deploy_memos_playbook.yml @@ -62,14 +62,6 @@ owner: root group: root - - name: Clean up temporary files - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/memos.tar.gz - - /tmp/memos - - name: Create memos environment file copy: dest: "{{ memos_config_dir }}/memos.env" @@ -136,18 +128,7 @@ msg: "Memos is running on port {{ memos_port }}. Access via Tailscale at http://{{ memos_tailscale_hostname }}:{{ memos_port }}" handlers: - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - name: Restart memos - when: uptime_kuma_enabled | default(false) systemd: name: memos state: restarted @@ -161,7 +142,6 @@ vars: memos_subdomain: "{{ subdomains.memos }}" memos_domain: "{{ memos_subdomain }}.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - name: Publish Memos through Caddy (via Tailscale) @@ -172,117 +152,3 @@ caddy_site_domain: "{{ memos_domain }}" caddy_site_upstream: "{{ memos_tailscale_hostname }}:{{ memos_port }}" caddy_site_resolvers: "100.100.100.100" - - - name: Create Uptime Kuma monitor setup script for Memos - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_memos_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - # Load configs - with open('/tmp/ansible_memos_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - # Connect to Uptime Kuma - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - error_msg = str(e) if str(e) else repr(e) - print(f"ERROR: {error_msg}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_memos_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ memos_domain }}/healthz" - monitor_name: "Memos" - mode: '0644' - - - name: Run Uptime Kuma monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_memos_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_memos_monitor.py - - /tmp/ansible_memos_config.yml diff --git a/ansible/services/mempool/deploy_mempool_playbook.yml b/ansible/services/mempool/deploy_mempool_playbook.yml index 76a3eb0..0042fb0 100644 --- a/ansible/services/mempool/deploy_mempool_playbook.yml +++ b/ansible/services/mempool/deploy_mempool_playbook.yml @@ -8,9 +8,12 @@ # about Uptime Kuma — these are just "URLs that accept a ping", and whatever # replaces it sets the same values. mempool_healthchecks: - - {name: mariadb, label: MariaDB, push_url: "{{ healthcheck_push_urls.mempool.mariadb | default('') }}"} - - {name: backend, label: Backend, push_url: "{{ healthcheck_push_urls.mempool.backend | default('') }}"} - - {name: frontend, label: Frontend, push_url: "{{ healthcheck_push_urls.mempool.frontend | default('') }}"} + - {name: mariadb, label: MariaDB, push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/probe_mempool-mariadb/external"} + - {name: backend, label: Backend, push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/probe_mempool-backend/external"} + - {name: frontend, label: Frontend, push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/probe_mempool-frontend/external"} + # One token for all three components: they run on the same host, so the + # blast radius is already that host. + healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" roles: - mempool diff --git a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml index c759592..9850e56 100644 --- a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml +++ b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml @@ -10,7 +10,6 @@ ntfy_emergency_app_ntfy_url: "https://{{ ntfy_service_domain }}" ntfy_emergency_app_ntfy_user: "{{ ntfy_username | default('') }}" ntfy_emergency_app_ntfy_password: "{{ ntfy_password | default('') }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - name: Create ntfy-emergency-app directory @@ -52,127 +51,3 @@ caddy_site_name: ntfy-emergency-app caddy_site_domain: "{{ ntfy_emergency_app_domain }}" caddy_site_upstream: "localhost:{{ ntfy_emergency_app_port }}" - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create Uptime Kuma monitor setup script for ntfy-emergency-app - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_ntfy_emergency_app_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - # Load configs - with open('/tmp/ansible_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - # Connect to Uptime Kuma - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - error_msg = str(e) if str(e) else repr(e) - print(f"ERROR: {error_msg}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ ntfy_emergency_app_domain }}" - monitor_name: "ntfy-emergency-app" - mode: '0644' - - - name: Run Uptime Kuma monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_ntfy_emergency_app_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_ntfy_emergency_app_monitor.py - - /tmp/ansible_config.yml diff --git a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml index 5e3780d..21d8b43 100644 --- a/ansible/services/personal-blog/deploy_personal_blog_playbook.yml +++ b/ansible/services/personal-blog/deploy_personal_blog_playbook.yml @@ -6,7 +6,6 @@ vars: personal_blog_subdomain: "{{ subdomains.personal_blog }}" personal_blog_domain: "{{ personal_blog_subdomain }}.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - name: Ensure user is in www-data group @@ -54,123 +53,3 @@ caddy_site_name: personal-blog caddy_site_domain: "{{ personal_blog_domain }}" caddy_site_root: "{{ personal_blog_web_root }}" - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create Uptime Kuma monitor setup script for Personal Blog - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_personal_blog_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - with open('/tmp/ansible_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - print(f"ERROR: {str(e)}", file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ personal_blog_domain }}" - monitor_name: "Personal Blog" - mode: '0644' - - - name: Run Uptime Kuma monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_personal_blog_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_personal_blog_monitor.py - - /tmp/ansible_config.yml - diff --git a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml index 0df1223..b9cf243 100644 --- a/ansible/services/phoenixd/deploy_phoenixd_playbook.yml +++ b/ansible/services/phoenixd/deploy_phoenixd_playbook.yml @@ -9,6 +9,7 @@ # decommissioning — its systemd Environment= was left empty. Leaving it empty # preserves that; the check still runs and its exit code is still the answer. # Set this to plug in whatever monitoring replaces it. - healthcheck_push_url: "{{ healthcheck_push_urls.phoenixd | default('') }}" + healthcheck_push_url: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints/probe_phoenixd/external" + healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" roles: - phoenixd diff --git a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml index d03363c..282fc2e 100644 --- a/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml +++ b/ansible/services/vaultwarden/deploy_vaultwarden_playbook.yml @@ -6,7 +6,6 @@ vars: vaultwarden_subdomain: "{{ subdomains.vaultwarden }}" vaultwarden_domain: "{{ vaultwarden_subdomain }}.{{ root_domain }}" - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" tasks: - name: Create vaultwarden directory @@ -84,128 +83,3 @@ caddy_site_name: vaultwarden caddy_site_domain: "{{ vaultwarden_domain }}" caddy_site_upstream: "localhost:{{ vaultwarden_port }}" - - # ═════════════════════════════════════════════════════════════════════════ - # DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. - # - # Every task below is inert: uptime_kuma_enabled is false in - # group_vars/all/main.yml, so they all skip and the deployment above still - # runs normally. Kept because the health-check logic is the durable part — - # when a replacement exists, rewire the push transport and flip the flag. - # - # What was being monitored: archive/uptime_kuma/MONITORS.md - # ═════════════════════════════════════════════════════════════════════════ - - name: Create Uptime Kuma monitor setup script for Vaultwarden - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/setup_vaultwarden_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import traceback - import yaml - from uptime_kuma_api import UptimeKumaApi, MonitorType - - try: - # Load configs - with open('/tmp/ansible_config.yml', 'r') as f: - config = yaml.safe_load(f) - - url = config['uptime_kuma_url'] - username = config['username'] - password = config['password'] - monitor_url = config['monitor_url'] - monitor_name = config['monitor_name'] - - # Connect to Uptime Kuma - api = UptimeKumaApi(url, timeout=30) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Find or create "services" group - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name='services') - # Refresh to get the group with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == 'services' and m.get('type') == 'group'), None) - - # Check if monitor already exists - existing_monitor = None - for monitor in monitors: - if monitor.get('name') == monitor_name: - existing_monitor = monitor - break - - # Get ntfy notification ID - notifications = api.get_notifications() - ntfy_notification_id = None - for notif in notifications: - if notif.get('type') == 'ntfy': - ntfy_notification_id = notif.get('id') - break - - if existing_monitor: - print(f"Monitor '{monitor_name}' already exists (ID: {existing_monitor['id']})") - print("Skipping - monitor already configured") - else: - print(f"Creating monitor '{monitor_name}'...") - api.add_monitor( - type=MonitorType.HTTP, - name=monitor_name, - url=monitor_url, - parent=group['id'], - interval=60, - maxretries=3, - retryInterval=60, - notificationIDList={ntfy_notification_id: True} if ntfy_notification_id else {} - ) - - api.disconnect() - print("SUCCESS") - - except Exception as e: - error_msg = str(e) if str(e) else repr(e) - print(f"ERROR: {error_msg}", file=sys.stderr) - traceback.print_exc(file=sys.stderr) - sys.exit(1) - mode: '0755' - - - name: Create temporary config for monitor setup - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - copy: - dest: /tmp/ansible_config.yml - content: | - uptime_kuma_url: "{{ uptime_kuma_api_url }}" - username: "{{ uptime_kuma_username }}" - password: "{{ uptime_kuma_password }}" - monitor_url: "https://{{ vaultwarden_domain }}/alive" - monitor_name: "Vaultwarden" - mode: '0644' - - - name: Run Uptime Kuma monitor setup - when: uptime_kuma_enabled | default(false) - command: python3 /tmp/setup_vaultwarden_monitor.py - delegate_to: localhost - become: no - register: monitor_setup - changed_when: "'SUCCESS' in monitor_setup.stdout" - ignore_errors: yes - - - name: Clean up temporary files - when: uptime_kuma_enabled | default(false) - delegate_to: localhost - become: no - file: - path: "{{ item }}" - state: absent - loop: - - /tmp/setup_vaultwarden_monitor.py - - /tmp/ansible_config.yml - diff --git a/ansible/site.yml b/ansible/site.yml index a5f8576..8e9151e 100644 --- a/ansible/site.yml +++ b/ansible/site.yml @@ -29,6 +29,10 @@ - import_playbook: infra/400_host_monitoring.yml - import_playbook: infra/401_service_monitoring.yml - import_playbook: infra/402_public_monitoring.yml +# Registers where the per-service probes report. The probes themselves are +# deployed by each service's own playbook further down; the endpoints must exist +# before the first push arrives. +- import_playbook: infra/403_service_probe_registration.yml # 910_docker says `hosts: managed`, but only 5 of 11 managed hosts have or need # Docker. Left out until it has a [docker] group — see the note in PLAN_7. diff --git a/ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml b/archive/uptime_kuma/setup_ntfy_uptime_kuma_notification.yml similarity index 100% rename from ansible/services/ntfy/setup_ntfy_uptime_kuma_notification.yml rename to archive/uptime_kuma/setup_ntfy_uptime_kuma_notification.yml From 85040d5f67fa4c9b124c0cd951f5ca64b94f79db Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 11:26:47 +0200 Subject: [PATCH 65/67] watchtower: remove from the estate, and with it ntfy watchtower is being destroyed. Removed from [vps], with its host_vars, its push token, and the six Gatus endpoints that referenced it (liveness, disk, two systemd services, the ntfy DNS record and the ntfy HTTP check). ntfy went with it - it ran nowhere else - so services/ntfy is deleted, subdomains.ntfy and ntfy_topic are gone from group_vars, and the ntfy playbook is out of site.yml. ntfy_topic already had no readers: the three infra/4xx plays that used it were deleted when their checks were superseded. Two things this exposed. services/ntfy/deploy_ntfy_playbook.yml was pointing at the WRONG MACHINE. It said `hosts: observability`, which resolves to the host `monitoring` (64.226.70.190) - but ntfy ran on watchtower, and ntfy.contrapeso.xyz pointed there. Running it would have installed ntfy on the new VPS. Moot now, but it is the same stale-identity failure as the rest: the group meant watchtower when the play was written, and nobody revisited it when the group changed. Watchtower was in [vps] and NO role group at all, while running caddy, ntfy and Uptime Kuma - nothing in the repo managed any of it. More seriously: ntfy-emergency-app on vipy (avisame.contrapeso.xyz) sends its notifications to https://ntfy.contrapeso.xyz, topic "emergencia". Destroying watchtower breaks it, and it is an EMERGENCY notifier - it would fail silently at exactly the moment it matters. That is NOT resolved here, deliberately: standing ntfy up elsewhere, pointing at ntfy.sh, or retiring the app are all decisions, not cleanups. What this change does is make the break impossible to miss. The URL was derived from subdomains.ntfy, so deleting that would have turned it into an undefined variable buried in a template. It is now an explicit ntfy_service_url in the app's own vars, still holding the old value, with the three options written above it. The ntfy credentials stay in the vault because that app still needs them - the vault was restored from HEAD and only watchtower's push token removed, rather than re-handling the plaintext. Verified: no reference to watchtower or its IP anywhere in the repo; Gatus down from 91 to 85 endpoints, 85 UP, 0 DOWN. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/main.yml | 5 +- ansible/group_vars/all/vault.yml | 309 +++++++++--------- ansible/host_vars/watchtower/main.yml | 13 - ansible/infra/402_public_monitoring.yml | 2 - ansible/inventory.ini | 1 - .../deploy_ntfy_emergency_app_playbook.yml | 13 +- .../ntfy_emergency_app_vars.yml | 6 + .../services/ntfy/deploy_ntfy_playbook.yml | 97 ------ ansible/services/ntfy/ntfy_vars.yml | 3 - ansible/site.yml | 4 - 10 files changed, 171 insertions(+), 282 deletions(-) delete mode 100644 ansible/host_vars/watchtower/main.yml delete mode 100644 ansible/services/ntfy/deploy_ntfy_playbook.yml delete mode 100644 ansible/services/ntfy/ntfy_vars.yml diff --git a/ansible/group_vars/all/main.yml b/ansible/group_vars/all/main.yml index 316b178..f77ae48 100644 --- a/ansible/group_vars/all/main.yml +++ b/ansible/group_vars/all/main.yml @@ -26,7 +26,6 @@ backup_pull_public_key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIOfIixKMhA9z+Nvyx6T subdomains: # Monitoring gatus: status - ntfy: ntfy # VPN infrastructure (spacey) headscale: headscale @@ -49,9 +48,7 @@ subdomains: # DATUM Gateway dashboard (knots-box, proxied via vipy) datum_gateway: datum -# Read by plays targeting managed, monitoring, vpn_control and edge - no one -# group covers them, so these are global rather than group_vars/. -ntfy_topic: alerts +# Read by plays across several groups, so global rather than group_vars/. headscale_namespace: counter-net # ───────────────────────────────────────────────────────────────────────────── diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index eafc633..068a490 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,157 +1,154 @@ $ANSIBLE_VAULT;1.1;AES256 -30373030336530356533636231303630303132666434393339663833366639366234313434333635 -3763366235343633303836383563333734643766346336390a383732313137623263343836616531 -35656262386162346434343462396166653435633330613366633263356638373061643833386336 -6438373137326563380a353962396135393365376133626538303862396236363330636564323538 -33616565363066633234633330386338316133613931613931653963356562323336396437313130 -33316331333166393161376635306237363033633535613761373762393761363165623462336466 -63663265376532663262323463643462336136373639316565396436653966336461656164316135 -64303235386236316136623063653133363961303733376166363462303565393761636632666230 -66333263626537343935383765303565626237346235356338373063633133326465653133366362 -62326563666531343166663563393932393938363663393132616561336363643037363735393233 -38336233346138356262616533643835396230663563623237616461626265666339613161616331 -34353263363065646336663538663561373536656364623061633863643137633230643931333662 -34386433326661336362636466386566636561623339383438626530333431313039623037616430 -62333832363531323363646238363733316237646133343337363434373465353463363935323233 -37326335343333636337626337643763323336623238636463623933646663356165363062383835 -30636630303534636436326530313030626565393439643238353433396638363133353865643135 -39666464323537383435636636303930396338333332363264306664383538666236616436353939 -61316564623964613839343533316438393337616363633033636231313937366139373031616236 -61373763303231376437313332343438376132633361353661336232333966313338373262383663 -36363931376132333763363332313534616133326437303637343139383861353138623465623736 -62643062396536633730663130353730356637353533616635343166643636616163343832363866 -65646266383062373635306239323565383039646334346535653962316434393365646461653366 -37396163396362313535393138663833653435353934623432616538663565353165363732316662 -35316332363030363963666136653437316639303766376431333738343061663835306339336337 -64346137313763656531306636333030323162646533633735646435613566616532363066353361 -32326439633432656633303732626361636238323735636631633732383635656436373435343931 -64356437326633326163306430306432326134373366616364656361383133656531653333633164 -37653934643061343237376664643331623633653437336534313261643765303134626536393938 -33393161363539313130376132313864666539363865393035626463313539376635393135633565 -35306435636363613735343136663939663435653135376337643034396634623135393039366265 -36326332386564303233623863363236356535323538616237303262613261363964643839363837 -63336536613139333939343964323835626433373464343730353864303532366535343636663362 -34333935316630323137366130356135643962373532346334333931366139356434316431393234 -61386632353434323432306130386239663931306538323335333231366234383132623936343465 -32393765316136663862663334616234326165336364316137396234343532643639366537343731 -33383938363530323337333838643835653238376532376431393439636266306634386233383937 -64623230353533643362633563663537343732646261353763366363343231376562336565333362 -65623665326136383764373761626461653238303335616463636333383235393037323530636630 -65333036653062386332653433366532383132376362356264373432653564316530656163396161 -34386133326133393638653961383032323830366232346134323663653436383034356334643137 -37663062396534616563363166333535336466373335613733323864363736633263646438336662 -39643333366565383130346430623637666132656330653833393836313566393461656638653938 -32623664363835376366353532646564643235356637363432363164353837643066643932383163 -34643232623734613738613638306664326333346439383830326662653934636536386166323239 -63626264613661336663316636353436643262353733303666663866346431353533353435346461 -34343763316364303061353030626431613434383062363737356131366234626137656331383433 -61653339616336376562326638343462343435613632656632353532376438623462656266326539 -35336465366133386534343466323936663934306363353462356530323531383961633336393234 -37366266343831386462626539306637623165363066616164393635373631393737613830363661 -34656530336634336531373830613039616231343238653637623532386538626365386364353930 -32396535643435306231386636363562623239393461393237303030353361633639626632653562 -31663261653663356266376264623762353261613163303430623738343939336438333532353436 -62343465353033343034373837376538323239653064623135613766343463373436663236616366 -30313062663861646331316362653230653139623937343538346230613832623965366666323365 -35623635646263396436346162333835343335623364393037353366343537336462636163303761 -37346663306339333264643034393630613833343163306430396333656637316662616439343136 -35336366306535333733633465343538383462616631326433666663303638633837313065656666 -34326136363433303264353331393133626639343166303364343065333266646439663463353833 -62346637363265643736616238663033666563326462633562643530383862616265306439376439 -63353432343534303433353138663461303165313433353866303838346338363761376333366138 -34353035356132383836313134643363636532353834323438313933346534333238646532326263 -30663964326564623164306664323265383134616165353634653261383931326663356137363636 -36383663666537373863623532376165343334633031626661306664663139616637636162373838 -32663863393765343431613536386133653731363266663566303665316435363830303035656230 -31366139636332336531326131346237653732313337343336393463373438373531653262363035 -39613739323832353332643030343861373836653064326139343965613563363166363236343733 -39653666333438613131396566623237643765393435313163333031383461303366396439353437 -31376562323231386133356231343366623363396437323866326234613034353664323163646631 -36386436653465306665646130616563646564343764346532363961663762303132336137356231 -61376565333036313033633732616430343266356435376434386266306432633633316361656538 -66393266383735336632653834373561636261663738333039653934636434343165353065353566 -34616431616430306234313032346235383065633734323632643065613634343834386664313336 -33333861306362333032346633636630326562323430323863373532613831313931303164303438 -36353831623262663030313564616634376235383666656434373261643464363264653831386534 -63663931643733313239323138393936373435303263663662613266306162373363353334336430 -62393835363263666136646631653738393432346331353538306666663864323233386665306332 -39353839313731313832333139613830396539306133363236653138643161323337356534643037 -38626636353465636166313965346364663238363032383537306532303332643839346230666564 -66613261396633346365633730633738306438326234386438373234393537353635626664616239 -35363737363537326132376638373839326139643135333234333764626162343766623531643265 -32656464383064626566653332323366306138373537663634333833613932666264306164666435 -39623335323163636135313361376231316231626335343765323261653134636439643561663463 -33376130356362316334386162306333333038316664636464653463313835356461386464653035 -38626230376339663361306161386332656230353737376133306464626466653038653266646139 -62313937383262633938633735393765616365646165396436653434303835666162373164333938 -61643432343237386133316433613030633638633731323936393562376139353033643562633835 -36393837383261663761393763613039623263643266303637613461626462313162613762383535 -33373064646533366136376563383663633331373161646534653330653566303332616262386564 -30313330346463643032363464346537613430633163306365313866373031393965383134323031 -63653238343338613033633336366235623332646239613235643637313666616263313833366362 -66666633393232303639363966306230663731353730333264663732386235653633303336316139 -33646263303164613437343735616362636364376261323364616131383433633864323132353063 -32646530323838633962613834663564326137663466623935323665343132653037386339613961 -61666363323763636361396330663537386630306637663638343565653437623966313738306366 -37633436373434396231396461656634316638393336653964396266363532343864346264323830 -32616461353630353730393639306231363662663034626434316661653936636262393761633934 -33393863383461316532373839663432336666356332393562393634386533346262306331643733 -62373162643763616637316233636163636363363639303662313630336233353162343437356533 -30303339666136343665313038643930383539313035373566336139643766353534386232343330 -32633836356435326332666266393237353162646537376564656236353138386531323032626636 -37373963666666346462343938656466343766323434373064363736386134656232636631653535 -38643739333230313265643132356430393035656662356465623836323038663237383439396137 -33383335356433383030393131666233663832643638646537363665396435323261613732303363 -63653634666164636632306462346166623237303936326435326431636630323562613233386462 -64376562316362653937646266366232653733663764633738303162616538656661326261643139 -30376233633835333637356134626266376566393962353639663039323439616639316465353038 -32373136643162633836663937373664306665643638366237653631313631393934643135393261 -35353338393761383431326631633166653932313335346165653364316431333831306135643339 -63316538343963336237653366343836343330656364373661343866656235303931633432313966 -35393631333263363533316537616132323038303230653838326266336434336532346439626537 -30656334316239356261386633393635643563336230326362303034633235323830343435646136 -63333237346164323763633665636234313263393366636233393538353335383933636331656261 -39313733383238613634363139326536616237353031666232663161653763326162386464626533 -39633931353035356561643564656637633163396561356430663636313231333962363765373536 -62613366616538306331623537613564353437366239616332626237623466613931653339333431 -39646239383438623063663063333861356536326637656337303561346432333539656431373733 -31316461343636313430303434336535353430353932373536366162333938646464653763356237 -39363938373065663732393862383031336438396164366664323130343734323130306662343138 -35383734663465303034393561306264316139623265386134323162373734393364393230333234 -31383336363832373962613535346136353037383065363066653935373435383037313764636266 -65383133333433326662626639356162303261393866373732366665353465356534333431633639 -39303539346266316132643537393564346562353533363438376363613036613739633939353564 -37313764316661663939396664376538626437366534383630613030323465396535366537346263 -31636338353233306333653962373763353439346337323839633238636336653439386362333533 -31653264343435633432646633386531386139366461646133396334326631333632393462666232 -35383735356232336335336435343932613433393530653633616466373965383935383861366136 -31303936616333396638326465633531373164626232356362613130376434303633366238336666 -37643035303932393265363037646430343866323163386234633739393938656530646533633163 -61653135633436653566303832383564663462353235353134313033616364636561643663383835 -30326561653934636164636363363736346630316263616664386233373535356339336239653333 -32336362363033616337663936623331663134656665643739626637353739666231643766346134 -63353631663631643436633935306261373939323530363639666366373531626231366138643766 -34623137643233333232303339323464333566633038333539356337306134383330353632383965 -34626461316261363830376664343336666132376365333635666131653933323562666635353763 -66376335663861623332623436333161353161336337313136613531646632363531383138343130 -39306330336239393965383939333765633532376539653661353965373030313038666161313265 -37653164343832333937356434646230623736646561623561643135626531343063626636663333 -30363937613834663434313038653739613933656534626532326136323336386164366339313830 -35396130386634666363656237373835313563343961633838383766663933376439326364616530 -64383136623739623137363562666431333565643130346166663531373738643761363337333536 -38313866363339393066643339653734316266393037396432356138303137343536313134623434 -62623362383863303539663264386430353436613261643865666534356538626538623630643766 -66646266336238396230613838626161336637313564303465663034653232306363306430633438 -35303364356531636565303965323539393233663566646562383863323461396166623032396338 -36346537343633643864323163623464623539616434376137313538646333666339626437656330 -31326463666335626361616566383065363264356663323435646337663566636436356336376531 -38636466386539663937326462636464613638653833666263333134396236313432643030366138 -66326364623435356633666463333066373238363864356230633634353764616331303163636266 -36306239323561326134393665643031663362636535656139376237383130616663316362326366 -61623236356562346233643332663261626333646638373262613664623935626266306230356164 -32336363383538333935333062386366653839303139366565343933346266623637363534393335 -31346431313631373535306138633930383536613636393034343434623664376237373066633036 -30323864613430333239326366613439613632633933656631356237333930303330326264646563 -38303066323563303262663530326637663964373039336431353339626436363335326466313262 -35386533633833363038636339663265626261373464633565666166633135333461363165353931 -63633332613966663133 +38396132363836373161663662323635613764643231316132333265396363353833383839613233 +3564323734323561366536303932636435373339363233650a343964613465353137396663666363 +63646331366162336537333035386263623264383566646334613332376561613231643235313565 +3638326434613462640a653364646361626636376464633734343332393030393232616362633532 +61313161643939383930303362363966326161313237346565323532626135643361633936313135 +38396534616239383735353532613834373564336537323531303533323434336166326263376238 +35303032336237666235396562313738363130326633336338306632353161336366346332393465 +61396163366432323335393362613539313661333431656465336662323336393330373332383834 +39636230313566663630656531323530653130643362363562326465643665326335323036346237 +32303439356432316564313730653332323731343235336535373462326162656638383735396135 +36633264613136323866643162353161306334666463303565333936653861643635336162623831 +65386231313465363864623035623364303039323961363237393637303133386530626266663435 +37336662643632666164386264646330363135623565306433646437333535363839663330663533 +64373231386231643863303933326132646364333339326431646531643962356663326363643734 +66346239663438363666613636336430336465623866643861666564353739623238376135346431 +38346234333364646230616632646561643638656535306630326332626136396538333534653061 +33643562326436636664343339643834323466313131323462653234303834373635313831353538 +35333739346563646163333937643232343866383337653462653530666165643361303332363364 +36326333333234343761386234666534323430323734616266326231386136336463636231346237 +33356361333562656666393630323530333731613964643966383065663635613734383066333532 +31366533373533656666666336363666313839316133333062356432633038366363623935373566 +33356563663933653539386263663530363239613461383831646262353165343638326663383364 +37386330356539343131353633633233323738396339336230666430363965353536343031646163 +38306162643733326563373835396264343433666335366136353439356333646131373764336438 +61393836303038306132383363633838333665343665303136316131616335643337313362306661 +37373137653830636635366266323538393338333463333963343931343433323437663661313366 +36616164346265626637616665663461336163343666623162663637633037666564363533623833 +35386537306366636533633930326339343136303839306539393866306462303632363163313535 +64326530323866313539623661636266663933343165653662373331663334376333393834303938 +30373233323664663935343364363731636334343065643662663266343366353031336439383766 +39313862616230346661313632613662613335666234333432336262663033633437383565383336 +35316365643030616331643333356361373236643731643733386237666430366566303933326563 +34336662623431333539616337303739373035313165396535393435383235313633616632616261 +37383633306235633939326339303861346335313361316438373630383863643864376462373464 +64663631656165316664633336333634643939333464303266366438633039653038356439393565 +35343237303533363032396365346232616131306666353766386433393737626639656139373332 +65346336373936393866666236346262623666666435363065386563613431626537636364396238 +64373133653132366538626262313434353930366338383866653139376538666136383363333666 +39636538323437616633356337626130356164363663333637333232663435303537336233643135 +64613638643564653431386634376538643564333438343835376633626230303264316337376434 +66316231386533313364343131316162623334373065626533303239383439326134343830376438 +34333831333035383830386436643239383138633166303033363563313562613564353538373536 +33343962346439616439366463383236373437613132326162323437343231666231396332623038 +37643932323538633365393033386462663735333034326332663461636363613932346637623638 +37323534613439373865646139656462303266373636356630306431323065656533663838303765 +30306534343965366566663337313263643832633861336334313835643738656334653135626563 +64353062343561626665613162346236393731623533663431333866613932653331613361333039 +65336337666264386334373039343965323065626334353030336336396136633363323339313031 +62633932613934623065306638656135393438383361376465653431316261633633353031323066 +32323466323861316338343137306333326137376434363735393664353761303265653165316634 +66623866373730646138333162663830363063326231653636643633623936313164363065396661 +64336263333234356263636630666264393734343732613231333331616230333662356164393539 +37616533356566323936386331623133666335613765303732313439376461363837623136386638 +38313666633436316336396337643666623735396562326538616564363066356136306632313138 +36643333346433366233636463626538313465666334373561666433396137323532366662393363 +62313434346636386566656437313130663836306265326463383264646564643538333738313436 +65343531356138353632346434616134316261306162636439666666326439373836333761333332 +30323836633438383166633364366365363561333731306435363433623333316665646334623061 +34396131666438613938343364663237393538313632356166346432346530386131623435643566 +30613233313739323562373332343537323665623330353135373735663035376138383931333938 +30363066303932643465633534663863336339356431393335343261333235396130616165636538 +32643335373137373638636537373635616133646333653230376633656561323533376136376564 +61306563653736656364383935326635363661636663626530313938646130356132373037373135 +64373034656232623530386435633838363262376464376361323161343163373561393839336165 +34323137363837616466326364376336646636396364393464626565623934636231643161343032 +61616663626130313764393735633866393566633533343737633035623138613662313335656136 +61303132333935613236336164396330383136303164633033356638643835313065393836326236 +36356235623838656534363639343261623939633864663631343363663262396231343562396530 +30653165663332626434636231303637356538666337643361653338303330633932643763323034 +66366538303939323232383631383562613732313065323933343761656535366134323164396432 +65663339366631623738323135633965623966613433376132363930383061613434386537373332 +39623136646638393835646163656364316665646464356535343839393966363631383734633536 +62393832346632323330356365393133346134653837663165353834313931356236373664656234 +39363431646239323665386639646165303836303731383063663266373236343436633163636535 +31393537653962303236383930613339613232643864356230616664313038396266333635396436 +36373930646435333863626333333439626663663930663730323236666233386231303730653132 +36613763303739376135653531326637313437623338343232303464303639633662373138326134 +33616632373031303139313439383465336565323837656632336532373737396230613837376564 +63663334363636666534616265623062366132376234326132393732336563313165386334626264 +34373934323930663832623338393964353964316132376262353662643534303436643537646361 +31663931613164646435656462613433356366633862636461633938343039353531633364373232 +31623131616534653533656566613038346133343266373637346663373739656161343765666636 +61656466636563643464616565643837623238636466316538636565626261313165346238306261 +62346264313739666338376439666633363033336435323932613664373234623931323962393035 +33313639626438656134363565396361613965376662386139366163383238346630623064343634 +38633962323232666636303830663538333737376437393465626338643632396238353863646263 +61363764346330663866396337626237653630633335363436613863613231356262383039613831 +62356465626564386537373134663536396231326136376234393966303065333264376630316633 +66303262383237386632656231663835383534326333633862383033356263306436623966383161 +65663530303836653235616466333736636463323065626335356333373334376337653637666133 +30363232383533646331323030303434633162643733623339346132666564396663343131303233 +62373431633635356235396130393330643463663239616365663930366138393930343334343432 +31616431666361353636316163313430633730366462616332333830353937616466343265343165 +36366265626363373164373463353561323437303936393337336238393137633931323537653666 +66346231363636326636333736383637376535393961303731386535386138633462323030346330 +64353936383162356433306164616264366164623731373733626261653238306362623264666633 +64316366393765356632323966343634653235636162353636306337383864326264343434613961 +65656433323164363738313535636637383365636435343933333265653333326536363864303866 +38396263626166316537326534363637393132363334616134653036663830343764303737383663 +31616639373330363365336231333938303138623165666531336336353133653263313263343231 +39666639383931653231373535396365386637643738346438333961393138633961376134303664 +39336331646236396634356337363336316537643133653561363937373963636238323837303833 +33396164656434666239393163363638373364383839633364393137313765613531626337316532 +33386465613034663630323638653839666264373235663364643461333238376138303765643364 +63623764363830356264306363613266393930376564616438636438373063323030633032326333 +66316235643036376165323035386164633536396365326333346638346133336235616132663532 +38396636376539656332613963646264316637616235316131316232623131386662333164333534 +31366534303166383062316263393833326165346333623830653266646637663364373365626335 +61623837323934343131353336393163303831613832303530356639393465633461326566363138 +39363730353862383365386339353536366537363933663566306132626135336339646533323763 +65616664393166666461643038336232303934306338323331366233643061373532623163663931 +37343839346637376337613438616131633239623234613539323138353233343238653661626462 +34353666643230633562323631303732373332616335326131396131653038643531646562646431 +33626130343834396439323138393162363831393534383839663638663838633061666534336138 +35313234653565316437336439376563666138653661613062643262656233316262343736363235 +32623466626366343965373638323239393933386434633531303363313061313832663166653634 +39333835303066303231363663666461323434363237646133323661386663633939313965363864 +38303861383565303766306533633132376632373234646338306264343065633631303935313266 +30333836616164363735363934386165663839393731633039383538623833353632373636656436 +62343934363834613464626364303434393262373365643130373437393730393964316130613565 +35373938306436666364646136306166663132393736383862623635316265386361663333356431 +35336363623535336631366531316661333130393034383566613138653834383333623437356139 +36353434626531616361333138386338636132646533326565336534643632316363313732336233 +38656134383565303132643161393032333334316265343862666532303438653161626665326632 +31343435393136313865356166363165613730373633623963303633343165613230393430633831 +31376361656361666434343866353037333834353735396566643935393439343334376461323133 +61663835656137356665623265376664336332313432636664313735643739626333383566336331 +35306539386433633636356636383734636633666137653334333162386564643464306538366461 +33323130656331313834323839303736366436633634356261363335343231303331353339313031 +61613563383937623162616133313633653833666634643566643766336533303566383561633237 +61316562353933333666323037376334326430623035643630323565386532313530383835343265 +61616165396664616332313262666534396537356564656236313864326533306338373362613034 +38623162336138366262626538366333373830623634616261353764666563663138663862656339 +65346137646263393064393536613963666434373663636336373531316464336264313061316439 +39623832666536356334383261653431343837663163623035366162396465646239613130643837 +64393735333438376262323834313330363434366163653833376334666666653535633330636535 +66346665643332393235313062313164333030353265386632326431333036333861356231666531 +63323966623361343363623561386633363734323864623961393832653037353766326637663735 +63396565326632303366396539396435626235643833323864663731396636313237373832646566 +66333839643734303734376338353037393266353232316264353736346338646262386636616335 +38313838363064643739633065343535643762626165343862646362613462386532356166383437 +38646130313462393863376339336436336630643061363130663831313532663466363936343934 +31656230323564303638653832626364633631336336613663356562636436643336323238393436 +63343462366436653365383766366262373332313339626431326538333562346438303730343937 +33326365636631633365336332636638353663303633623137643764303930656336323963643835 +34376136353532613938326432663835363530653264383561353533663863376661636137336437 +61353365333661623038363535643035393335613632656330333938623335343436663763646362 +63633130373530386235663661613335323531363234343465313861656639646461613936336134 +30613937376666623761626438396635313563306564633433613330633664356636636565613166 +32633430646139383362623133346435633063633466656635333034343937643639346536393134 +39383664383234643265323563646465373636303261646331366439666466343636333330633131 +37363963373361613331663263623737373464663831326630663432316336336163343866663834 +6630 diff --git a/ansible/host_vars/watchtower/main.yml b/ansible/host_vars/watchtower/main.yml deleted file mode 100644 index efdf7d6..0000000 --- a/ansible/host_vars/watchtower/main.yml +++ /dev/null @@ -1,13 +0,0 @@ ---- - -# Systemd services deployed on this host, monitored every 5 minutes. -# -# The fact lives with the machine rather than in a central map, for the same -# reason the cross-host ports do: "what runs here" is a property of the host, -# and a central list is one more thing to forget to update when a service moves. -# -# Only units WE deploy belong here. Distro units (ssh, cron) have their own -# supervision and would be noise. -monitored_services: - - caddy - - ntfy diff --git a/ansible/infra/402_public_monitoring.yml b/ansible/infra/402_public_monitoring.yml index 7ff7cf8..b66d10f 100644 --- a/ansible/infra/402_public_monitoring.yml +++ b/ansible/infra/402_public_monitoring.yml @@ -22,7 +22,6 @@ # ansible_host - if a box is renumbered, inventory is the one edit. dns_records: - {sub: "{{ subdomains.gatus }}", host: monitoring} - - {sub: "{{ subdomains.ntfy }}", host: watchtower} - {sub: "{{ subdomains.headscale }}", host: spacey} - {sub: "{{ subdomains.vaultwarden }}", host: vipy} - {sub: "{{ subdomains.forgejo }}", host: vipy} @@ -42,7 +41,6 @@ # there would go green precisely when the auth broke. public_sites: - {name: gatus, sub: "{{ subdomains.gatus }}", path: "/", status: 401} - - {name: ntfy, sub: "{{ subdomains.ntfy }}", path: "/", status: 200} - {name: headscale, sub: "{{ subdomains.headscale }}", path: "/health", status: 200} - {name: vaultwarden, sub: "{{ subdomains.vaultwarden }}", path: "/", status: 200} - {name: forgejo, sub: "{{ subdomains.forgejo }}", path: "/", status: 200} diff --git a/ansible/inventory.ini b/ansible/inventory.ini index 73ac21e..c39439d 100644 --- a/ansible/inventory.ini +++ b/ansible/inventory.ini @@ -1,6 +1,5 @@ [vps] vipy ansible_host=167.172.107.33 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua -watchtower ansible_host=164.92.239.72 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua spacey ansible_host=64.227.112.128 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua monitoring ansible_host=64.226.70.190 ansible_user=counterweight ansible_port=22 ansible_ssh_private_key_file=~/.ssh/counterganzua diff --git a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml index 9850e56..58f0348 100644 --- a/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml +++ b/ansible/services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml @@ -6,8 +6,17 @@ vars: ntfy_emergency_app_subdomain: "{{ subdomains.ntfy_emergency_app }}" ntfy_emergency_app_domain: "{{ ntfy_emergency_app_subdomain }}.{{ root_domain }}" - ntfy_service_domain: "{{ subdomains.ntfy }}.{{ root_domain }}" - ntfy_emergency_app_ntfy_url: "https://{{ ntfy_service_domain }}" + # ⚠ UNRESOLVED: this app sends its notifications to an ntfy server, and the + # server it points at ran on watchtower, which is being destroyed. This was + # derived from subdomains.ntfy, which is now gone with it. + # + # Until an ntfy server exists again this URL is dead, and the app fails + # silently at exactly the moment it matters - it is an EMERGENCY notifier. + # Three ways out, none of them automatic: + # * stand ntfy up somewhere else (the monitoring VPS, or vipy) + # * point this at the public ntfy.sh + # * retire the app + ntfy_emergency_app_ntfy_url: "{{ ntfy_service_url }}" ntfy_emergency_app_ntfy_user: "{{ ntfy_username | default('') }}" ntfy_emergency_app_ntfy_password: "{{ ntfy_password | default('') }}" diff --git a/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml b/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml index 59ae5d6..bfa4bd1 100644 --- a/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml +++ b/ansible/services/ntfy-emergency-app/ntfy_emergency_app_vars.yml @@ -14,3 +14,9 @@ remote_host: "{{ hostvars.get(remote_host_name, {}).get('ansible_host', remote_h remote_user: "{{ hostvars.get(remote_host_name, {}).get('ansible_user', 'counterweight') }}" remote_key_file: "{{ hostvars.get(remote_host_name, {}).get('ansible_ssh_private_key_file', '') }}" remote_port: "{{ hostvars.get(remote_host_name, {}).get('ansible_port', 22) }}" + +# Where the emergency notifications are sent. This pointed at ntfy.contrapeso.xyz +# on watchtower; that host is being destroyed, so this MUST be repointed before +# the app can work again. Left at the old value so the break is visible rather +# than silently defaulted to something plausible. +ntfy_service_url: "https://ntfy.contrapeso.xyz" diff --git a/ansible/services/ntfy/deploy_ntfy_playbook.yml b/ansible/services/ntfy/deploy_ntfy_playbook.yml deleted file mode 100644 index 88312d8..0000000 --- a/ansible/services/ntfy/deploy_ntfy_playbook.yml +++ /dev/null @@ -1,97 +0,0 @@ -- name: Deploy ntfy and configure Caddy reverse proxy - hosts: observability - become: yes - vars_files: - - ./ntfy_vars.yml - vars: - ntfy_subdomain: "{{ subdomains.ntfy }}" - ntfy_domain: "{{ ntfy_subdomain }}.{{ root_domain }}" - - tasks: - - name: Ensure /etc/apt/keyrings exists - file: - path: /etc/apt/keyrings - state: directory - mode: '0755' - - - name: Download and dearmor ntfy GPG key - shell: curl -fsSL https://archive.heckel.io/apt/pubkey.txt | gpg --dearmor -o /etc/apt/keyrings/archive.heckel.io.gpg - args: - creates: /etc/apt/keyrings/archive.heckel.io.gpg - - - name: Add ntfy APT repository - copy: - dest: /etc/apt/sources.list.d/archive.heckel.io.list - content: | - deb [arch=amd64 signed-by=/etc/apt/keyrings/archive.heckel.io.gpg] https://archive.heckel.io/apt debian main - mode: '0644' - - - name: Update APT cache - apt: - update_cache: yes - - - name: Install ntfy - apt: - name: ntfy - state: present - - - name: Ensure ntfy cache directories exist - file: - path: "{{ item }}" - state: directory - owner: ntfy - group: ntfy - mode: '0755' - loop: - - /var/cache/ntfy - - /var/cache/ntfy/attachments - - - name: Deploy ntfy configuration file - copy: - dest: /etc/ntfy/server.yml - content: | - base-url: "http://{{ ntfy_domain }}" - listen-http: ":{{ ntfy_port }}" - cache-file: "/var/cache/ntfy/cache.db" - attachment-cache-dir: "/var/cache/ntfy/attachments" - behind-proxy: true - auth-file: "/var/lib/ntfy/user.db" - auth-default-access: "deny-all" - owner: root - group: root - mode: '0644' - notify: Restart ntfy - - - name: Enable and start ntfy service - systemd: - name: ntfy - enabled: yes - state: started - - - name: Create ntfy admin user - shell: | - (echo "{{ ntfy_password }}"; echo "{{ ntfy_password }}") | ntfy user add --role=admin "{{ ntfy_username }}" - - - name: Publish ntfy through Caddy - ansible.builtin.include_role: - name: caddy_site - vars: - caddy_site_name: ntfy - caddy_site_domain: "{{ ntfy_domain }}, http://{{ ntfy_domain }}" - # Raw body: ntfy needs a plain-HTTP listener for its CLI/app clients, - # with only GETs to the docs and topic paths redirected to HTTPS. - caddy_site_body: | - reverse_proxy 127.0.0.1:{{ ntfy_port }} - - @httpget { - protocol http - method GET - path_regexp ^/([-_a-z0-9]{0,64}$|docs/|static/) - } - redir @httpget https://{host}{uri} - - handlers: - - name: Restart ntfy - systemd: - name: ntfy - state: restarted \ No newline at end of file diff --git a/ansible/services/ntfy/ntfy_vars.yml b/ansible/services/ntfy/ntfy_vars.yml deleted file mode 100644 index ba51792..0000000 --- a/ansible/services/ntfy/ntfy_vars.yml +++ /dev/null @@ -1,3 +0,0 @@ -ntfy_port: 6674 - -# ntfy_topic lives in group_vars/all/main.yml \ No newline at end of file diff --git a/ansible/site.yml b/ansible/site.yml index 8e9151e..10ee4a7 100644 --- a/ansible/site.yml +++ b/ansible/site.yml @@ -56,7 +56,6 @@ - import_playbook: services/vaultwarden/deploy_vaultwarden_playbook.yml - import_playbook: services/forgejo/deploy_forgejo_playbook.yml - import_playbook: services/lnbits/deploy_lnbits_playbook.yml -- import_playbook: services/ntfy/deploy_ntfy_playbook.yml - import_playbook: services/ntfy-emergency-app/deploy_ntfy_emergency_app_playbook.yml - import_playbook: services/personal-blog/deploy_personal_blog_playbook.yml @@ -80,9 +79,6 @@ # infra/nodito/30_proxmox_bootstrap one-shot: bare-metal bootstrap, run once # infra/nodito/33_..._cloud_template one-shot: builds the VM template # -# services/ntfy/setup_ntfy_uptime_ creates a notification channel INSIDE -# kuma_notification.yml Uptime Kuma. Kuma-specific tooling, not -# a deployment, and Kuma is being retired. # # services/vaultwarden/disable_ deliberate manual actions, not convergence # vaultwarden_sign_ups_playbook.yml From 3a9e1d58513c07dd0f3ea255e2d09cf5309e3d45 Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 21:40:37 +0200 Subject: [PATCH 66/67] alerting: Signal via signal-cli-rest-api, and faster failure detection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three changes: detection windows tightened, a Signal transport deployed, and alerts attached to all 84 non-transport endpoints. ── Failing faster ────────────────────────────────────────────────────────── The heartbeat is a TICKER, not a deadline: Gatus wakes every interval and asks "did anything arrive in the last interval", so real detection is 1-2x the window. And a window can only ever be as tight as the push frequency - which is why two checks moved rather than just having their numbers changed. liveness, cpu, ups, service-health, probes 16m -> 11m disk, zfs daily/30h -> 6-hourly/7h backup store + pull job daily/30h -> 6-hourly/7h DNS records 24h -> 6h backup dump 30h -> 26h backup-dump stays slow because the dump genuinely is daily. The store-side check catches the same fault within 6h by reading the source's dump timestamp out of the artefact filename, so 26h is a backstop rather than the primary signal. ── Thresholds differ by check type, deliberately ─────────────────────────── `failure-threshold` counts consecutive failures, but "consecutive" is a different amount of wall-clock time per check: a push endpoint produces one failure per heartbeat window, a pulled one per interval. The default of 3 would mean 33 minutes on an 11m heartbeat and over a day on a 7h one. More importantly the heartbeat window ALREADY encodes the tolerance - an 11m window on a 5-minute push is exactly "one missed push forgiven" - so stacking a threshold of 3 triples a tolerance that was already chosen. Hence: push/heartbeat endpoints failure-threshold 1 pulled, 5m (public) failure-threshold 3 (= 15 minutes) pulled, 6h/24h (dns, domain) failure-threshold 1 ── The Signal transport ──────────────────────────────────────────────────── roles/signal_api runs signal-cli-rest-api on the observability host, pinned by digest, MODE=native. It publishes NO PORTS. The API has no authentication of any kind - anything that reaches it can send messages as you and read your Signal. Gatus talks to it over a shared docker network by service name, which is also WHY the network exists: Gatus runs in a container, so the host's loopback is unreachable from it and a port published on 127.0.0.1 would not have worked. MODE=native and not json-rpc because this VPS has 464MB of RAM and already runs Gatus and Caddy. The json-rpc modes hold a resident JVM; native runs a binary per request, and alerts are rare enough that startup cost per alert is the right trade. Monitored - Gatus polls /v1/health over the same network path the alerts take, so it proves the delivery route rather than mere container liveness. Deliberately NOT backed up: the data directory holds Signal private keys and the recovery path is to link the device again from the phone. Neither the signal-api endpoint nor Gatus's self-check carries a Signal alert. If either is down, Signal is precisely what cannot deliver the alert. ── Four traps hit while linking, all now in the role README ──────────────── * /v1/qrcodelink is BROKEN in native mode - returns "no data to encode" while the binary itself emits a perfectly good URI. Not worked around by switching MODE, which would put a JVM in the path of every alert permanently. * The data dir must be owned by uid 1000, not root. A root-owned 0700 dir cannot be traversed by the container user, so linking silently never completes and /v1/accounts returns "Failed to read local accounts list". * `docker exec` runs as ROOT while the service runs as uid 1000, so without --config the account is written to /root/... on the container's ephemeral layer. It reports success and is destroyed on the next recreate. * The phone reporting "network error" was IPv6: chat.signal.org resolves to dualstack AAAA records first, the container has no IPv6 address, and this host's IPv6 path is broken - the same edge that 404'd the Go tarball. Fixed with a mounted gai.conf that prefers IPv4. Verified: provider loads (configuredProviders=[signal]), a test message was delivered and confirmed received, 84 endpoints carry alerts, 86 UP / 0 DOWN. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/group_vars/all/vault.yml | 316 +++++++++--------- ansible/infra/400_host_monitoring.yml | 61 +++- ansible/infra/401_service_monitoring.yml | 28 +- ansible/infra/402_public_monitoring.yml | 29 +- .../infra/403_service_probe_registration.yml | 26 +- ansible/playbooks/backups.yml | 18 +- ansible/roles/backup_store/defaults/main.yml | 9 +- ansible/roles/gatus/defaults/main.yml | 6 + ansible/roles/gatus/tasks/configure.yml | 10 + .../gatus/templates/docker-compose.yml.j2 | 9 + .../roles/gatus_endpoint/defaults/main.yml | 11 + .../templates/endpoints.yaml.j2 | 10 +- ansible/roles/signal_api/README.md | 133 ++++++++ ansible/roles/signal_api/defaults/main.yml | 39 +++ ansible/roles/signal_api/tasks/main.yml | 93 ++++++ .../templates/docker-compose.yml.j2 | 47 +++ .../roles/signal_api/templates/gai.conf.j2 | 17 + .../services/gatus/deploy_gatus_playbook.yml | 23 ++ .../signal-api/deploy_signal_api_playbook.yml | 44 +++ ansible/site.yml | 4 + 20 files changed, 752 insertions(+), 181 deletions(-) create mode 100644 ansible/roles/signal_api/README.md create mode 100644 ansible/roles/signal_api/defaults/main.yml create mode 100644 ansible/roles/signal_api/tasks/main.yml create mode 100644 ansible/roles/signal_api/templates/docker-compose.yml.j2 create mode 100644 ansible/roles/signal_api/templates/gai.conf.j2 create mode 100644 ansible/services/signal-api/deploy_signal_api_playbook.yml diff --git a/ansible/group_vars/all/vault.yml b/ansible/group_vars/all/vault.yml index 068a490..76eb7bf 100644 --- a/ansible/group_vars/all/vault.yml +++ b/ansible/group_vars/all/vault.yml @@ -1,154 +1,164 @@ $ANSIBLE_VAULT;1.1;AES256 -38396132363836373161663662323635613764643231316132333265396363353833383839613233 -3564323734323561366536303932636435373339363233650a343964613465353137396663666363 -63646331366162336537333035386263623264383566646334613332376561613231643235313565 -3638326434613462640a653364646361626636376464633734343332393030393232616362633532 -61313161643939383930303362363966326161313237346565323532626135643361633936313135 -38396534616239383735353532613834373564336537323531303533323434336166326263376238 -35303032336237666235396562313738363130326633336338306632353161336366346332393465 -61396163366432323335393362613539313661333431656465336662323336393330373332383834 -39636230313566663630656531323530653130643362363562326465643665326335323036346237 -32303439356432316564313730653332323731343235336535373462326162656638383735396135 -36633264613136323866643162353161306334666463303565333936653861643635336162623831 -65386231313465363864623035623364303039323961363237393637303133386530626266663435 -37336662643632666164386264646330363135623565306433646437333535363839663330663533 -64373231386231643863303933326132646364333339326431646531643962356663326363643734 -66346239663438363666613636336430336465623866643861666564353739623238376135346431 -38346234333364646230616632646561643638656535306630326332626136396538333534653061 -33643562326436636664343339643834323466313131323462653234303834373635313831353538 -35333739346563646163333937643232343866383337653462653530666165643361303332363364 -36326333333234343761386234666534323430323734616266326231386136336463636231346237 -33356361333562656666393630323530333731613964643966383065663635613734383066333532 -31366533373533656666666336363666313839316133333062356432633038366363623935373566 -33356563663933653539386263663530363239613461383831646262353165343638326663383364 -37386330356539343131353633633233323738396339336230666430363965353536343031646163 -38306162643733326563373835396264343433666335366136353439356333646131373764336438 -61393836303038306132383363633838333665343665303136316131616335643337313362306661 -37373137653830636635366266323538393338333463333963343931343433323437663661313366 -36616164346265626637616665663461336163343666623162663637633037666564363533623833 -35386537306366636533633930326339343136303839306539393866306462303632363163313535 -64326530323866313539623661636266663933343165653662373331663334376333393834303938 -30373233323664663935343364363731636334343065643662663266343366353031336439383766 -39313862616230346661313632613662613335666234333432336262663033633437383565383336 -35316365643030616331643333356361373236643731643733386237666430366566303933326563 -34336662623431333539616337303739373035313165396535393435383235313633616632616261 -37383633306235633939326339303861346335313361316438373630383863643864376462373464 -64663631656165316664633336333634643939333464303266366438633039653038356439393565 -35343237303533363032396365346232616131306666353766386433393737626639656139373332 -65346336373936393866666236346262623666666435363065386563613431626537636364396238 -64373133653132366538626262313434353930366338383866653139376538666136383363333666 -39636538323437616633356337626130356164363663333637333232663435303537336233643135 -64613638643564653431386634376538643564333438343835376633626230303264316337376434 -66316231386533313364343131316162623334373065626533303239383439326134343830376438 -34333831333035383830386436643239383138633166303033363563313562613564353538373536 -33343962346439616439366463383236373437613132326162323437343231666231396332623038 -37643932323538633365393033386462663735333034326332663461636363613932346637623638 -37323534613439373865646139656462303266373636356630306431323065656533663838303765 -30306534343965366566663337313263643832633861336334313835643738656334653135626563 -64353062343561626665613162346236393731623533663431333866613932653331613361333039 -65336337666264386334373039343965323065626334353030336336396136633363323339313031 -62633932613934623065306638656135393438383361376465653431316261633633353031323066 -32323466323861316338343137306333326137376434363735393664353761303265653165316634 -66623866373730646138333162663830363063326231653636643633623936313164363065396661 -64336263333234356263636630666264393734343732613231333331616230333662356164393539 -37616533356566323936386331623133666335613765303732313439376461363837623136386638 -38313666633436316336396337643666623735396562326538616564363066356136306632313138 -36643333346433366233636463626538313465666334373561666433396137323532366662393363 -62313434346636386566656437313130663836306265326463383264646564643538333738313436 -65343531356138353632346434616134316261306162636439666666326439373836333761333332 -30323836633438383166633364366365363561333731306435363433623333316665646334623061 -34396131666438613938343364663237393538313632356166346432346530386131623435643566 -30613233313739323562373332343537323665623330353135373735663035376138383931333938 -30363066303932643465633534663863336339356431393335343261333235396130616165636538 -32643335373137373638636537373635616133646333653230376633656561323533376136376564 -61306563653736656364383935326635363661636663626530313938646130356132373037373135 -64373034656232623530386435633838363262376464376361323161343163373561393839336165 -34323137363837616466326364376336646636396364393464626565623934636231643161343032 -61616663626130313764393735633866393566633533343737633035623138613662313335656136 -61303132333935613236336164396330383136303164633033356638643835313065393836326236 -36356235623838656534363639343261623939633864663631343363663262396231343562396530 -30653165663332626434636231303637356538666337643361653338303330633932643763323034 -66366538303939323232383631383562613732313065323933343761656535366134323164396432 -65663339366631623738323135633965623966613433376132363930383061613434386537373332 -39623136646638393835646163656364316665646464356535343839393966363631383734633536 -62393832346632323330356365393133346134653837663165353834313931356236373664656234 -39363431646239323665386639646165303836303731383063663266373236343436633163636535 -31393537653962303236383930613339613232643864356230616664313038396266333635396436 -36373930646435333863626333333439626663663930663730323236666233386231303730653132 -36613763303739376135653531326637313437623338343232303464303639633662373138326134 -33616632373031303139313439383465336565323837656632336532373737396230613837376564 -63663334363636666534616265623062366132376234326132393732336563313165386334626264 -34373934323930663832623338393964353964316132376262353662643534303436643537646361 -31663931613164646435656462613433356366633862636461633938343039353531633364373232 -31623131616534653533656566613038346133343266373637346663373739656161343765666636 -61656466636563643464616565643837623238636466316538636565626261313165346238306261 -62346264313739666338376439666633363033336435323932613664373234623931323962393035 -33313639626438656134363565396361613965376662386139366163383238346630623064343634 -38633962323232666636303830663538333737376437393465626338643632396238353863646263 -61363764346330663866396337626237653630633335363436613863613231356262383039613831 -62356465626564386537373134663536396231326136376234393966303065333264376630316633 -66303262383237386632656231663835383534326333633862383033356263306436623966383161 -65663530303836653235616466333736636463323065626335356333373334376337653637666133 -30363232383533646331323030303434633162643733623339346132666564396663343131303233 -62373431633635356235396130393330643463663239616365663930366138393930343334343432 -31616431666361353636316163313430633730366462616332333830353937616466343265343165 -36366265626363373164373463353561323437303936393337336238393137633931323537653666 -66346231363636326636333736383637376535393961303731386535386138633462323030346330 -64353936383162356433306164616264366164623731373733626261653238306362623264666633 -64316366393765356632323966343634653235636162353636306337383864326264343434613961 -65656433323164363738313535636637383365636435343933333265653333326536363864303866 -38396263626166316537326534363637393132363334616134653036663830343764303737383663 -31616639373330363365336231333938303138623165666531336336353133653263313263343231 -39666639383931653231373535396365386637643738346438333961393138633961376134303664 -39336331646236396634356337363336316537643133653561363937373963636238323837303833 -33396164656434666239393163363638373364383839633364393137313765613531626337316532 -33386465613034663630323638653839666264373235663364643461333238376138303765643364 -63623764363830356264306363613266393930376564616438636438373063323030633032326333 -66316235643036376165323035386164633536396365326333346638346133336235616132663532 -38396636376539656332613963646264316637616235316131316232623131386662333164333534 -31366534303166383062316263393833326165346333623830653266646637663364373365626335 -61623837323934343131353336393163303831613832303530356639393465633461326566363138 -39363730353862383365386339353536366537363933663566306132626135336339646533323763 -65616664393166666461643038336232303934306338323331366233643061373532623163663931 -37343839346637376337613438616131633239623234613539323138353233343238653661626462 -34353666643230633562323631303732373332616335326131396131653038643531646562646431 -33626130343834396439323138393162363831393534383839663638663838633061666534336138 -35313234653565316437336439376563666138653661613062643262656233316262343736363235 -32623466626366343965373638323239393933386434633531303363313061313832663166653634 -39333835303066303231363663666461323434363237646133323661386663633939313965363864 -38303861383565303766306533633132376632373234646338306264343065633631303935313266 -30333836616164363735363934386165663839393731633039383538623833353632373636656436 -62343934363834613464626364303434393262373365643130373437393730393964316130613565 -35373938306436666364646136306166663132393736383862623635316265386361663333356431 -35336363623535336631366531316661333130393034383566613138653834383333623437356139 -36353434626531616361333138386338636132646533326565336534643632316363313732336233 -38656134383565303132643161393032333334316265343862666532303438653161626665326632 -31343435393136313865356166363165613730373633623963303633343165613230393430633831 -31376361656361666434343866353037333834353735396566643935393439343334376461323133 -61663835656137356665623265376664336332313432636664313735643739626333383566336331 -35306539386433633636356636383734636633666137653334333162386564643464306538366461 -33323130656331313834323839303736366436633634356261363335343231303331353339313031 -61613563383937623162616133313633653833666634643566643766336533303566383561633237 -61316562353933333666323037376334326430623035643630323565386532313530383835343265 -61616165396664616332313262666534396537356564656236313864326533306338373362613034 -38623162336138366262626538366333373830623634616261353764666563663138663862656339 -65346137646263393064393536613963666434373663636336373531316464336264313061316439 -39623832666536356334383261653431343837663163623035366162396465646239613130643837 -64393735333438376262323834313330363434366163653833376334666666653535633330636535 -66346665643332393235313062313164333030353265386632326431333036333861356231666531 -63323966623361343363623561386633363734323864623961393832653037353766326637663735 -63396565326632303366396539396435626235643833323864663731396636313237373832646566 -66333839643734303734376338353037393266353232316264353736346338646262386636616335 -38313838363064643739633065343535643762626165343862646362613462386532356166383437 -38646130313462393863376339336436336630643061363130663831313532663466363936343934 -31656230323564303638653832626364633631336336613663356562636436643336323238393436 -63343462366436653365383766366262373332313339626431326538333562346438303730343937 -33326365636631633365336332636638353663303633623137643764303930656336323963643835 -34376136353532613938326432663835363530653264383561353533663863376661636137336437 -61353365333661623038363535643035393335613632656330333938623335343436663763646362 -63633130373530386235663661613335323531363234343465313861656639646461613936336134 -30613937376666623761626438396635313563306564633433613330633664356636636565613166 -32633430646139383362623133346435633063633466656635333034343937643639346536393134 -39383664383234643265323563646465373636303261646331366439666466343636333330633131 -37363963373361613331663263623737373464663831326630663432316336336163343866663834 -6630 +34383033613438623432656438313763613639313139626432303637383363643339613166626665 +6537663336353661373437303636613364643736663031300a623461353135623537343763383333 +37323731646262613436646231656263396532616132316237353762376363383062646132303435 +3435316666316235660a323163613932616661303935336633656332353766393536393736356539 +61396164386266303464373262373862363361363365363235383363613335663933313062663565 +39613466353161356462656539613335613536363532333431393935303430373435373737363537 +62383266323339373062313134316464303263313830646161333530383736336637333865373764 +31313165396366653763343466643561383236643836616662383439666165343139326232326364 +32633133373863616163343365656231353939366435346534383462396134653064663566343566 +66343464373865383864313264646635316233336133346365396634626261613561373965643339 +30353831393037316334313632376537393361336366326561343832373531303139323937613531 +34386164303430316162363536386137316330336638303365393462386465633862663932376563 +34323862636435306563333065383134343362333733623066656439613339376666353331373665 +32353938663364313036373237323062393430306661633732356434333232376338313835663330 +32306536393666323464663238306562613936376236613935616664653865373330626365626330 +64613066653632646638306338393331376430383165316363323437373362613866313962663432 +62356663636630356239613831353066303530363966396135376134633936313530643262353965 +38383330613962313935366633353635363638303937316362303866343433353437363334396431 +32393134626133353564643233616632376633303934653065613262353436333533376137303862 +35303861383833326537376433336634633730373833313036643365373431666438653033653930 +33346230396661636333613031353133353631306331666362386366633062666534303539316238 +39366634656333646634643032393738333330636230663366623265666533356534623465316665 +35336535326237343231653661613736626239316262386363666264613636326332366232333964 +38353635616662663133376534306366643037613732343233373336653166316330643638353539 +35353532643534613864333839393365313439343663333337373639633832383630316164306132 +63636162373034363839656638353534613733663737626432636436373164303239316436313566 +66623466396663353136323565643262383865323830616232373466373431373261306537646463 +61353966623062356438626635663461396165343366393132393965383734633632666335313030 +37333631373239303231396264393031373162373936303462363736313538343562343737356332 +63663762633032356130303964386664366137373432353533386566613037613163383963623732 +32343765376535343630656633333362393765666131653632613361363163383038323138333537 +37373333316466316562613439356365316163336366623435666166653835616531653563343664 +30303464663838646333373837363961356465613234636138633162366636633030383337643131 +62636236663066316630396132333934353139383465363034363033656161383663656462313939 +62353738396132343734343363636164323163616237613861643862303433633934656165343263 +35313737373333386335663731643664393630653839396464376639306231653934386336353630 +61393735336662663435613431386361373561363531643232303831363163636139653538393931 +66313965646562616237383439343637623835303065333730613865666638383131616261346463 +31396464306537613135653830653138323539393731626264326335336432666333333735646534 +30616439393464326631633635363466303336613135346231626232313361303666323661616638 +64363937343231363536353034363966326132333734386638613737636130646232363666643633 +61353563636466356630346232613761636432303430336461663434333636623962643764336639 +35303565613861323431393036303534303061666437366538306236373930313439353330313632 +61363363303562316664303065613863613339626632643931386438326330373938613762303334 +33316430363931393262373661623137633835656136313235613666613932313236343966306331 +36343031303632323432366337633637336335343638393564313738386162386164613339343662 +62383939663265373265613932633265626263623939303638383838646531343433393864363235 +37313761396436623833376435393537376136373162393465343764326533666533393061643965 +61306365346562333737313962633764313232323161623861336566343735353737633539656162 +34323939306661373662663331653333346264323930663636633134666532623438323537383165 +66343033363962633766303331623866306462623235373838616565653066646132633034363535 +31613462356361613535383963643662373162363363653334393937313039666266366363653537 +64356166396139393262313565303731346534646462333638316661383139346362333364613466 +38303561316263636136303431393366343936653161336238613439366266346136636465303337 +61373634376464333037303862616335623031656133636165636264386265643261373735663266 +39396265366138363430343035356337303438346165366361316230656239326633653537626339 +39613334316563376133343336343563363564653237633764393532336437303334363830396635 +31333963666462313830333163646465356337643263636462356363613266323630393866626535 +35336436663063333364353663616564626535396133386461326433323936323161613633353539 +66323935626433353162666335613839346561343264323763303034613366663233613037343538 +66376236373462383430313030373235373337653338333033306431616562353435656433356365 +37653063363435366532306334373438306632306435653334396630343863623938666335656164 +37386634366430383762316435396237313362313965353634613266616532633465646464313662 +35303433363236393465396161346431306231666432666531626136623966663864376663373562 +37653239633732346536323265643434616136326561666330633435383264333937353366653736 +61353430633363323263323163646639663330663632343038623964396433626337353232373334 +32343938356539303938666137323831646562393233323733336464316165623232326132316362 +61313631626234336164656533616231393663363364623130313439623466396231383537656562 +35393163323131383838356130643863633736613635306635306135666261623731613239303962 +61373432336638353339646665653561653738353731373032366361313562306466373365653333 +35306236336331333961613864346165613763383266613236343066623432303766333434333331 +66376263363132616132366165393235626336323665373135653462653533376138346632393363 +35396165383966326635333134663138366136336430393532373935373264616530613762386430 +37643862316232613261343430616438333831623835663239656531356666613032653837373531 +30386232643537306530323936386561646265616339376265353763663833623539653831656634 +63626564633964343663656435636635656562666236386639353332356566316631643431353330 +32643238623932643732353332633535643738663066383830376432346463393464333236386163 +64376165333266323166613564313061643832656239333165393035343836376434626162643061 +31373861393534636236383765313834356638613332626666653762343537363339383939666263 +61366230623930643534353565316165656132396664393138326139613536343738396330653735 +37623463343834626133313231393130646134363331393238663065643930383237326439343831 +32393536613361616637373761393361303936363863333838666633343063386235316237323531 +62633362316436623761616366383336613765373362393161303765613464643632666462343565 +63616430396164346263666436336463623638313130656231643131336663336565616135373433 +39326533643733333564373434333738343634383033353036646534646632383730613161393130 +62653836376435663435353739373536366466646332653030653332353761626466353631663163 +38613734383634663662376464653733306564313761336566336336656366653536643834653631 +35333963306562393234623039613338643266333762333236366331333431343830393864623830 +34323439386133613035363861623332663933313734313637623739613163663730316134303434 +35363239363836666463366334356435666632396132306163303037663430383535393934323537 +30353266373864346662333035393761646632313032383738346439343232663666343833333565 +35363564623038376666363436663830356165323234653230636264303066666230643632356237 +66633734346633366533623966393462313438333965623033343337323135353730613561363236 +30636333613034626162393137323762306432663761393766303163393536656338366664643432 +34636630323965396166356538623165663335333739323034333237306531393734306138343436 +36613937373932633234303336373530663533646635396664343935326334333239646534343336 +35396666353938326337646431353965653836323731313861613031356561376262363965623934 +65336164623134366365303062393161353965363937646363396564313138323938656330333964 +64663862363766613630333531623463616331363533343962653164376665303463336134346662 +66376238663632663932656230323638323135653666363736643065336330636236323064386361 +37646631303138616565383665663634333338613336346438366333636165366635313337373635 +62656236653961393061373262326534393339363236353336613339656462336661663739396261 +35363636653237316436656662633462306333313336313865333037636661383731336466666536 +36646537653431666261353232653233653930346538383438633466346636656432383963333765 +62346363633265306665366630383831303434336139343837616361666539656262633735363064 +30376138386531336430613563366666333964663231363465326230316233323532366632663930 +63653164373931353861653833326130643365373530663039363166366331303134366336613438 +63633336363539636265396331653630633162316366316130326531363463616464376635376538 +32376638353337343639633434613663666437313564333264383766623561393933333561663235 +37666163626538353736383435373530663438363636643665626266316338336532623235353337 +61653762383630616266353438386137653137616532633066623165383663646437343565366237 +33333463613539346136613135386466656135616136663036313661323738363066633038656139 +38353230313861666463383737303038613731316334333031643438343463666164633536313031 +39613663376436643165333362636462303230656336333234363336633262336363643736336231 +66316630613762383832613836333263303038306235663662323162633932653733626130356665 +65316630346534356261373961353130393334396365376237353439326261336339393336303030 +63323439666439313831626639323735653730356565643938376564316337613333616664643865 +30633530383865663731383635366566613565393632336337306334663937396135303539363536 +65656331386532306438383031313164626237393762303636663831666239336637666465633630 +63383961336333663537666561376532376631343132363638633364616531306363623432396332 +31343864626335613131653762363338393738343531626330663663376531336230656365363661 +64363530333965386439376161313534333435393966656231613161633034303964653564363436 +63396131623562303962383137666439373132633330363237353031306561306362666561376435 +66383535333233663736363762323662363264393365643135323834366565653932313863353437 +38363536323739336165356561613335653464363637636234313637383066633865353239393961 +31613532373361383062393238303430353662383930373764373237383035633364323661623661 +62653661623234366666616532663563316234663137616338393036333937633963323635366165 +61613032363531666338353132346365323661626265626461396362646338373536306362303338 +31646564343434633433663039366533656665643235336136333134366231376336343465336539 +37363139346239666232313366613432646265626564313838346364306133306434313337626562 +32346430316532623037633838303363346130623636386134313566383333613565316135373264 +38393261313435323433326130303032393538333963646430646366653363613830613534353832 +32643434613964613239363966376138376661636335373130356430336631333461623734383535 +35356461326136636234663738343738303064386536336462303632303461303733333666383036 +38383563393739343134363038623638393766316533373163336439336238613734656132623163 +32626564666430663339316163393934323133306238353562323866633738363737643937376435 +66306430623334643766623564363239346361393666663766306637313265313833396435626234 +35366239616132623863343365663833363934316362636638303536656631633364646235366566 +39386637613735373761643339396132323031323438316633363464336636316534643435323861 +62663732366430646538626663313035616235643537643234356434323635373962343633336266 +30323031306438656537626430356231393466313334633934623632663661306361313330633339 +33396234386536336331616161363132343765306562303932313963623037366633383765663134 +62636466636433666530363930633765626531613539363832313361666565323635326334326533 +36336439363836663033643262336139613437623633616138656564393032393263393132303031 +31626336663039633362366663306662333432643261613464303939326562653562346239383261 +37373966383635626433363831653936333964626262363839383936356634633233336365623765 +62313031353538636464303234383865613932323164386336316362323731303263346637346534 +32353734343130343364646161643431353230336364366330363261326334613936633234646264 +39623163346663613431323630303034383761663835373565663166366239303130386139396236 +62393565353761636636366462353038366664663430616332326465373364396264323064666536 +61653035633562316634336263636332363733663666626131383161656535383133613939623034 +66363632613733306336323165336534346337386163656631343332323636353539353737316131 +34616336343865656565646338653037333838663736333330376330663834373138363739633064 +61646439636365323838376131663266636235333062616532613936616339633661303634323531 +63356335386534353639323566376134373565353137333134363761666532366431633634333430 +39326662633765626230326130386333393463323433366162363432613234336634663439313339 +33653737333830336264343839396563316462613032376335356634383436353962613636333866 +32333032623635323430356636623138366635386533646133626164316438393937626462313239 +31636635363064363763396566306234363965346438653738333961623435303233396634643763 +65343763386464366636383734336335646464306262623363303934393734333666636635356365 +65353931303862343765343665373139376530656263356365366333323135383838383030613166 +30303963383833643937636539656164336537383337346436303534303035313935323633663765 +333464336235373230316337333632326138 diff --git a/ansible/infra/400_host_monitoring.yml b/ansible/infra/400_host_monitoring.yml index ac81289..3f6b119 100644 --- a/ansible/infra/400_host_monitoring.yml +++ b/ansible/infra/400_host_monitoring.yml @@ -34,6 +34,23 @@ # slow apt run, a reboot - does not raise an alarm, but a check that has # genuinely stopped does. # ───────────────────────────────────────────────────────────────────────────── +# ───────────────────────────────────────────────────────────────────────────── +# Alerting thresholds, and why they differ by check type. +# +# `failure-threshold` counts CONSECUTIVE failures, but "consecutive" means a +# different amount of wall-clock time per check: +# +# push/heartbeat endpoints a failure is produced once per heartbeat window +# pulled endpoints a failure is produced once per interval +# +# So the default of 3 would mean 33 minutes on an 11m heartbeat and over a day +# on a 7h one - and the heartbeat window ALREADY encodes the tolerance. An 11m +# window on a 5-minute push is precisely "one missed push forgiven"; stacking a +# threshold of 3 on top triples a tolerance that was already chosen. +# +# Hence: push endpoints alert on the FIRST heartbeat failure. Pulled endpoints +# have no built-in tolerance, so the threshold is where it belongs for them. +# ───────────────────────────────────────────────────────────────────────────── - name: Register the host checks with Gatus hosts: observability become: yes @@ -47,7 +64,7 @@ 'name': item, 'group': 'liveness', 'token': gatus_push_tokens[item], - 'heartbeat': '16m'}] }}" + 'heartbeat': '11m'}] }}" loop: "{{ monitored }}" - name: Build the disk endpoint list @@ -56,13 +73,20 @@ 'name': item, 'group': 'disk', 'token': gatus_push_tokens[item], - 'heartbeat': '30h'}] }}" + 'heartbeat': '7h'}] }}" loop: "{{ monitored }}" - name: Register liveness endpoints ansible.builtin.include_role: name: gatus_endpoint vars: + gatus_endpoint_default_alerts: + - type: signal + # 1, not 3: the heartbeat window is the tolerance. See the note above. + failure-threshold: 1 + success-threshold: 2 + send-on-resolved: true + minimum-reminder-interval: 6h gatus_endpoint_name: liveness gatus_endpoint_external: "{{ liveness_endpoints }}" @@ -70,6 +94,13 @@ ansible.builtin.include_role: name: gatus_endpoint vars: + gatus_endpoint_default_alerts: + - type: signal + # 1, not 3: the heartbeat window is the tolerance. See the note above. + failure-threshold: 1 + success-threshold: 2 + send-on-resolved: true + minimum-reminder-interval: 6h gatus_endpoint_name: disk gatus_endpoint_external: "{{ disk_endpoints }}" @@ -77,20 +108,27 @@ ansible.builtin.include_role: name: gatus_endpoint vars: + gatus_endpoint_default_alerts: + - type: signal + # 1, not 3: the heartbeat window is the tolerance. See the note above. + failure-threshold: 1 + success-threshold: 2 + send-on-resolved: true + minimum-reminder-interval: 6h gatus_endpoint_name: hypervisor gatus_endpoint_external: - name: cpu group: hypervisor token: "{{ gatus_push_tokens['nodito'] }}" - heartbeat: "16m" + heartbeat: "11m" - name: zfs group: hypervisor token: "{{ gatus_push_tokens['nodito'] }}" - heartbeat: "30h" + heartbeat: "7h" - name: ups group: hypervisor token: "{{ gatus_push_tokens['nodito'] }}" - heartbeat: "16m" + heartbeat: "11m" - name: Deploy host liveness and disk checks hosts: managed @@ -120,10 +158,13 @@ healthcheck_name: disk-usage healthcheck_description: "Disk usage for {{ inventory_hostname }}" healthcheck_check: disk-usage - # Daily. RandomizedDelaySec spreads twelve hosts across the hour rather - # than having them all report in the same second. - healthcheck_on_calendar: "*-*-* 07:00:00" - healthcheck_randomized_delay: "3600" + # Every 6h rather than daily. Disk usage itself moves slowly, but the + # heartbeat can only be as tight as the push frequency - a daily push + # forces a >24h window, and a stuck check then hides for a day and a + # half. Six-hourly buys a 7h window. RandomizedDelaySec spreads the + # hosts so twelve boxes do not all report in the same second. + healthcheck_on_calendar: "*-*-* 00/6:00:00" + healthcheck_randomized_delay: "900" healthcheck_boot_delay: "5min" healthcheck_push_url: "{{ gatus_api }}/disk_{{ host_key }}/external" healthcheck_push_token: "{{ host_token }}" @@ -157,7 +198,7 @@ healthcheck_check: zfs-health healthcheck_packages: [curl, jq] healthcheck_zfs_pool: "{{ zfs_pool_name }}" - healthcheck_on_calendar: "*-*-* 07:20:00" + healthcheck_on_calendar: "*-*-* 00/6:20:00" healthcheck_boot_delay: "10min" healthcheck_push_url: "{{ gatus_api }}/hypervisor_zfs/external" healthcheck_push_token: "{{ host_token }}" diff --git a/ansible/infra/401_service_monitoring.yml b/ansible/infra/401_service_monitoring.yml index f07a906..ab564a1 100644 --- a/ansible/infra/401_service_monitoring.yml +++ b/ansible/infra/401_service_monitoring.yml @@ -1,7 +1,7 @@ --- # Is each systemd-deployed service actually running? # -# Every 5 minutes, with a 16-minute Gatus heartbeat - three missed runs before +# Every 5 minutes, with an 11-minute Gatus heartbeat - one missed run before # it alarms, so a reboot or a slow check does not page anyone, but a host that # stops reporting does. # @@ -28,6 +28,23 @@ # Register one endpoint per unit. Runs first: Gatus reloads within 30s, and the # host play above takes minutes, so every endpoint exists before its first push. # ───────────────────────────────────────────────────────────────────────────── +# ───────────────────────────────────────────────────────────────────────────── +# Alerting thresholds, and why they differ by check type. +# +# `failure-threshold` counts CONSECUTIVE failures, but "consecutive" means a +# different amount of wall-clock time per check: +# +# push/heartbeat endpoints a failure is produced once per heartbeat window +# pulled endpoints a failure is produced once per interval +# +# So the default of 3 would mean 33 minutes on an 11m heartbeat and over a day +# on a 7h one - and the heartbeat window ALREADY encodes the tolerance. An 11m +# window on a 5-minute push is precisely "one missed push forgiven"; stacking a +# threshold of 3 on top triples a tolerance that was already chosen. +# +# Hence: push endpoints alert on the FIRST heartbeat failure. Pulled endpoints +# have no built-in tolerance, so the threshold is where it belongs for them. +# ───────────────────────────────────────────────────────────────────────────── - name: Register the service checks with Gatus hosts: observability become: yes @@ -47,13 +64,20 @@ 'name': (item.0.host | lower | regex_replace('[/_.,# +&]', '-')) ~ '/' ~ item.1, 'group': 'services', 'token': gatus_push_tokens[item.0.host], - 'heartbeat': '16m'}] }}" + 'heartbeat': '11m'}] }}" loop: "{{ host_units | subelements('units') }}" - name: Register the service endpoints ansible.builtin.include_role: name: gatus_endpoint vars: + gatus_endpoint_default_alerts: + - type: signal + # 1, not 3: the heartbeat window is the tolerance. See the note above. + failure-threshold: 1 + success-threshold: 2 + send-on-resolved: true + minimum-reminder-interval: 6h gatus_endpoint_name: services gatus_endpoint_external: "{{ service_endpoints }}" diff --git a/ansible/infra/402_public_monitoring.yml b/ansible/infra/402_public_monitoring.yml index b66d10f..d924468 100644 --- a/ansible/infra/402_public_monitoring.yml +++ b/ansible/infra/402_public_monitoring.yml @@ -75,23 +75,36 @@ 'group': 'domain', 'url': 'https://' ~ item, 'interval': '24h', - 'conditions': ['[DOMAIN_EXPIRATION] > 336h']}] }}" + 'conditions': ['[DOMAIN_EXPIRATION] > 336h'], + 'alerts': [{'type': 'signal', 'failure-threshold': 1, + 'success-threshold': 1, 'send-on-resolved': true, + 'minimum-reminder-interval': '168h'}]}] }}" loop: "{{ monitored_domains }}" # ── DNS ────────────────────────────────────────────────────────────────── + # 6h, not daily: a DNS query is cheap and a wrong record is an outage. The + # domain check stays at 24h because it does a WHOIS/RDAP lookup against a + # free service. Alert on the first failure - at a 6h interval, waiting for + # three would be nearly a day. - name: Build the DNS endpoints ansible.builtin.set_fact: dns_endpoints: "{{ dns_endpoints | default([]) + [{ 'name': item.sub ~ '.' ~ root_domain, 'group': 'dns', 'url': dns_resolver, - 'interval': '24h', + 'interval': '6h', 'dns': {'query-type': 'A', 'query-name': item.sub ~ '.' ~ root_domain}, 'conditions': ['[DNS_RCODE] == NOERROR', - '[BODY] == ' ~ hostvars[item.host].ansible_host]}] }}" + '[BODY] == ' ~ hostvars[item.host].ansible_host], + 'alerts': [{'type': 'signal', 'failure-threshold': 1, + 'success-threshold': 1, 'send-on-resolved': true, + 'minimum-reminder-interval': '24h'}]}] }}" loop: "{{ dns_records }}" # ── Public HTTP ────────────────────────────────────────────────────────── + # failure-threshold 3 at a 5m interval = 15 minutes. A pulled endpoint has + # no heartbeat window, so unlike the push checks the tolerance has to live + # in the threshold - and one failed poll of a public site is usually a blip. - name: Build the public HTTP endpoints ansible.builtin.set_fact: http_endpoints: "{{ http_endpoints | default([]) + [{ @@ -100,7 +113,10 @@ 'url': 'https://' ~ item.sub ~ '.' ~ root_domain ~ item.path, 'interval': '5m', 'conditions': ['[STATUS] == ' ~ item.status, - '[CERTIFICATE_EXPIRATION] > 168h']}] }}" + '[CERTIFICATE_EXPIRATION] > 168h'], + 'alerts': [{'type': 'signal', 'failure-threshold': 3, + 'success-threshold': 2, 'send-on-resolved': true, + 'minimum-reminder-interval': '6h'}]}] }}" loop: "{{ public_sites }}" # ── Public TCP ─────────────────────────────────────────────────────────── @@ -111,7 +127,10 @@ 'group': 'public', 'url': 'tcp://' ~ hostvars[item.host].ansible_host ~ ':' ~ item.port, 'interval': '5m', - 'conditions': ['[CONNECTED] == true']}] }}" + 'conditions': ['[CONNECTED] == true'], + 'alerts': [{'type': 'signal', 'failure-threshold': 3, + 'success-threshold': 2, 'send-on-resolved': true, + 'minimum-reminder-interval': '6h'}]}] }}" loop: "{{ public_tcp }}" - name: Register the public-facing endpoints diff --git a/ansible/infra/403_service_probe_registration.yml b/ansible/infra/403_service_probe_registration.yml index a8c8b35..c92b767 100644 --- a/ansible/infra/403_service_probe_registration.yml +++ b/ansible/infra/403_service_probe_registration.yml @@ -14,6 +14,23 @@ # They used to push to Uptime Kuma. The scripts now POST with a bearer token # instead of GETting ?status=up, and each host uses its own token. +# ───────────────────────────────────────────────────────────────────────────── +# Alerting thresholds, and why they differ by check type. +# +# `failure-threshold` counts CONSECUTIVE failures, but "consecutive" means a +# different amount of wall-clock time per check: +# +# push/heartbeat endpoints a failure is produced once per heartbeat window +# pulled endpoints a failure is produced once per interval +# +# So the default of 3 would mean 33 minutes on an 11m heartbeat and over a day +# on a 7h one - and the heartbeat window ALREADY encodes the tolerance. An 11m +# window on a 5-minute push is precisely "one missed push forgiven"; stacking a +# threshold of 3 on top triples a tolerance that was already chosen. +# +# Hence: push endpoints alert on the FIRST heartbeat failure. Pulled endpoints +# have no built-in tolerance, so the threshold is where it belongs for them. +# ───────────────────────────────────────────────────────────────────────────── - name: Register the per-service probes with Gatus hosts: observability become: yes @@ -36,12 +53,19 @@ 'name': item.name, 'group': 'probe', 'token': gatus_push_tokens[item.host], - 'heartbeat': '16m'}] }}" + 'heartbeat': '11m'}] }}" loop: "{{ probes }}" - name: Register the probe endpoints ansible.builtin.include_role: name: gatus_endpoint vars: + gatus_endpoint_default_alerts: + - type: signal + # 1, not 3: the heartbeat window is the tolerance. See the note above. + failure-threshold: 1 + success-threshold: 2 + send-on-resolved: true + minimum-reminder-interval: 6h gatus_endpoint_name: probes gatus_endpoint_external: "{{ probe_endpoints }}" diff --git a/ansible/playbooks/backups.yml b/ansible/playbooks/backups.yml index 61c005d..0be3807 100644 --- a/ansible/playbooks/backups.yml +++ b/ansible/playbooks/backups.yml @@ -65,13 +65,17 @@ store_sources: [arbret, headscale, memos, vaultwarden, lnbits, forgejo] tasks: + # 26h, not 7h: the DUMP is genuinely daily, so the window cannot be tighter + # than a day plus slack. The store-side check catches the same fault within + # 6h by reading the artefact's dump timestamp out of the filename, so this is + # the slow backstop rather than the primary signal. - name: Build the dump endpoint list ansible.builtin.set_fact: dump_endpoints: "{{ dump_endpoints | default([]) + [{ 'name': item.name, 'group': 'backup-dump', 'token': gatus_push_tokens[item.host], - 'heartbeat': '30h'}] }}" + 'heartbeat': '26h'}] }}" loop: "{{ dump_sources }}" - name: Build the store endpoint list @@ -80,16 +84,24 @@ 'name': item, 'group': 'backup-store', 'token': gatus_push_tokens['small_backups_local'], - 'heartbeat': '30h'}] }}" + 'heartbeat': '7h'}] }}" loop: "{{ store_sources }}" - name: Register the backup endpoints ansible.builtin.include_role: name: gatus_endpoint vars: + # Push endpoints: the heartbeat window is the tolerance, so alert on + # the first failure rather than waiting for three 7h windows to pass. + gatus_endpoint_default_alerts: + - type: signal + failure-threshold: 1 + success-threshold: 2 + send-on-resolved: true + minimum-reminder-interval: 12h gatus_endpoint_name: backups gatus_endpoint_external: "{{ dump_endpoints + store_endpoints + [{ 'name': 'pull job', 'group': 'backup-store', 'token': gatus_push_tokens['small_backups_local'], - 'heartbeat': '30h'}] }}" + 'heartbeat': '7h'}] }}" diff --git a/ansible/roles/backup_store/defaults/main.yml b/ansible/roles/backup_store/defaults/main.yml index 3a0754c..e6d17a3 100644 --- a/ansible/roles/backup_store/defaults/main.yml +++ b/ansible/roles/backup_store/defaults/main.yml @@ -17,9 +17,12 @@ backup_store_sources: [] backup_store_check_push_base: "" backup_store_check_push_token: "" -# Runs after the 04:00 pull. Late enough that a slow pull has finished, early -# enough that a failure is visible before the working day. -backup_store_check_on_calendar: "*-*-* 05:30:00" +# Every six hours, offset past the 04:00 pull so the first run of the day sees a +# finished pull. The BACKUPS are daily, but this check is not - it reads the +# source's dump timestamp out of the artefact filename, so running it more often +# catches "the source stopped dumping" within hours rather than a day, and lets +# the Gatus heartbeat be 7h instead of 30h. +backup_store_check_on_calendar: "*-*-* 05:30:00,11:30:00,17:30:00,23:30:00" # An artefact older than this is stale. Sources dump daily at 02:00-02:30 and the # pull is at 04:00, so 26h tolerates exactly one missed night before alarming. diff --git a/ansible/roles/gatus/defaults/main.yml b/ansible/roles/gatus/defaults/main.yml index 9c7cb12..e8bf22f 100644 --- a/ansible/roles/gatus/defaults/main.yml +++ b/ansible/roles/gatus/defaults/main.yml @@ -100,3 +100,9 @@ gatus_self_check: true # container would get free only if it ran as root; it does not. Set false if # you never use icmp:// checks and want the capability dropped entirely. gatus_allow_icmp: true + +# ── Shared network ─────────────────────────────────────────────────────────── +# Gatus runs in a container, so the HOST's loopback is not reachable from it. +# Anything Gatus must talk to locally - the Signal API that sends its alerts - +# has to be on a shared docker network and addressed by service name. +gatus_network: monitoring diff --git a/ansible/roles/gatus/tasks/configure.yml b/ansible/roles/gatus/tasks/configure.yml index 94f71f0..ef5d7e7 100644 --- a/ansible/roles/gatus/tasks/configure.yml +++ b/ansible/roles/gatus/tasks/configure.yml @@ -5,6 +5,16 @@ changed_when: false failed_when: gatus_docker_check.rc != 0 +# Created explicitly rather than by either compose file, so neither the gatus +# stack nor the signal-api stack has to be deployed before the other. +- name: Ensure the shared monitoring network exists + ansible.builtin.command: "docker network create {{ gatus_network }}" + register: gatus_net + changed_when: "'already exists' not in gatus_net.stderr" + failed_when: + - gatus_net.rc != 0 + - "'already exists' not in gatus_net.stderr" + - name: Create the gatus directories ansible.builtin.file: path: "{{ item.path }}" diff --git a/ansible/roles/gatus/templates/docker-compose.yml.j2 b/ansible/roles/gatus/templates/docker-compose.yml.j2 index 67d7573..512c9c5 100644 --- a/ansible/roles/gatus/templates/docker-compose.yml.j2 +++ b/ansible/roles/gatus/templates/docker-compose.yml.j2 @@ -37,8 +37,17 @@ services: - NET_RAW {% endif %} + networks: + # Shared with signal-api, so alerts can be delivered by service name. + # 127.0.0.1 inside this container is the container, not the host. + - {{ gatus_network }} + logging: driver: json-file options: max-size: "10m" max-file: "3" + +networks: + {{ gatus_network }}: + external: true diff --git a/ansible/roles/gatus_endpoint/defaults/main.yml b/ansible/roles/gatus_endpoint/defaults/main.yml index 89b669d..5c330cc 100644 --- a/ansible/roles/gatus_endpoint/defaults/main.yml +++ b/ansible/roles/gatus_endpoint/defaults/main.yml @@ -31,3 +31,14 @@ gatus_endpoint_external: [] gatus_config_dir: /opt/gatus/config gatus_endpoints_dir: "{{ gatus_config_dir }}/endpoints" gatus_gid: 10001 + +# Alerts attached to every endpoint in this file that does not specify its own. +# +# Gatus's provider-level `default-alert` only supplies DEFAULTS - an endpoint +# still has to opt in with `alerts: - type: signal` or it alerts on nothing at +# all. With ~90 endpoints that cannot be written by hand, so it is applied here. +# +# failure-threshold is set by the CALLER, because the right value depends on the +# check's cadence and there is no single correct default. See the note in +# infra/400_host_monitoring.yml. +gatus_endpoint_default_alerts: [] diff --git a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 index fa13d1f..54340cd 100644 --- a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 +++ b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 @@ -14,9 +14,10 @@ external-endpoints: heartbeat: interval: {{ e.heartbeat }} {% endif %} -{% if e.alerts | default([]) %} +{% set _alerts = e.alerts | default(gatus_endpoint_default_alerts) %} +{% if _alerts %} alerts: -{{ e.alerts | to_nice_yaml(indent=2) | indent(6, true) }} +{{ _alerts | to_nice_yaml(indent=2) | indent(6, true) }} {% endif %} {% endfor %} {% endif %} @@ -44,9 +45,10 @@ endpoints: {% for c in e.conditions %} - "{{ c }}" {% endfor %} -{% if e.alerts | default([]) %} +{% set _alerts = e.alerts | default(gatus_endpoint_default_alerts) %} +{% if _alerts %} alerts: -{{ e.alerts | to_nice_yaml(indent=2) | indent(6, true) }} +{{ _alerts | to_nice_yaml(indent=2) | indent(6, true) }} {% endif %} {% endfor %} {% endif %} diff --git a/ansible/roles/signal_api/README.md b/ansible/roles/signal_api/README.md new file mode 100644 index 0000000..eadcfb7 --- /dev/null +++ b/ansible/roles/signal_api/README.md @@ -0,0 +1,133 @@ +# signal_api + +Runs [signal-cli-rest-api](https://github.com/bbernhard/signal-cli-rest-api) on +the `observability` host. Gatus uses it to deliver alerts over Signal. + +Gatus does not speak Signal — it POSTs JSON to this service, which holds the +Signal identity and does the protocol work. + +## It is never published, and that is not optional + +**This API has no authentication of any kind.** No key, no token, no basic auth. +Anything that can reach the port can send messages as your identity and read +your Signal. So the compose file publishes **no ports at all** and there is no +Caddy vhost. + +Gatus reaches it over a shared docker network (`monitoring`) by service name: +`http://signal-api:8080`. That is also *why* a shared network is needed rather +than a published port — Gatus runs in a container, so `127.0.0.1` for Gatus is +the Gatus container, not the host. + +The network is created by an explicit Ansible task in both this role and +`gatus`, so neither stack has to be deployed before the other. + +## MODE, and why `native` + +Upstream offers `normal`, `native`, `json-rpc` and `json-rpc-native`. The +json-rpc modes keep a resident JVM daemon and upstream describes them as +"increased memory". + +**This VPS has 464 MB of RAM**, already running Gatus and Caddy. A resident JVM +is not affordable. `native` runs a precompiled GraalVM binary per request — no +daemon, no resident cost — and alerts are rare enough that paying startup cost +per alert is the right trade. + +## Linking the device — a one-time manual step + +Ansible cannot scan a QR code, so this is manual. **Do not use +`/v1/qrcodelink`** — it is broken in `native` mode. + +### The trap + +`GET /v1/qrcodelink?device_name=...` returns: + +```json +{"error":"Couldn't create QR code: no data to encode"} +``` + +The linking itself is fine: running the binary directly inside the container +emits a perfectly good provisioning URI. + +``` +$ docker exec signal-api signal-cli-native link -n gatus +sgnl://linkdevice?uuid=...&pub_key=... +``` + +It is the REST wrapper that fails to capture that output in `native` mode. + +**Do not "fix" this by switching MODE to `normal` or `json-rpc`.** That puts a +JVM in the path of *every alert* on a 464 MB host, permanently degrading the +running system to work around a step performed once. Generate the QR yourself +instead. + +### The procedure + +**`docker exec` runs as root, but the service runs as uid 1000.** Without +`--config`, signal-cli writes the linked account to `/root/.local/share/signal-cli` +— the container's ephemeral layer, NOT the mounted volume. It looks like it +worked (`Associated with: +34…`), `/v1/accounts` keeps returning `[]`, and the +account is destroyed on the next `docker compose up`. Always pass `--config`. + +1. Start the link and capture the URI. It must keep running while you scan: + + docker exec signal-api sh -c "rm -f /tmp/link.uri; \ + nohup signal-cli-native --config /home/.local/share/signal-cli \ + link -n gatus > /tmp/link.uri 2>/tmp/link.log & echo started" + sleep 10 + docker exec signal-api cat /tmp/link.uri + + Do **not** add `setsid`, and do **not** background `docker exec` itself from + the host — the first stops the URI appearing, the second is killed when the + Ansible task returns. The output is block-buffered because stdout is a file, + so the URI appears only after several seconds; `stdbuf` does not help, as the + buffering is GraalVM's, not libc's. + +2. Render the QR on your own machine and scan it: + + qrencode -o /tmp/qr.png -s 12 -m 4 "sgnl://linkdevice?uuid=...&pub_key=..." + +3. Phone: Signal → Settings → Linked devices → **+** → scan. Provisioning links + expire in a couple of minutes, so generate and scan in one sitting. + +4. Confirm — this must list the number, not `[]`: + + docker exec signal-api curl -s http://localhost:8080/v1/accounts + +5. Send a test message: + + docker exec signal-api curl -s -X POST -H "Content-Type: application/json" \ + -d '{"message":"test","number":"+34…","recipients":["+34…"]}' \ + http://localhost:8080/v2/send + +Alerts are sent **from your own number**, so sending to yourself lands in Note +to Self. If the device is ever unlinked from the phone, alerts stop silently — +which is why this service is itself monitored. + +### If the phone says "network error" + +The phone is not the problem. `chat.signal.org` resolves to AWS Global +Accelerator **dualstack** addresses with the AAAA records first, this container +has no IPv6 address at all, and this host's IPv6 path is broken — the same edge +that returned a bogus 404 for the Go tarball. signal-cli reaches for an +unreachable IPv6 address and dies with `Link request error: Connection closed!`, +while the phone can only report a failed handshake. + +That is what `gai.conf` (mounted at `/etc/gai.conf`) fixes. If linking starts +failing again, check it is still mounted and that `getent ahosts chat.signal.org` +returns an IPv4 address first. + +## Backups + +Deliberately **not** backed up. The data directory holds Signal private keys, +and the recovery path is to link again from the phone — which takes a minute and +does not depend on any stored artefact. Backing it up would copy a credential +off the host to buy nothing. + +## Verifying + +```bash +docker ps --filter name=signal-api +docker exec signal-api curl -fsS http://localhost:8080/v1/health +docker exec signal-api curl -fsS http://localhost:8080/v1/accounts +docker logs signal-api --tail 50 +``` diff --git a/ansible/roles/signal_api/defaults/main.yml b/ansible/roles/signal_api/defaults/main.yml new file mode 100644 index 0000000..6bb7d4f --- /dev/null +++ b/ansible/roles/signal_api/defaults/main.yml @@ -0,0 +1,39 @@ +--- +# signal-cli-rest-api: the transport Gatus uses to send Signal messages. +# +# Gatus does not speak Signal. It POSTs JSON to this service, which holds the +# actual Signal identity and does the protocol work. + +# Pinned by digest for the same reason as Gatus: a tag is mutable. +# Upstream publishes no versioned tags worth pinning to, so this pins the +# DIGEST that `latest` resolved to when this was reviewed. `latest` is a moving +# target; a digest is a content address, and `docker compose pull` either +# fetches exactly this image or fails. +signal_api_image_digest: "sha256:2399d449123cdad56c4d859277e3b9127e1a00c4d2ab4601c239882609286cf8" +signal_api_image: "bbernhard/signal-cli-rest-api@{{ signal_api_image_digest }}" + +signal_api_dir: /opt/signal-api +signal_api_data_dir: "{{ signal_api_dir }}/data" + +# MODE matters on this host. Upstream offers normal / native / json-rpc / +# json-rpc-native. json-rpc keeps a resident JVM daemon and upstream describes it +# as "increased memory" - this VPS has 464MB total and already runs Gatus and +# Caddy, so a resident JVM is not affordable. `native` runs a precompiled +# GraalVM binary per request: no daemon, no resident cost, and alerts are rare +# enough that paying startup per alert is the right trade. +signal_api_mode: native + +# Port INSIDE the shared docker network. Never published to the host: this API +# has NO AUTHENTICATION of any kind. Anyone who can reach it can send messages +# as you and read your Signal. +signal_api_port: 8080 + +# Both this and Gatus join this network so Gatus can reach the API by service +# name. Gatus runs in a container, so the host's loopback is NOT reachable from +# it - this is why a shared network is required rather than a published port. +signal_api_network: monitoring +signal_api_service_name: signal-api + +# The uid the upstream image drops to (`setpriv --reuid=1000`). The data +# directory must be owned by it or signal-cli cannot write the account. +signal_api_uid: 1000 diff --git a/ansible/roles/signal_api/tasks/main.yml b/ansible/roles/signal_api/tasks/main.yml new file mode 100644 index 0000000..1aa92b4 --- /dev/null +++ b/ansible/roles/signal_api/tasks/main.yml @@ -0,0 +1,93 @@ +--- +- name: Assert Docker is available + ansible.builtin.command: docker --version + register: signal_docker_check + changed_when: false + +# Created explicitly rather than by either compose file, so neither stack has to +# be deployed before the other and neither owns it. +- name: Ensure the shared monitoring network exists + ansible.builtin.command: "docker network create {{ signal_api_network }}" + register: signal_net + changed_when: "'already exists' not in signal_net.stderr" + failed_when: + - signal_net.rc != 0 + - "'already exists' not in signal_net.stderr" + +- name: Create the signal-api directory + ansible.builtin.file: + path: "{{ signal_api_dir }}" + state: directory + owner: root + group: root + mode: "0755" + +# Owned by the container's uid, NOT root. +# +# The image drops to uid 1000 (`setpriv --reuid=1000`), and a root-owned 0700 +# directory cannot be traversed by uid 1000 - signal-cli then fails to write the +# account and linking silently never completes, leaving a 39-byte accounts.json +# with no accounts and the API returning "Failed to read local accounts list". +# +# 0700 on uid 1000 is still private: only that uid and root can read the Signal +# private keys, which is the property actually wanted. +- name: Create the signal-api data directory owned by the container user + ansible.builtin.file: + path: "{{ signal_api_data_dir }}" + state: directory + owner: "{{ signal_api_uid }}" + group: "{{ signal_api_uid }}" + mode: "0700" + +- name: Write the IPv4-preference resolver config + ansible.builtin.template: + src: gai.conf.j2 + dest: "{{ signal_api_dir }}/gai.conf" + owner: root + group: root + mode: "0644" + +- name: Write the docker compose file + ansible.builtin.template: + src: docker-compose.yml.j2 + dest: "{{ signal_api_dir }}/docker-compose.yml" + owner: root + group: root + mode: "0644" + +- name: Pull the pinned signal-api image + ansible.builtin.command: + cmd: docker compose pull + chdir: "{{ signal_api_dir }}" + register: signal_pull + changed_when: "'Downloaded newer image' in signal_pull.stderr or 'Pull complete' in signal_pull.stderr" + +- name: Start signal-api + ansible.builtin.command: + cmd: docker compose up -d --remove-orphans + chdir: "{{ signal_api_dir }}" + register: signal_up + changed_when: "'Started' in signal_up.stderr or 'Created' in signal_up.stderr or 'Recreated' in signal_up.stderr" + +- name: Wait for the API to answer + ansible.builtin.command: + cmd: "docker exec {{ signal_api_service_name }} curl -fsS http://localhost:{{ signal_api_port }}/v1/health" + register: signal_health + until: signal_health.rc == 0 + retries: 12 + delay: 5 + changed_when: false + +- name: Report whether an account is linked yet + ansible.builtin.command: + cmd: "docker exec {{ signal_api_service_name }} curl -fsS http://localhost:{{ signal_api_port }}/v1/accounts" + register: signal_accounts + changed_when: false + failed_when: false + +- name: Show the linking status + ansible.builtin.debug: + msg: >- + {{ 'Linked account(s): ' ~ signal_accounts.stdout + if (signal_accounts.stdout | default('[]') | trim) not in ['[]', '', 'null'] + else 'NO ACCOUNT LINKED YET - this is a one-time manual step, see the role README.' }} diff --git a/ansible/roles/signal_api/templates/docker-compose.yml.j2 b/ansible/roles/signal_api/templates/docker-compose.yml.j2 new file mode 100644 index 0000000..9825299 --- /dev/null +++ b/ansible/roles/signal_api/templates/docker-compose.yml.j2 @@ -0,0 +1,47 @@ +# Managed by Ansible (roles/signal_api) +services: + {{ signal_api_service_name }}: + image: {{ signal_api_image }} + container_name: {{ signal_api_service_name }} + restart: unless-stopped + + environment: + MODE: "{{ signal_api_mode }}" + + volumes: + # Prefer IPv4. See gai.conf.j2 - without this, signal-cli reaches for + # chat.signal.org's IPv6 address, which is unreachable from here, and + # linking fails with an opaque "network error" on the phone. + - {{ signal_api_dir }}/gai.conf:/etc/gai.conf:ro + # Holds the Signal identity: the linked-device keys and registration + # state. Lose this and the device must be linked again by scanning a new + # QR code from the phone. It is also the most sensitive thing on this + # host - anyone with these keys can send and read Signal as you. + - {{ signal_api_data_dir }}:/home/.local/share/signal-cli + + networks: + - {{ signal_api_network }} + + # NO PORTS. Deliberately. + # + # This API has no authentication whatsoever - no key, no token, nothing. + # Publishing it, even on 127.0.0.1, would expose "send a Signal message as + # this identity" to anything that can reach the host. Gatus talks to it over + # the shared docker network by service name instead, which is why no port is + # published and why there is no Caddy vhost. + + healthcheck: + test: ["CMD", "curl", "-fsS", "http://localhost:8080/v1/health"] + interval: 60s + timeout: 5s + retries: 3 + + logging: + driver: json-file + options: + max-size: "10m" + max-file: "3" + +networks: + {{ signal_api_network }}: + external: true diff --git a/ansible/roles/signal_api/templates/gai.conf.j2 b/ansible/roles/signal_api/templates/gai.conf.j2 new file mode 100644 index 0000000..9b2cca4 --- /dev/null +++ b/ansible/roles/signal_api/templates/gai.conf.j2 @@ -0,0 +1,17 @@ +# Managed by Ansible (roles/signal_api) +# +# Prefer IPv4 over IPv6 in getaddrinfo. +# +# chat.signal.org resolves to AWS Global Accelerator dualstack addresses, and +# DNS returns the AAAA records first. This container has NO IPv6 address at all, +# and this host's IPv6 path is unreliable anyway - the same edge that made +# Google's IPv6 endpoint return a confident 404 for the Go tarball. +# +# signal-cli would connect to the AAAA address, fail, and report +# Link request error: Connection closed! +# while the phone showed a bare "network error" - a failure with no obvious +# cause on either end. +# +# This line flips the precedence so IPv4-mapped addresses sort first, which is +# the standard glibc fix. It does NOT disable IPv6; it only changes the order. +precedence ::ffff:0:0/96 100 diff --git a/ansible/services/gatus/deploy_gatus_playbook.yml b/ansible/services/gatus/deploy_gatus_playbook.yml index 704ada9..d6a5ac1 100644 --- a/ansible/services/gatus/deploy_gatus_playbook.yml +++ b/ansible/services/gatus/deploy_gatus_playbook.yml @@ -7,6 +7,29 @@ - name: Deploy Gatus on the observability host hosts: observability become: yes + vars: + gatus_alerting: + signal: + # NOTE: the key is `api-url`, not `url` as upstream's own README table + # says - see alerting/provider/signal/signal.go. Gatus appends /v2/send + # itself if the suffix is missing. + # + # Reached by service name over the shared docker network. Gatus runs in + # a container, so 127.0.0.1 here would be the Gatus container, not the + # host - and the Signal API deliberately publishes no ports because it + # has no authentication. + api-url: "http://signal-api:8080" + number: "{{ signal_number }}" + recipients: "{{ signal_recipients }}" + default-alert: + # Overridden per group by the registration playbooks; these are the + # values that apply if a caller sets nothing. + failure-threshold: 1 + success-threshold: 2 + send-on-resolved: true + # An ongoing outage should not become an ongoing phone buzz. + minimum-reminder-interval: 6h + roles: - gatus diff --git a/ansible/services/signal-api/deploy_signal_api_playbook.yml b/ansible/services/signal-api/deploy_signal_api_playbook.yml new file mode 100644 index 0000000..d4690f9 --- /dev/null +++ b/ansible/services/signal-api/deploy_signal_api_playbook.yml @@ -0,0 +1,44 @@ +--- +# The Signal transport for Gatus alerts. +# +# Deliberately NOT published and NOT fronted by Caddy: the API has no +# authentication of any kind, so it is reachable only from the shared docker +# network that Gatus is on. See roles/signal_api/README.md, including the +# one-time manual step to link the device. +- name: Deploy the Signal API on the observability host + hosts: observability + become: yes + roles: + - signal_api + + # post_tasks, not a second play: the registration below needs the role's + # defaults (service name, port) in scope, and a separate play would not have + # them. + post_tasks: + # Monitored, because a dead alert transport is the worst kind of dead: every + # check could be failing and nothing would tell you. Gatus polls it over the + # shared network - the same path the alerts take - so this proves the actual + # delivery route rather than merely that a container is running. + # + # Deliberately NOT backed up: the data directory holds Signal private keys, + # and the recovery path is to link the device again from the phone. Backing + # it up would copy a credential off the host to buy nothing. + - name: Register the signal-api health endpoint with Gatus + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: signal-api + gatus_endpoint_pulled: + - name: signal-api + group: infrastructure + url: "http://{{ signal_api_service_name }}:{{ signal_api_port }}/v1/health" + interval: 5m + # Deliberately NOT alerted via Signal: if this endpoint is down, + # Signal is exactly what cannot deliver the alert. It is visible on + # the dashboard, and its failure shows up indirectly as every other + # alert going missing. + conditions: + # /v1/health answers 204 No Content, not 200 - checked live. Any + # 2xx is asserted rather than the exact code, so an upstream + # change from 204 to 200 does not read as an outage. + - "[STATUS] < 300" diff --git a/ansible/site.yml b/ansible/site.yml index 10ee4a7..35f5867 100644 --- a/ansible/site.yml +++ b/ansible/site.yml @@ -26,6 +26,10 @@ # Gatus first: the three plays below register endpoints with it, and registering # against a host that is not serving yet would simply fail. - import_playbook: services/gatus/deploy_gatus_playbook.yml +# The Signal transport for Gatus alerts. Shares a docker network with Gatus and +# publishes no ports - the API has no authentication. Needs a one-time manual +# device link; see roles/signal_api/README.md. +- import_playbook: services/signal-api/deploy_signal_api_playbook.yml - import_playbook: infra/400_host_monitoring.yml - import_playbook: infra/401_service_monitoring.yml - import_playbook: infra/402_public_monitoring.yml From f6656b0ff78845740c9a3733265c55350b25efd5 Mon Sep 17 00:00:00 2001 From: counterweight Date: Mon, 14 Sep 2026 22:14:44 +0200 Subject: [PATCH 67/67] gatus: show resolved values on successful conditions, ui as a pass-through Asked to un-hide endpoint properties; the answer is that nothing was hidden. All six hide-* options (hide-hostname, hide-url, hide-port, hide-conditions, hide-errors, and dont-resolve-failed-conditions) already default to false upstream, so hostname, URL, port, conditions and errors were all being shown. The one setting that genuinely displays MORE is resolve-successful-conditions. By default a FAILING check resolves its placeholders - "[STATUS] (502) == 200" - while a PASSING one drops the value and shows only "[STATUS] == 200". With it on, a healthy DNS check now reads: [DNS_RCODE] (NOERROR) == NOERROR [BODY] (64.226.70.190) == 64.226.70.190 which says what it actually resolved to rather than merely that the assertion held. Applied to all 27 pulled endpoints via a gatus_endpoint_default_ui that each caller can override. It applies to pulled endpoints ONLY: an external (push) endpoint has no `ui` field upstream at all, because it carries no conditions - success comes from the push. The template was initially emitting the block in both loops; emitting an unknown key into the external-endpoints list risks a parse rejection, and a rejected config is exactly what skip-invalid-config-update exists to survive. Also made the page-level `ui` a pass-through dict, the same shape as gatus_alerting, so every upstream option (description, dashboard-heading, logo, link, favicon, buttons, custom-css, dark-mode, default-sort-by, default-filter-by) is reachable without a variable per key. Replaces the two one-off gatus_ui_title / gatus_ui_header variables. Set default-sort-by: group, because the dashboard's own grouping toggle starts OFF and remembers per browser in localStorage - without it the ten groups render as one flat list of 86 rows for anyone who has not clicked it. Note the limit of all this: it is configuration. Layout, card design and group rendering come from the Vue app compiled into the binary (//go:embed static), so changing those means forking and rebuilding the image - which would discard the pinned-digest property the deployment relies on. Co-Authored-By: Claude Opus 5 (1M context) --- ansible/roles/gatus/defaults/main.yml | 14 ++++++++++++-- ansible/roles/gatus/templates/config.yaml.j2 | 3 +-- ansible/roles/gatus_endpoint/defaults/main.yml | 14 ++++++++++++++ .../gatus_endpoint/templates/endpoints.yaml.j2 | 6 ++++++ 4 files changed, 33 insertions(+), 4 deletions(-) diff --git a/ansible/roles/gatus/defaults/main.yml b/ansible/roles/gatus/defaults/main.yml index e8bf22f..6f6ec09 100644 --- a/ansible/roles/gatus/defaults/main.yml +++ b/ansible/roles/gatus/defaults/main.yml @@ -47,8 +47,18 @@ gatus_port: 8080 # publish this on 0.0.0.0: the external-endpoint push API shares the dashboard's # listener, and Caddy is what should be in front of both. gatus_bind_address: "127.0.0.1" -gatus_ui_title: "Status" -gatus_ui_header: "Status" +# Pass-through, the same shape as gatus_alerting: whatever is set here is +# rendered verbatim under `ui:`, so every option upstream supports is reachable +# without adding a variable per key. See config/ui/ui.go for the full set - +# title, description, header, dashboard-heading/subheading, logo, link, favicon, +# buttons, custom-css, dark-mode, default-sort-by, default-filter-by. +gatus_ui: + title: "Status" + header: "Status" + # Group by default. The dashboard's own groupByGroup toggle starts OFF and + # remembers per browser in localStorage, so without this the ten groups render + # as one flat list for anyone who has not clicked it. + default-sort-by: group # ── Storage ────────────────────────────────────────────────────────────────── # sqlite, not memory: history has to survive a restart, or the dashboard lies diff --git a/ansible/roles/gatus/templates/config.yaml.j2 b/ansible/roles/gatus/templates/config.yaml.j2 index 6a2a72e..faf61fb 100644 --- a/ansible/roles/gatus/templates/config.yaml.j2 +++ b/ansible/roles/gatus/templates/config.yaml.j2 @@ -40,8 +40,7 @@ storage: maximum-number-of-events: {{ gatus_storage_max_events }} ui: - title: {{ gatus_ui_title }} - header: {{ gatus_ui_header }} +{{ gatus_ui | to_nice_yaml(indent=2) | indent(2, true) }} {% if gatus_alerting %} alerting: diff --git a/ansible/roles/gatus_endpoint/defaults/main.yml b/ansible/roles/gatus_endpoint/defaults/main.yml index 5c330cc..90a2c19 100644 --- a/ansible/roles/gatus_endpoint/defaults/main.yml +++ b/ansible/roles/gatus_endpoint/defaults/main.yml @@ -42,3 +42,17 @@ gatus_gid: 10001 # check's cadence and there is no single correct default. See the note in # infra/400_host_monitoring.yml. gatus_endpoint_default_alerts: [] + +# UI options applied to every PULLED endpoint that does not set its own. +# +# Only pulled endpoints can carry this - an external (push) endpoint has no `ui` +# field at all, because it has no conditions to display. +# +# Note that every hide-* option already defaults to false upstream, so there is +# nothing to un-hide. The one setting that genuinely shows MORE is +# resolve-successful-conditions: by default a failing check displays the real +# value - "[STATUS] (502) == 200" - while a passing one drops it and shows only +# "[STATUS] == 200". Turning it on resolves both, so a healthy DNS check shows +# the IP it actually resolved rather than just the assertion. +gatus_endpoint_default_ui: + resolve-successful-conditions: true diff --git a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 index 54340cd..680691e 100644 --- a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 +++ b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 @@ -19,6 +19,7 @@ external-endpoints: alerts: {{ _alerts | to_nice_yaml(indent=2) | indent(6, true) }} {% endif %} +{# external endpoints have no `ui` field upstream - they carry no conditions #} {% endfor %} {% endif %} {% if gatus_endpoint_pulled %} @@ -50,5 +51,10 @@ endpoints: alerts: {{ _alerts | to_nice_yaml(indent=2) | indent(6, true) }} {% endif %} +{% set _ui = e.ui | default(gatus_endpoint_default_ui) %} +{% if _ui %} + ui: +{{ _ui | to_nice_yaml(indent=2) | indent(6, true) }} +{% endif %} {% endfor %} {% endif %}