diff --git a/ansible/host_vars/forgejo_runner_local/main.yml b/ansible/host_vars/forgejo_runner_local/main.yml new file mode 100644 index 0000000..6e852ce --- /dev/null +++ b/ansible/host_vars/forgejo_runner_local/main.yml @@ -0,0 +1,12 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - forgejo-runner diff --git a/ansible/host_vars/fulcrum_box_local/main.yml b/ansible/host_vars/fulcrum_box_local/main.yml index 2cf3c65..aa189b0 100644 --- a/ansible/host_vars/fulcrum_box_local/main.yml +++ b/ansible/host_vars/fulcrum_box_local/main.yml @@ -4,3 +4,14 @@ # which publishes the port. See host_vars/knots_box_local/main.yml for why this # lives in host_vars rather than in the role's defaults. fulcrum_ssl_port: 50002 + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - fulcrum diff --git a/ansible/host_vars/knots_box_local/main.yml b/ansible/host_vars/knots_box_local/main.yml index fb15905..3866541 100644 --- a/ansible/host_vars/knots_box_local/main.yml +++ b/ansible/host_vars/knots_box_local/main.yml @@ -12,3 +12,15 @@ bitcoin_p2p_port: 8333 datum_gateway_api_port: 7152 datum_gateway_stratum_port: 23334 + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - bitcoind + - datum-gateway diff --git a/ansible/host_vars/memos_box_local/main.yml b/ansible/host_vars/memos_box_local/main.yml new file mode 100644 index 0000000..f1fc19b --- /dev/null +++ b/ansible/host_vars/memos_box_local/main.yml @@ -0,0 +1,12 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - memos diff --git a/ansible/host_vars/monitoring/main.yml b/ansible/host_vars/monitoring/main.yml new file mode 100644 index 0000000..2ef805d --- /dev/null +++ b/ansible/host_vars/monitoring/main.yml @@ -0,0 +1,12 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - caddy diff --git a/ansible/host_vars/nodito/main.yml b/ansible/host_vars/nodito/main.yml index a6f6a58..c8176af 100644 --- a/ansible/host_vars/nodito/main.yml +++ b/ansible/host_vars/nodito/main.yml @@ -31,3 +31,15 @@ ups_port: auto ups_user: counterweight ups_offdelay: 120 # Seconds after shutdown before UPS cuts outlet power ups_ondelay: 30 # Seconds after mains returns before UPS restores outlet power + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - nut-server + - nut-monitor diff --git a/ansible/host_vars/spacey/main.yml b/ansible/host_vars/spacey/main.yml new file mode 100644 index 0000000..a50d116 --- /dev/null +++ b/ansible/host_vars/spacey/main.yml @@ -0,0 +1,13 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - headscale + - caddy diff --git a/ansible/host_vars/vipy/main.yml b/ansible/host_vars/vipy/main.yml new file mode 100644 index 0000000..3b077df --- /dev/null +++ b/ansible/host_vars/vipy/main.yml @@ -0,0 +1,15 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - forgejo + - lnbits + - caddy + - phoenixd diff --git a/ansible/host_vars/watchtower/main.yml b/ansible/host_vars/watchtower/main.yml new file mode 100644 index 0000000..efdf7d6 --- /dev/null +++ b/ansible/host_vars/watchtower/main.yml @@ -0,0 +1,13 @@ +--- + +# Systemd services deployed on this host, monitored every 5 minutes. +# +# The fact lives with the machine rather than in a central map, for the same +# reason the cross-host ports do: "what runs here" is a property of the host, +# and a central list is one more thing to forget to update when a service moves. +# +# Only units WE deploy belong here. Distro units (ssh, cron) have their own +# supervision and would be noise. +monitored_services: + - caddy + - ntfy diff --git a/ansible/infra/401_service_monitoring.yml b/ansible/infra/401_service_monitoring.yml new file mode 100644 index 0000000..f07a906 --- /dev/null +++ b/ansible/infra/401_service_monitoring.yml @@ -0,0 +1,83 @@ +--- +# Is each systemd-deployed service actually running? +# +# Every 5 minutes, with a 16-minute Gatus heartbeat - three missed runs before +# it alarms, so a reboot or a slow check does not page anyone, but a host that +# stops reporting does. +# +# This closes the gap that let a real bug run unnoticed: a backup script left +# forgejo, lnbits, headscale and memos stopped, and NOTHING caught it. The dumps +# exited 0, the artefacts were correct, the deploy said failed=0, and liveness +# only proves the HOST is up - not that anything on it is serving. +# +# One endpoint PER UNIT, not per host. A host running four services needs four +# endpoints, or a single red light says "something on vipy is down" without +# saying which - and that is the question you actually have at 3am. But only ONE +# timer per host: the check iterates that host's units and pushes a result for +# each, the same way check-backups.sh reports per source. Four units on vipy +# would otherwise mean four scripts, four services and four timers. +# +# Which units each host runs is in host_vars//main.yml as +# monitored_services, because "what runs here" is a property of the machine. +# +# Keys are host-qualified because unit names collide - caddy runs on four +# machines. Gatus computes sanitize(group)_sanitize(name), so group "services" +# and name "vipy/caddy" give services_vipy-caddy. + +# ───────────────────────────────────────────────────────────────────────────── +# Register one endpoint per unit. Runs first: Gatus reloads within 30s, and the +# host play above takes minutes, so every endpoint exists before its first push. +# ───────────────────────────────────────────────────────────────────────────── +- name: Register the service checks with Gatus + hosts: observability + become: yes + + tasks: + # Two plain steps rather than one clever expression: first collect which + # units each host declares, then flatten that into endpoints. + - name: Collect the units each host declares + ansible.builtin.set_fact: + host_units: "{{ host_units | default([]) + [{'host': item, 'units': hostvars[item].monitored_services}] }}" + loop: "{{ groups['managed'] | sort }}" + when: hostvars[item].monitored_services | default([]) | length > 0 + + - name: Build one endpoint per unit + ansible.builtin.set_fact: + service_endpoints: "{{ service_endpoints | default([]) + [{ + 'name': (item.0.host | lower | regex_replace('[/_.,# +&]', '-')) ~ '/' ~ item.1, + 'group': 'services', + 'token': gatus_push_tokens[item.0.host], + 'heartbeat': '16m'}] }}" + loop: "{{ host_units | subelements('units') }}" + + - name: Register the service endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: services + gatus_endpoint_external: "{{ service_endpoints }}" + +- name: Monitor systemd services on every host that has them + hosts: managed + become: yes + vars: + gatus_api: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints" + host_key: "{{ inventory_hostname | lower | regex_replace('[/_.,# +&]', '-') }}" + + tasks: + - name: Is every deployed service running? + ansible.builtin.include_role: + name: healthcheck + vars: + healthcheck_name: service-health + healthcheck_description: "systemd services on {{ inventory_hostname }}" + healthcheck_check: systemd-units + healthcheck_units: "{{ monitored_services }}" + healthcheck_units_key_prefix: "services_{{ host_key }}" + healthcheck_interval: "5min" + healthcheck_boot_delay: "2min" + # The per-unit results go to keys under this collection; the role's own + # single-result push is unused here, so only the base is set. + healthcheck_push_base: "{{ gatus_api }}" + healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" + when: monitored_services | default([]) | length > 0 diff --git a/ansible/infra/402_public_monitoring.yml b/ansible/infra/402_public_monitoring.yml new file mode 100644 index 0000000..b2c329b --- /dev/null +++ b/ansible/infra/402_public_monitoring.yml @@ -0,0 +1,125 @@ +--- +# Domain expiry, DNS correctness, and public endpoint reachability. +# +# These are the first checks in the estate that PULL rather than push, and that +# is the right way round for them: all three are about how the outside world +# sees us, so they must be measured from outside. Gatus polls from the +# observability host and needs nothing installed anywhere else - there is no +# script, no timer and no token, because nothing is reporting in. +# +# That also means these have no heartbeat. A heartbeat answers "did the thing +# that was supposed to report in do so"; when Gatus does the checking itself, +# failure is immediate and self-evident. + +- name: Register the public-facing checks with Gatus + hosts: observability + become: yes + + vars: + # Expected A records, derived from inventory rather than written down again. + # The estate's recurring bug is an address recorded in a second place and + # then left behind when the machine moved, so the check asserts against + # ansible_host - if a box is renumbered, inventory is the one edit. + dns_records: + - {sub: "{{ subdomains.gatus }}", host: monitoring} + - {sub: "{{ subdomains.ntfy }}", host: watchtower} + - {sub: "{{ subdomains.headscale }}", host: spacey} + - {sub: "{{ subdomains.vaultwarden }}", host: vipy} + - {sub: "{{ subdomains.forgejo }}", host: vipy} + - {sub: "{{ subdomains.lnbits }}", host: vipy} + - {sub: "{{ subdomains.ntfy_emergency_app }}", host: vipy} + - {sub: "{{ subdomains.personal_blog }}", host: vipy} + - {sub: "{{ subdomains.memos }}", host: vipy} + - {sub: "{{ subdomains.mempool }}", host: vipy} + - {sub: "{{ subdomains.datum_gateway }}", host: vipy} + + # A public resolver on purpose: this must test what the internet sees, not + # what a local cache or the tailnet's MagicDNS happens to answer. + dns_resolver: "1.1.1.1" + + # Expected status per site, checked live before being written down. + # 401 is the CORRECT answer for the two behind basic auth - asserting 200 + # there would go green precisely when the auth broke. + public_sites: + - {name: gatus, sub: "{{ subdomains.gatus }}", path: "/", status: 401} + - {name: ntfy, sub: "{{ subdomains.ntfy }}", path: "/", status: 200} + - {name: headscale, sub: "{{ subdomains.headscale }}", path: "/health", status: 200} + - {name: vaultwarden, sub: "{{ subdomains.vaultwarden }}", path: "/", status: 200} + - {name: forgejo, sub: "{{ subdomains.forgejo }}", path: "/", status: 200} + - {name: lnbits, sub: "{{ subdomains.lnbits }}", path: "/", status: 200} + - {name: avisame, sub: "{{ subdomains.ntfy_emergency_app }}", path: "/", status: 200} + - {name: blog, sub: "{{ subdomains.personal_blog }}", path: "/", status: 200} + - {name: memos, sub: "{{ subdomains.memos }}", path: "/", status: 200} + - {name: mempool, sub: "{{ subdomains.mempool }}", path: "/", status: 200} + - {name: datum, sub: "{{ subdomains.datum_gateway }}", path: "/", status: 401} + + # Ports published from the edge host by socket_proxy. + public_tcp: + - {name: bitcoin-p2p, host: vipy, port: "{{ hostvars['knots_box_local'].bitcoin_p2p_port }}"} + - {name: fulcrum-ssl, host: vipy, port: "{{ hostvars['fulcrum_box_local'].fulcrum_ssl_port }}"} + - {name: datum-stratum, host: vipy, port: "{{ hostvars['knots_box_local'].datum_gateway_stratum_port }}"} + + tasks: + # ── Domain expiry ──────────────────────────────────────────────────────── + - name: Build the domain endpoint + ansible.builtin.set_fact: + domain_endpoints: + - name: "{{ root_domain }}" + group: domain + # Needs a scheme: Gatus derives the endpoint TYPE from the URL prefix + # (endpoint.Type()), and a bare domain is UNKNOWN and rejected. The + # apex points at the registrar's parking page, which is irrelevant - + # the only condition here is the WHOIS expiry, and no status check is + # asserted, so what the page serves does not matter. + url: "https://{{ root_domain }}" + # 24h, and upstream enforces a 5m minimum for DOMAIN_EXPIRATION + # anyway because it uses a free whois service that must not be + # hammered and whose data updates slowly. + interval: 24h + # 336h = 14 days. Renewal is manual at the registrar, so this needs + # enough runway to act on. + conditions: + - "[DOMAIN_EXPIRATION] > 336h" + + # ── DNS ────────────────────────────────────────────────────────────────── + - name: Build the DNS endpoints + ansible.builtin.set_fact: + dns_endpoints: "{{ dns_endpoints | default([]) + [{ + 'name': item.sub ~ '.' ~ root_domain, + 'group': 'dns', + 'url': dns_resolver, + 'interval': '24h', + 'dns': {'query-type': 'A', 'query-name': item.sub ~ '.' ~ root_domain}, + 'conditions': ['[DNS_RCODE] == NOERROR', + '[BODY] == ' ~ hostvars[item.host].ansible_host]}] }}" + loop: "{{ dns_records }}" + + # ── Public HTTP ────────────────────────────────────────────────────────── + - name: Build the public HTTP endpoints + ansible.builtin.set_fact: + http_endpoints: "{{ http_endpoints | default([]) + [{ + 'name': item.name, + 'group': 'public', + 'url': 'https://' ~ item.sub ~ '.' ~ root_domain ~ item.path, + 'interval': '5m', + 'conditions': ['[STATUS] == ' ~ item.status, + '[CERTIFICATE_EXPIRATION] > 168h']}] }}" + loop: "{{ public_sites }}" + + # ── Public TCP ─────────────────────────────────────────────────────────── + - name: Build the public TCP endpoints + ansible.builtin.set_fact: + tcp_endpoints: "{{ tcp_endpoints | default([]) + [{ + 'name': item.name, + 'group': 'public', + 'url': 'tcp://' ~ hostvars[item.host].ansible_host ~ ':' ~ item.port, + 'interval': '5m', + 'conditions': ['[CONNECTED] == true']}] }}" + loop: "{{ public_tcp }}" + + - name: Register the public-facing endpoints + ansible.builtin.include_role: + name: gatus_endpoint + vars: + gatus_endpoint_name: public + gatus_endpoint_pulled: "{{ domain_endpoints + dns_endpoints + http_endpoints + tcp_endpoints }}" diff --git a/ansible/roles/gatus_endpoint/defaults/main.yml b/ansible/roles/gatus_endpoint/defaults/main.yml index 7d49d7c..89b669d 100644 --- a/ansible/roles/gatus_endpoint/defaults/main.yml +++ b/ansible/roles/gatus_endpoint/defaults/main.yml @@ -9,6 +9,11 @@ gatus_endpoint_name: "" # PULLED endpoints - Gatus makes the request and evaluates conditions. # - {name, group, url, interval, conditions: [...], alerts: [...]} +# +# A DNS check adds `dns: {query-type, query-name}` - and note that for those, +# `url` is the RESOLVER to ask, not the name being looked up. +# A domain-expiry check is just `url: ` with a [DOMAIN_EXPIRATION] +# condition; it uses WHOIS/RDAP and needs no scheme. gatus_endpoint_pulled: [] # EXTERNAL endpoints - the host pushes its own result. Gatus never reaches out, diff --git a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 index 1c84996..fa13d1f 100644 --- a/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 +++ b/ansible/roles/gatus_endpoint/templates/endpoints.yaml.j2 @@ -28,6 +28,18 @@ endpoints: group: {{ e.group }} url: "{{ e.url }}" interval: {{ e.interval | default('60s') }} +{% if e.dns is defined %} + # A DNS endpoint: `url` is the RESOLVER to ask, not the thing being asked + # about. The name being queried lives in query-name, and [BODY] holds the + # resolved record for the conditions below. + dns: + query-type: {{ e.dns['query-type'] }} + query-name: {{ e.dns['query-name'] }} +{% endif %} +{% if e.client is defined %} + client: +{{ e.client | to_nice_yaml(indent=2) | indent(6, true) }} +{% endif %} conditions: {% for c in e.conditions %} - "{{ c }}" diff --git a/ansible/roles/healthcheck/defaults/main.yml b/ansible/roles/healthcheck/defaults/main.yml index d29cbd1..526347a 100644 --- a/ansible/roles/healthcheck/defaults/main.yml +++ b/ansible/roles/healthcheck/defaults/main.yml @@ -28,6 +28,17 @@ healthcheck_boot_delay: "2min" healthcheck_push_url: "" healthcheck_push_token: "" +# Some checks report MORE THAN ONE result - a host with four systemd services +# needs four endpoints, or a single red light cannot tell you which one died. +# Those set healthcheck_push_base to the endpoints COLLECTION and the check body +# appends each key itself, the same way check-backups.sh reports per source. +healthcheck_push_base: "" + +# Units for the systemd-units check. Each becomes its own Gatus endpoint. +healthcheck_units: [] +# Prefix for the per-unit endpoint keys, e.g. "services_vipy" -> services_vipy-caddy. +healthcheck_units_key_prefix: "" + healthcheck_script_dir: /usr/local/bin healthcheck_log_dir: /var/log/healthchecks diff --git a/ansible/roles/healthcheck/templates/checks/systemd-units.sh.j2 b/ansible/roles/healthcheck/templates/checks/systemd-units.sh.j2 new file mode 100644 index 0000000..90c6193 --- /dev/null +++ b/ansible/roles/healthcheck/templates/checks/systemd-units.sh.j2 @@ -0,0 +1,33 @@ + # One result PER UNIT, not one for the host. A host running four services + # needs four endpoints: a single red light would tell you "something on vipy + # is down" without saying which, and that is the question you actually have. + # + # Keys are host-qualified because unit names collide - caddy runs on four + # machines. Gatus builds the key as sanitize(group)_sanitize(name), so + # group "services" + name "vipy/caddy" becomes services_vipy-caddy. + local prefix="{{ healthcheck_units_key_prefix }}" + local down="" + + for unit in {{ healthcheck_units | join(' ') }}; do + local state substate + state=$(systemctl is-active "$unit" 2>/dev/null || true) + substate=$(systemctl show -p SubState --value "$unit" 2>/dev/null || true) + + if [ "$state" = "active" ]; then + report_key "${prefix}-${unit}" "true" "${unit} active (${substate:-running})" + else + # `failed` and `inactive` are different stories: one crashed, one was + # stopped. Both are down, but the message should say which. + local detail="${unit} is ${state:-unknown}" + [ "$state" = "failed" ] && detail="${detail} (${substate:-failed})" + report_key "${prefix}-${unit}" "false" "$detail" + down="${down}${down:+, }${unit}=${state:-unknown}" + fi + done + + if [ -n "$down" ]; then + MESSAGE="down: ${down}" + return 1 + fi + MESSAGE="all {{ healthcheck_units | length }} units active" + return 0 diff --git a/ansible/roles/healthcheck/templates/healthcheck.service.j2 b/ansible/roles/healthcheck/templates/healthcheck.service.j2 index 574e37c..3ddd8aa 100644 --- a/ansible/roles/healthcheck/templates/healthcheck.service.j2 +++ b/ansible/roles/healthcheck/templates/healthcheck.service.j2 @@ -9,6 +9,7 @@ User=root ExecStart={{ healthcheck_script_dir }}/{{ healthcheck_name }}-healthcheck.sh Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} Environment=HEALTHCHECK_PUSH_TOKEN={{ healthcheck_push_token }} +Environment=HEALTHCHECK_PUSH_BASE={{ healthcheck_push_base }} StandardOutput=journal StandardError=journal diff --git a/ansible/roles/healthcheck/templates/healthcheck.sh.j2 b/ansible/roles/healthcheck/templates/healthcheck.sh.j2 index 92c9d27..6667878 100644 --- a/ansible/roles/healthcheck/templates/healthcheck.sh.j2 +++ b/ansible/roles/healthcheck/templates/healthcheck.sh.j2 @@ -11,6 +11,8 @@ set -uo pipefail LOG_FILE="{{ healthcheck_log_dir }}/{{ healthcheck_name }}.log" PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" PUSH_TOKEN="${HEALTHCHECK_PUSH_TOKEN:-}" +# Set only by checks that report several results; see healthcheck_push_base. +PUSH_BASE="${HEALTHCHECK_PUSH_BASE:-}" log() { echo "$(date '+%Y-%m-%d %H:%M:%S') - $*" >> "$LOG_FILE"; } @@ -40,6 +42,18 @@ report() { fi } +# Report to an arbitrary endpoint key under PUSH_BASE. Used by checks that +# produce one result per item rather than a single verdict. +report_key() { + local key="$1" success="$2" message="$3" + [ -n "$PUSH_BASE" ] || return 0 + local encoded + encoded=$(printf '%s' "$message" | sed 's/%/%25/g; s/ /%20/g; s/&/%26/g; s/+/%2B/g; s/#/%23/g') + curl -s -o /dev/null --max-time 15 --retry 2 --retry-delay 3 -X POST \ + -H "Authorization: Bearer ${PUSH_TOKEN}" \ + "${PUSH_BASE}/${key}/external?success=${success}&error=${encoded}" 2>/dev/null || true +} + # ── the check ──────────────────────────────────────────────────────────────── # # The blank line before the closing brace below is load-bearing. Jinja strips an