--- # Host-level monitoring for the whole estate, reported to Gatus. # # Every check here PUSHES. Gatus never reaches out, which matters because nodito # and its VMs sit behind NAT, and because four of the five checks are internal # state with no pollable surface at all - disk usage, CPU temperature, ZFS pool # health and UPS mains status cannot be observed from outside the machine. # # Liveness is a push too, and that is a choice rather than a limitation. A # heartbeat proves the host is running AND can reach the internet; an ICMP probe # from one vantage point only proves it answers pings from there. And because # Gatus alerts when a heartbeat window expires, a check that stops running # raises the alarm by itself - a dead timer looks exactly like a dead host, # which is the correct reading. # # Each host has ONE bearer token, shared across its own checks: a token can only # write results for that host's endpoints, so a compromised host can lie about # itself, which it could do anyway. # # The push URL must use Gatus's own key format (config/key/key.go): # key = sanitize(group) + "_" + sanitize(name) # where sanitize lowercases and replaces / _ . , space # + & with "-". So # knots_box_local becomes knots-box-local in the URL but stays readable in the # name. host_key below is the Jinja equivalent; do not hand-write these. # ───────────────────────────────────────────────────────────────────────────── # Register everything with Gatus. # # This play runs FIRST on purpose. Gatus reloads its config within 30s, and the # host plays below take minutes, so every endpoint exists before its first push # arrives. Registering afterwards would 404 every first report. # # Heartbeat windows are several times the check interval, so one missed run - a # slow apt run, a reboot - does not raise an alarm, but a check that has # genuinely stopped does. # ───────────────────────────────────────────────────────────────────────────── # ───────────────────────────────────────────────────────────────────────────── # Alerting thresholds, and why they differ by check type. # # `failure-threshold` counts CONSECUTIVE failures, but "consecutive" means a # different amount of wall-clock time per check: # # push/heartbeat endpoints a failure is produced once per heartbeat window # pulled endpoints a failure is produced once per interval # # So the default of 3 would mean 33 minutes on an 11m heartbeat and over a day # on a 7h one - and the heartbeat window ALREADY encodes the tolerance. An 11m # window on a 5-minute push is precisely "one missed push forgiven"; stacking a # threshold of 3 on top triples a tolerance that was already chosen. # # Hence: push endpoints alert on the FIRST heartbeat failure. Pulled endpoints # have no built-in tolerance, so the threshold is where it belongs for them. # ───────────────────────────────────────────────────────────────────────────── - name: Register the host checks with Gatus hosts: observability become: yes vars: monitored: "{{ groups['managed'] | sort }}" tasks: - name: Build the liveness endpoint list ansible.builtin.set_fact: liveness_endpoints: "{{ liveness_endpoints | default([]) + [{ 'name': item, 'group': 'liveness', 'token': gatus_push_tokens[item], 'heartbeat': '11m'}] }}" loop: "{{ monitored }}" - name: Build the disk endpoint list ansible.builtin.set_fact: disk_endpoints: "{{ disk_endpoints | default([]) + [{ 'name': item, 'group': 'disk', 'token': gatus_push_tokens[item], 'heartbeat': '7h'}] }}" loop: "{{ monitored }}" - name: Register liveness endpoints ansible.builtin.include_role: name: gatus_endpoint vars: gatus_endpoint_default_alerts: - type: signal # 1, not 3: the heartbeat window is the tolerance. See the note above. failure-threshold: 1 success-threshold: 2 send-on-resolved: true minimum-reminder-interval: 6h gatus_endpoint_name: liveness gatus_endpoint_external: "{{ liveness_endpoints }}" - name: Register disk endpoints ansible.builtin.include_role: name: gatus_endpoint vars: gatus_endpoint_default_alerts: - type: signal # 1, not 3: the heartbeat window is the tolerance. See the note above. failure-threshold: 1 success-threshold: 2 send-on-resolved: true minimum-reminder-interval: 6h gatus_endpoint_name: disk gatus_endpoint_external: "{{ disk_endpoints }}" - name: Register the hypervisor endpoints ansible.builtin.include_role: name: gatus_endpoint vars: gatus_endpoint_default_alerts: - type: signal # 1, not 3: the heartbeat window is the tolerance. See the note above. failure-threshold: 1 success-threshold: 2 send-on-resolved: true minimum-reminder-interval: 6h gatus_endpoint_name: hypervisor gatus_endpoint_external: - name: cpu group: hypervisor token: "{{ gatus_push_tokens['nodito'] }}" heartbeat: "11m" - name: zfs group: hypervisor token: "{{ gatus_push_tokens['nodito'] }}" heartbeat: "7h" - name: ups group: hypervisor token: "{{ gatus_push_tokens['nodito'] }}" heartbeat: "11m" - name: Deploy host liveness and disk checks hosts: managed become: yes vars: gatus_api: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints" host_key: "{{ inventory_hostname | lower | regex_replace('[/_.,# +&]', '-') }}" host_token: "{{ gatus_push_tokens[inventory_hostname] }}" tasks: - name: Is the host up? ansible.builtin.include_role: name: healthcheck vars: healthcheck_name: liveness healthcheck_description: "Liveness heartbeat for {{ inventory_hostname }}" healthcheck_check: liveness healthcheck_interval: "5min" healthcheck_boot_delay: "1min" healthcheck_push_url: "{{ gatus_api }}/liveness_{{ host_key }}/external" healthcheck_push_token: "{{ host_token }}" - name: Is the disk packed? ansible.builtin.include_role: name: healthcheck vars: healthcheck_name: disk-usage healthcheck_description: "Disk usage for {{ inventory_hostname }}" healthcheck_check: disk-usage # Every 6h rather than daily. Disk usage itself moves slowly, but the # heartbeat can only be as tight as the push frequency - a daily push # forces a >24h window, and a stuck check then hides for a day and a # half. Six-hourly buys a 7h window. RandomizedDelaySec spreads the # hosts so twelve boxes do not all report in the same second. healthcheck_on_calendar: "*-*-* 00/6:00:00" healthcheck_randomized_delay: "900" healthcheck_boot_delay: "5min" healthcheck_push_url: "{{ gatus_api }}/disk_{{ host_key }}/external" healthcheck_push_token: "{{ host_token }}" - name: Deploy the hypervisor-only checks hosts: hypervisor become: yes vars: gatus_api: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints" host_token: "{{ gatus_push_tokens[inventory_hostname] }}" tasks: - name: Is the CPU hot? ansible.builtin.include_role: name: healthcheck vars: healthcheck_name: cpu-temp healthcheck_description: "CPU temperature for {{ inventory_hostname }}" healthcheck_check: cpu-temp healthcheck_packages: [curl, lm-sensors] healthcheck_interval: "5min" healthcheck_push_url: "{{ gatus_api }}/hypervisor_cpu/external" healthcheck_push_token: "{{ host_token }}" - name: Is ZFS broken? ansible.builtin.include_role: name: healthcheck vars: healthcheck_name: zfs-health healthcheck_description: "ZFS pool health for {{ zfs_pool_name }}" healthcheck_check: zfs-health healthcheck_packages: [curl, jq] healthcheck_zfs_pool: "{{ zfs_pool_name }}" healthcheck_on_calendar: "*-*-* 00/6:20:00" healthcheck_boot_delay: "10min" healthcheck_push_url: "{{ gatus_api }}/hypervisor_zfs/external" healthcheck_push_token: "{{ host_token }}" - name: Is the UPS online? ansible.builtin.include_role: name: healthcheck vars: healthcheck_name: ups-status healthcheck_description: "UPS mains status for {{ ups_name }}" healthcheck_check: ups-status healthcheck_ups_name: "{{ ups_name }}" healthcheck_interval: "5min" healthcheck_push_url: "{{ gatus_api }}/hypervisor_ups/external" healthcheck_push_token: "{{ host_token }}"