--- # Is each systemd-deployed service actually running? # # Every 5 minutes, with an 11-minute Gatus heartbeat - one missed run before # it alarms, so a reboot or a slow check does not page anyone, but a host that # stops reporting does. # # This closes the gap that let a real bug run unnoticed: a backup script left # forgejo, lnbits, headscale and memos stopped, and NOTHING caught it. The dumps # exited 0, the artefacts were correct, the deploy said failed=0, and liveness # only proves the HOST is up - not that anything on it is serving. # # One endpoint PER UNIT, not per host. A host running four services needs four # endpoints, or a single red light says "something on vipy is down" without # saying which - and that is the question you actually have at 3am. But only ONE # timer per host: the check iterates that host's units and pushes a result for # each, the same way check-backups.sh reports per source. Four units on vipy # would otherwise mean four scripts, four services and four timers. # # Which units each host runs is in host_vars//main.yml as # monitored_services, because "what runs here" is a property of the machine. # # Keys are host-qualified because unit names collide - caddy runs on four # machines. Gatus computes sanitize(group)_sanitize(name), so group "services" # and name "vipy/caddy" give services_vipy-caddy. # ───────────────────────────────────────────────────────────────────────────── # Register one endpoint per unit. Runs first: Gatus reloads within 30s, and the # host play above takes minutes, so every endpoint exists before its first push. # ───────────────────────────────────────────────────────────────────────────── # ───────────────────────────────────────────────────────────────────────────── # Alerting thresholds, and why they differ by check type. # # `failure-threshold` counts CONSECUTIVE failures, but "consecutive" means a # different amount of wall-clock time per check: # # push/heartbeat endpoints a failure is produced once per heartbeat window # pulled endpoints a failure is produced once per interval # # So the default of 3 would mean 33 minutes on an 11m heartbeat and over a day # on a 7h one - and the heartbeat window ALREADY encodes the tolerance. An 11m # window on a 5-minute push is precisely "one missed push forgiven"; stacking a # threshold of 3 on top triples a tolerance that was already chosen. # # Hence: push endpoints alert on the FIRST heartbeat failure. Pulled endpoints # have no built-in tolerance, so the threshold is where it belongs for them. # ───────────────────────────────────────────────────────────────────────────── - name: Register the service checks with Gatus hosts: observability become: yes tasks: # Two plain steps rather than one clever expression: first collect which # units each host declares, then flatten that into endpoints. - name: Collect the units each host declares ansible.builtin.set_fact: host_units: "{{ host_units | default([]) + [{'host': item, 'units': hostvars[item].monitored_services}] }}" loop: "{{ groups['managed'] | sort }}" when: hostvars[item].monitored_services | default([]) | length > 0 - name: Build one endpoint per unit ansible.builtin.set_fact: service_endpoints: "{{ service_endpoints | default([]) + [{ 'name': (item.0.host | lower | regex_replace('[/_.,# +&]', '-')) ~ '/' ~ item.1, 'group': 'services', 'token': gatus_push_tokens[item.0.host], 'heartbeat': '11m'}] }}" loop: "{{ host_units | subelements('units') }}" - name: Register the service endpoints ansible.builtin.include_role: name: gatus_endpoint vars: gatus_endpoint_default_alerts: - type: signal # 1, not 3: the heartbeat window is the tolerance. See the note above. failure-threshold: 1 success-threshold: 2 send-on-resolved: true minimum-reminder-interval: 6h gatus_endpoint_name: services gatus_endpoint_external: "{{ service_endpoints }}" - name: Monitor systemd services on every host that has them hosts: managed become: yes vars: gatus_api: "https://{{ subdomains.gatus }}.{{ root_domain }}/api/v1/endpoints" host_key: "{{ inventory_hostname | lower | regex_replace('[/_.,# +&]', '-') }}" tasks: - name: Is every deployed service running? ansible.builtin.include_role: name: healthcheck vars: healthcheck_name: service-health healthcheck_description: "systemd services on {{ inventory_hostname }}" healthcheck_check: systemd-units healthcheck_units: "{{ monitored_services }}" healthcheck_units_key_prefix: "services_{{ host_key }}" healthcheck_interval: "5min" healthcheck_boot_delay: "2min" # The per-unit results go to keys under this collection; the role's own # single-result push is unused here, so only the base is set. healthcheck_push_base: "{{ gatus_api }}" healthcheck_push_token: "{{ gatus_push_tokens[inventory_hostname] }}" when: monitored_services | default([]) | length > 0