--- # One health check: a script, a systemd service, a timer, and an optional push. # # The exit code is the answer and systemd keeps it: # systemctl is-failed -healthcheck.service # Reporting anywhere else is optional and generic. Point healthcheck_push_url at # Gatus, or at whatever replaces it, or at nothing. healthcheck_name: "" # e.g. disk-usage -> disk-usage-healthcheck healthcheck_description: "" # The check itself. Pick ONE: # healthcheck_check: a template under templates/checks/ (without .sh.j2) # healthcheck_command: a shell one-liner that exits 0 for healthy healthcheck_check: "" healthcheck_command: "" # systemd timer. OnUnitActiveSec unless healthcheck_on_calendar is set. healthcheck_interval: "5min" healthcheck_on_calendar: "" healthcheck_boot_delay: "2min" # ── Reporting ──────────────────────────────────────────────────────────────── # Gatus external endpoints: # POST {url}?success=true|false&error=... # Authorization: Bearer {token} # Empty url = check and log only, which is a valid state and not an error. healthcheck_push_url: "" healthcheck_push_token: "" # Some checks report MORE THAN ONE result - a host with four systemd services # needs four endpoints, or a single red light cannot tell you which one died. # Those set healthcheck_push_base to the endpoints COLLECTION and the check body # appends each key itself, the same way check-backups.sh reports per source. healthcheck_push_base: "" # Units for the systemd-units check. Each becomes its own Gatus endpoint. healthcheck_units: [] # Prefix for the per-unit endpoint keys, e.g. "services_vipy" -> services_vipy-caddy. healthcheck_units_key_prefix: "" healthcheck_script_dir: /usr/local/bin healthcheck_log_dir: /var/log/healthchecks # Per-check knobs, consumed by the templates under checks/ healthcheck_disk_threshold: 85 # percent healthcheck_cpu_temp_threshold: 80 # celsius healthcheck_zfs_pool: "" # ZFS degrades badly once a pool passes roughly 80% - allocation gets slow and # fragmentation becomes hard to undo, and unlike a normal filesystem you cannot # simply delete your way back to good performance. So this alarms well before # the pool is actually out of space. healthcheck_zfs_capacity_threshold: 80 healthcheck_ups_name: ""