63 lines
2.4 KiB
YAML
63 lines
2.4 KiB
YAML
|
|
---
|
||
|
|
# Everything here answers "is Fulcrum healthy" and records the answer. The
|
||
|
|
# Uptime Kuma specifics that used to follow — an embedded Python script creating
|
||
|
|
# monitors over the API, a /tmp credentials file, push-URL extraction and a
|
||
|
|
# systemd Environment= rewrite — are gone. Where it reports is now one variable,
|
||
|
|
# healthcheck_push_url. See the role README.
|
||
|
|
- name: Create Fulcrum health check script
|
||
|
|
ansible.builtin.template:
|
||
|
|
src: healthcheck.sh.j2
|
||
|
|
dest: /usr/local/bin/fulcrum-healthcheck-push.sh
|
||
|
|
owner: root
|
||
|
|
group: root
|
||
|
|
mode: '0755'
|
||
|
|
validate: "bash -n %s"
|
||
|
|
|
||
|
|
- name: Create systemd service for Fulcrum health check
|
||
|
|
ansible.builtin.template:
|
||
|
|
src: healthcheck.service.j2
|
||
|
|
dest: /etc/systemd/system/fulcrum-healthcheck.service
|
||
|
|
owner: root
|
||
|
|
group: root
|
||
|
|
mode: '0644'
|
||
|
|
|
||
|
|
- name: Create systemd timer for Fulcrum health check
|
||
|
|
ansible.builtin.template:
|
||
|
|
src: healthcheck.timer.j2
|
||
|
|
dest: /etc/systemd/system/fulcrum-healthcheck.timer
|
||
|
|
owner: root
|
||
|
|
group: root
|
||
|
|
mode: '0644'
|
||
|
|
|
||
|
|
- name: Reload systemd daemon for health check
|
||
|
|
systemd:
|
||
|
|
daemon_reload: yes
|
||
|
|
|
||
|
|
# state: restarted, not started. The hand-written timer had got itself stuck
|
||
|
|
# `active` with no next elapse and had not fired since 2026-02-17; `started` on
|
||
|
|
# an already-active timer is a no-op and would have left it stuck. Restarting
|
||
|
|
# re-arms it. See the note in healthcheck.timer.j2.
|
||
|
|
- name: Enable and restart the Fulcrum health check timer
|
||
|
|
systemd:
|
||
|
|
name: fulcrum-healthcheck.timer
|
||
|
|
enabled: yes
|
||
|
|
state: restarted
|
||
|
|
daemon_reload: yes
|
||
|
|
|
||
|
|
# Run the check once, which is both a smoke test and the thing that actually
|
||
|
|
# arms the timer.
|
||
|
|
#
|
||
|
|
# This timer is OnBootSec + OnUnitActiveSec with no OnCalendar. OnBootSec is
|
||
|
|
# monotonic and had long since elapsed; OnUnitActiveSec schedules relative to the
|
||
|
|
# SERVICE last being active, and the service had not run since 2026-02-17 — so
|
||
|
|
# there was no reference to schedule from and the timer sat `active` and
|
||
|
|
# `enabled` with NextElapseUSecMonotonic=infinity. Restarting the timer alone
|
||
|
|
# does not supply that reference; running the service does.
|
||
|
|
#
|
||
|
|
# (Diagnosing this is easy to get wrong: NextElapseUSecRealtime is always empty
|
||
|
|
# for a monotonic timer, so it looks broken even when it is fine. Read
|
||
|
|
# NextElapseUSecMonotonic, or just use `systemctl list-timers`.)
|
||
|
|
- name: Run the Fulcrum health check once to arm the timer
|
||
|
|
command: systemctl start fulcrum-healthcheck.service
|
||
|
|
changed_when: false
|