diff --git a/ansible/infra/409_remove_legacy_monitoring.yml b/ansible/infra/409_remove_legacy_monitoring.yml new file mode 100644 index 0000000..8f57e46 --- /dev/null +++ b/ansible/infra/409_remove_legacy_monitoring.yml @@ -0,0 +1,109 @@ +--- +# Remove the Uptime-Kuma-era monitoring that 400/401/402 replaced. +# +# Deleting the playbooks that installed these is NOT enough: the units are on +# the hosts, enabled, and keep firing regardless of what the repo says. Two of +# them still push to https://uptime.contrapeso.xyz every 15 minutes. A playbook +# that is deleted without a cleanup leaves its output running forever, with +# nothing in the repo left to explain it. +# +# What replaced what, all verified against the deployed scripts before removal: +# +# disk-usage-monitor -> disk-usage-healthcheck (infra/400) +# The old one checked ONLY "/" at 80%. The replacement walks every real +# filesystem, excluding tmpfs/devtmpfs/squashfs/overlay, at 85%. Strictly +# more coverage, so nothing is lost. +# +# system-healthcheck -> liveness-healthcheck (infra/400) +# The old script computed uptime and pushed. That is exactly a liveness +# heartbeat and nothing more. +# +# nodito-cpu-temp-monitor -> cpu-temp-healthcheck (infra/400) +# zfs-health-monitor -> zfs-health-healthcheck (infra/400) +# The ZFS check logic was ported verbatim - same five conditions - so only +# the reporting transport changed. +# +# NOT removed, because they are not monitoring: +# zfs-monthly-scrub.{timer,service} the actual scrub (infra/nodito/32) +# pull-backups, check-backups the backup machinery (playbooks/backups) +# +# This play is idempotent and kept permanently rather than run once and deleted: +# on a host that never had these it does nothing, and it guarantees a rebuilt or +# restored machine cannot quietly bring them back. + +- name: Remove the legacy Uptime Kuma monitoring units + hosts: managed + become: yes + + vars: + legacy_units: + - disk-usage-monitor + - system-healthcheck + - nodito-cpu-temp-monitor + - zfs-health-monitor + legacy_dirs: + - /opt/disk-monitoring + - /opt/system-healthcheck + - /opt/nodito-monitoring + - /opt/zfs-monitoring + + tasks: + - name: Find which legacy units exist here + ansible.builtin.stat: + path: "/etc/systemd/system/{{ item.0 }}.{{ item.1 }}" + register: legacy_unit_files + loop: "{{ legacy_units | product(['timer', 'service']) | list }}" + + # Stop and disable BEFORE deleting the unit file: systemd cannot disable a + # unit whose file has already gone, which would leave a dangling symlink in + # multi-user.target.wants and a warning on every daemon-reload. + - name: Stop and disable the legacy units + ansible.builtin.systemd: + name: "{{ item.item.0 }}.{{ item.item.1 }}" + state: stopped + enabled: no + loop: "{{ legacy_unit_files.results }}" + loop_control: + label: "{{ item.item.0 }}.{{ item.item.1 }}" + when: item.stat.exists + failed_when: false + + - name: Remove the legacy unit files + ansible.builtin.file: + path: "/etc/systemd/system/{{ item.item.0 }}.{{ item.item.1 }}" + state: absent + loop: "{{ legacy_unit_files.results }}" + loop_control: + label: "{{ item.item.0 }}.{{ item.item.1 }}" + when: item.stat.exists + + - name: Reload systemd + ansible.builtin.systemd: + daemon_reload: yes + + - name: Remove the legacy monitoring scripts and their logs + ansible.builtin.file: + path: "{{ item }}" + state: absent + loop: "{{ legacy_dirs }}" + + # An orphan predating all of this: mode 0644, not executable, referenced by + # no unit and no cron entry, pushing to a Kuma monitor. Superseded by + # ups-status-healthcheck. + - name: Remove the orphaned hand-written UPS heartbeat + ansible.builtin.file: + path: /usr/local/bin/ups-heartbeat.sh + state: absent + + - name: Confirm nothing still pushes to Uptime Kuma + ansible.builtin.shell: >- + grep -rl "uptime.contrapeso.xyz" /etc/systemd/system /usr/local/bin /opt 2>/dev/null || true + register: kuma_refs + changed_when: false + + - name: Report any remaining references + ansible.builtin.debug: + msg: >- + {{ 'clean - nothing references Uptime Kuma' + if kuma_refs.stdout | trim | length == 0 + else 'STILL REFERENCING KUMA: ' ~ kuma_refs.stdout_lines | join(', ') }} diff --git a/ansible/infra/410_disk_usage_alerts.yml b/ansible/infra/410_disk_usage_alerts.yml deleted file mode 100644 index b286add..0000000 --- a/ansible/infra/410_disk_usage_alerts.yml +++ /dev/null @@ -1,338 +0,0 @@ -# ═════════════════════════════════════════════════════════════════════════════ -# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. -# -# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username -# and uptime_kuma_password were removed from the vault, so the "Validate Uptime -# Kuma configuration" assert stops it before anything is installed or changed. -# -# It is kept because the CHECK LOGIC is the durable part — what gets measured, -# the thresholds, and the systemd timer plumbing. When something replaces Uptime -# Kuma, only the push transport needs rewriting; the rest still applies. -# -# What was being monitored: archive/uptime_kuma/MONITORS.md -# ═════════════════════════════════════════════════════════════════════════════ -- name: Deploy Disk Usage Monitoring - hosts: managed - become: yes - - vars: - disk_usage_threshold_percent: 80 - disk_check_interval_minutes: 15 - monitored_mount_point: "/" - monitoring_script_dir: /opt/disk-monitoring - monitoring_script_path: "{{ monitoring_script_dir }}/disk_usage_monitor.sh" - log_file: "{{ monitoring_script_dir }}/disk_usage_monitor.log" - systemd_service_name: disk-usage-monitor - # Uptime Kuma configuration (auto-configured from group_vars/all/) - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - - name: Validate Uptime Kuma configuration - assert: - that: - - uptime_kuma_api_url is defined - - uptime_kuma_api_url != "" - - uptime_kuma_username is defined - - uptime_kuma_username != "" - - uptime_kuma_password is defined - - uptime_kuma_password != "" - fail_msg: "uptime_kuma_api_url, uptime_kuma_username and uptime_kuma_password must be set" - - - name: Get hostname for monitor identification - command: hostname - register: host_name - changed_when: false - - - name: Set monitor name and group based on hostname and mount point - set_fact: - monitor_name: "disk-usage-{{ host_name.stdout }}-{{ monitored_mount_point | replace('/', 'root') }}" - monitor_friendly_name: "Disk Usage: {{ host_name.stdout }} ({{ monitored_mount_point }})" - uptime_kuma_monitor_group: "{{ host_name.stdout }} - infra" - - - name: Create Uptime Kuma monitor setup script - copy: - dest: /tmp/setup_uptime_kuma_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - ntfy_topic = sys.argv[8] if len(sys.argv) > 8 else "alerts" - - api = UptimeKumaApi(api_url, timeout=60, wait_events=2.0) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Get all notifications and find ntfy notification - notifications = api.get_notifications() - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - # Find or create group - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name=group_name) - # Refresh to get the full group object with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - # Find or create/update push monitor - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': True, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - monitor = api.edit_monitor(existing_monitor['id'], **monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - else: - monitor_result = api.add_monitor(**monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - # Output result as JSON - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma monitor setup script - command: > - {{ ansible_playbook_python }} - /tmp/setup_uptime_kuma_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "{{ uptime_kuma_monitor_group }}" - "{{ monitor_name }}" - "{{ monitor_friendly_name }} - Alerts when usage exceeds {{ disk_usage_threshold_percent }}%" - "{{ (disk_check_interval_minutes * 60) + 60 }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL and monitor ID as facts - set_fact: - uptime_kuma_disk_usage_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - uptime_kuma_monitor_id: "{{ monitor_info_parsed.monitor_id }}" - - - name: Install required packages for disk monitoring - package: - name: - - curl - state: present - - - name: Create monitoring script directory - file: - path: "{{ monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create disk usage monitoring script - copy: - dest: "{{ monitoring_script_path }}" - content: | - #!/bin/bash - - # Disk Usage Monitoring Script - # Monitors disk usage and sends alerts to Uptime Kuma - # Mode: "No news is good news" - only sends alerts when disk usage is HIGH - - LOG_FILE="{{ log_file }}" - USAGE_THRESHOLD="{{ disk_usage_threshold_percent }}" - UPTIME_KUMA_URL="{{ uptime_kuma_disk_usage_push_url }}" - MOUNT_POINT="{{ monitored_mount_point }}" - - # Function to log messages - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - # Function to get disk usage percentage - get_disk_usage() { - local mount_point="$1" - local usage="" - - # Get disk usage percentage (without % sign) - usage=$(df -h "$mount_point" 2>/dev/null | awk 'NR==2 {gsub(/%/, "", $5); print $5}') - - if [ -z "$usage" ]; then - log_message "ERROR: Could not read disk usage for $mount_point" - return 1 - fi - - echo "$usage" - } - - # Function to get disk usage details - get_disk_details() { - local mount_point="$1" - df -h "$mount_point" 2>/dev/null | awk 'NR==2 {print "Used: "$3" / Total: "$2" ("$5" full)"}' - } - - # Function to send alert to Uptime Kuma when disk usage exceeds threshold - # With upside-down mode enabled, sending status=up will trigger an alert - send_uptime_kuma_alert() { - local usage="$1" - local details="$2" - local message="DISK FULL WARNING: ${MOUNT_POINT} is ${usage}% full (Threshold: ${USAGE_THRESHOLD}%) - ${details}" - - log_message "ALERT: $message" - - # Send push notification to Uptime Kuma with status=up - # In upside-down mode, status=up is treated as down/alert - response=$(curl -s -w "\n%{http_code}" -G \ - --data-urlencode "status=up" \ - --data-urlencode "msg=$message" \ - "$UPTIME_KUMA_URL" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Alert sent successfully to Uptime Kuma (HTTP $http_code)" - else - log_message "ERROR: Failed to send alert to Uptime Kuma (HTTP $http_code)" - fi - } - - # Main monitoring logic - main() { - log_message "Starting disk usage check for $MOUNT_POINT" - - # Get current disk usage - current_usage=$(get_disk_usage "$MOUNT_POINT") - - if [ $? -ne 0 ] || [ -z "$current_usage" ]; then - log_message "ERROR: Could not read disk usage" - exit 1 - fi - - # Get disk details - disk_details=$(get_disk_details "$MOUNT_POINT") - - log_message "Current disk usage: ${current_usage}% - $disk_details" - - # Check if usage exceeds threshold - if [ "$current_usage" -gt "$USAGE_THRESHOLD" ]; then - log_message "WARNING: Disk usage ${current_usage}% exceeds threshold ${USAGE_THRESHOLD}%" - send_uptime_kuma_alert "$current_usage" "$disk_details" - else - log_message "Disk usage is within normal range - no alert needed (no news is good news)" - fi - } - - # Run main function - main - owner: root - group: root - mode: '0755' - - - name: Create systemd service for disk usage monitoring - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.service" - content: | - [Unit] - Description=Disk Usage Monitor - After=network.target - - [Service] - Type=oneshot - ExecStart={{ monitoring_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for disk usage monitoring - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.timer" - content: | - [Unit] - Description=Run Disk Usage Monitor every {{ disk_check_interval_minutes }} minute(s) - Requires={{ systemd_service_name }}.service - - [Timer] - OnBootSec={{ disk_check_interval_minutes }}min - OnUnitActiveSec={{ disk_check_interval_minutes }}min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start disk usage monitoring timer - systemd: - name: "{{ systemd_service_name }}.timer" - enabled: yes - state: started - - - name: Test disk usage monitoring script - command: "{{ monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "Disk usage monitoring script failed to execute properly" - - - name: Clean up temporary Uptime Kuma setup script - file: - path: /tmp/setup_uptime_kuma_monitor.py - state: absent - delegate_to: localhost - become: no diff --git a/ansible/infra/420_system_healthcheck.yml b/ansible/infra/420_system_healthcheck.yml deleted file mode 100644 index 9813d0c..0000000 --- a/ansible/infra/420_system_healthcheck.yml +++ /dev/null @@ -1,320 +0,0 @@ -# ═════════════════════════════════════════════════════════════════════════════ -# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. -# -# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username -# and uptime_kuma_password were removed from the vault, so the "Validate Uptime -# Kuma configuration" assert stops it before anything is installed or changed. -# -# It is kept because the CHECK LOGIC is the durable part — what gets measured, -# the thresholds, and the systemd timer plumbing. When something replaces Uptime -# Kuma, only the push transport needs rewriting; the rest still applies. -# -# What was being monitored: archive/uptime_kuma/MONITORS.md -# ═════════════════════════════════════════════════════════════════════════════ -- name: Deploy System Healthcheck Monitoring - hosts: managed - become: yes - - vars: - healthcheck_interval_seconds: 60 # Send healthcheck every 60 seconds (1 minute) - healthcheck_timeout_seconds: 90 # Uptime Kuma should alert if no ping received within 90s - healthcheck_retries: 1 # Number of retries before alerting - monitoring_script_dir: /opt/system-healthcheck - monitoring_script_path: "{{ monitoring_script_dir }}/system_healthcheck.sh" - log_file: "{{ monitoring_script_dir }}/system_healthcheck.log" - systemd_service_name: system-healthcheck - # Uptime Kuma configuration (auto-configured from group_vars/all/) - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - - name: Validate Uptime Kuma configuration - assert: - that: - - uptime_kuma_api_url is defined - - uptime_kuma_api_url != "" - - uptime_kuma_username is defined - - uptime_kuma_username != "" - - uptime_kuma_password is defined - - uptime_kuma_password != "" - fail_msg: "uptime_kuma_api_url, uptime_kuma_username and uptime_kuma_password must be set" - - - name: Get hostname for monitor identification - command: hostname - register: host_name - changed_when: false - - - name: Set monitor name and group based on hostname - set_fact: - monitor_name: "system-healthcheck-{{ host_name.stdout }}" - monitor_friendly_name: "System Healthcheck: {{ host_name.stdout }}" - uptime_kuma_monitor_group: "{{ host_name.stdout }} - infra" - - - name: Create Uptime Kuma monitor setup script - copy: - dest: /tmp/setup_uptime_kuma_healthcheck_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - retries = int(sys.argv[8]) - ntfy_topic = sys.argv[9] if len(sys.argv) > 9 else "alerts" - - api = UptimeKumaApi(api_url, timeout=120, wait_events=2.0) - api.login(username, password) - - # Get all monitors - monitors = api.get_monitors() - - # Get all notifications and find ntfy notification - notifications = api.get_notifications() - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - # Find or create group - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - group_result = api.add_monitor(type='group', name=group_name) - # Refresh to get the full group object with id - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - # Find or create/update push monitor - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': False, # Normal mode: receiving pings = healthy - 'maxretries': retries, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - monitor = api.edit_monitor(existing_monitor['id'], **monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - else: - monitor_result = api.add_monitor(**monitor_data) - # Refresh to get the full monitor object with pushToken - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - # Output result as JSON - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma monitor setup script - command: > - {{ ansible_playbook_python }} - /tmp/setup_uptime_kuma_healthcheck_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "{{ uptime_kuma_monitor_group }}" - "{{ monitor_name }}" - "{{ monitor_friendly_name }} - Regular healthcheck ping every {{ healthcheck_interval_seconds }}s" - "{{ healthcheck_timeout_seconds }}" - "{{ healthcheck_retries }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL and monitor ID as facts - set_fact: - uptime_kuma_healthcheck_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - uptime_kuma_monitor_id: "{{ monitor_info_parsed.monitor_id }}" - - - name: Install required packages for healthcheck monitoring - package: - name: - - curl - state: present - - - name: Create monitoring script directory - file: - path: "{{ monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create system healthcheck script - copy: - dest: "{{ monitoring_script_path }}" - content: | - #!/bin/bash - - # System Healthcheck Script - # Sends regular heartbeat pings to Uptime Kuma - # This ensures the system is running and able to communicate - - LOG_FILE="{{ log_file }}" - UPTIME_KUMA_URL="{{ uptime_kuma_healthcheck_push_url }}" - HOSTNAME=$(hostname) - - # Function to log messages - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - # Function to send healthcheck ping to Uptime Kuma - send_healthcheck() { - local uptime_seconds=$(awk '{print int($1)}' /proc/uptime) - local uptime_days=$((uptime_seconds / 86400)) - local uptime_hours=$(((uptime_seconds % 86400) / 3600)) - local uptime_minutes=$(((uptime_seconds % 3600) / 60)) - - local message="System healthy - Uptime: ${uptime_days}d ${uptime_hours}h ${uptime_minutes}m" - - log_message "Sending healthcheck ping: $message" - - # Send push notification to Uptime Kuma with status=up - encoded_message=$(printf '%s\n' "$message" | sed 's/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') - response=$(curl -s -w "\n%{http_code}" "$UPTIME_KUMA_URL?status=up&msg=$encoded_message" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Healthcheck ping sent successfully (HTTP $http_code)" - else - log_message "ERROR: Failed to send healthcheck ping (HTTP $http_code)" - return 1 - fi - } - - # Main healthcheck logic - main() { - log_message "Starting system healthcheck for $HOSTNAME" - - # Send healthcheck ping - if send_healthcheck; then - log_message "Healthcheck completed successfully" - else - log_message "ERROR: Healthcheck failed" - exit 1 - fi - } - - # Run main function - main - owner: root - group: root - mode: '0755' - - - name: Create systemd service for system healthcheck - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.service" - content: | - [Unit] - Description=System Healthcheck Monitor - After=network.target - - [Service] - Type=oneshot - ExecStart={{ monitoring_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for system healthcheck - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.timer" - content: | - [Unit] - Description=Run System Healthcheck every minute - Requires={{ systemd_service_name }}.service - - [Timer] - OnBootSec=30sec - OnUnitActiveSec={{ healthcheck_interval_seconds }}sec - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start system healthcheck timer - systemd: - name: "{{ systemd_service_name }}.timer" - enabled: yes - state: started - - - name: Test system healthcheck script - command: "{{ monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "System healthcheck script failed to execute properly" - - - name: Display monitor information - debug: - msg: | - ✓ System healthcheck monitoring deployed successfully! - - Monitor Name: {{ monitor_friendly_name }} - Monitor Group: {{ uptime_kuma_monitor_group }} - Healthcheck Interval: Every {{ healthcheck_interval_seconds }} seconds (1 minute) - Timeout: {{ healthcheck_timeout_seconds }} seconds (90s) - Retries: {{ healthcheck_retries }} - - The system will send a heartbeat ping every minute. - Uptime Kuma will alert if no ping is received within 90 seconds (with 1 retry). - - - name: Clean up temporary Uptime Kuma setup script - file: - path: /tmp/setup_uptime_kuma_healthcheck_monitor.py - state: absent - delegate_to: localhost - become: no - diff --git a/ansible/infra/430_cpu_temp_alerts.yml b/ansible/infra/430_cpu_temp_alerts.yml deleted file mode 100644 index 2e00bdb..0000000 --- a/ansible/infra/430_cpu_temp_alerts.yml +++ /dev/null @@ -1,324 +0,0 @@ -# ═════════════════════════════════════════════════════════════════════════════ -# DEPRECATED — Uptime Kuma was decommissioned on 2026-09-11. -# -# This play WILL FAIL if run as-is, and that is deliberate: uptime_kuma_username -# and uptime_kuma_password were removed from the vault, so the "Validate Uptime -# Kuma configuration" assert stops it before anything is installed or changed. -# -# It is kept because the CHECK LOGIC is the durable part — what gets measured, -# the thresholds, and the systemd timer plumbing. When something replaces Uptime -# Kuma, only the push transport needs rewriting; the rest still applies. -# -# What was being monitored: archive/uptime_kuma/MONITORS.md -# ═════════════════════════════════════════════════════════════════════════════ -- name: Deploy CPU Temperature Monitoring - hosts: hypervisor - become: yes - - vars: - temp_threshold_celsius: 80 - temp_check_interval_minutes: 1 - monitoring_script_dir: /opt/nodito-monitoring - monitoring_script_path: "{{ monitoring_script_dir }}/cpu_temp_monitor.sh" - log_file: "{{ monitoring_script_dir }}/cpu_temp_monitor.log" - systemd_service_name: nodito-cpu-temp-monitor - uptime_kuma_api_url: "https://{{ subdomains.uptime_kuma }}.{{ root_domain }}" - - tasks: - - name: Validate Uptime Kuma configuration - assert: - that: - - uptime_kuma_api_url is defined - - uptime_kuma_api_url != "" - - uptime_kuma_username is defined - - uptime_kuma_username != "" - - uptime_kuma_password is defined - - uptime_kuma_password != "" - fail_msg: "uptime_kuma_api_url, uptime_kuma_username and uptime_kuma_password must be set" - - - name: Get hostname for monitor identification - command: hostname - register: host_name - changed_when: false - - - name: Set monitor name and group based on hostname - set_fact: - monitor_name: "cpu-temp-{{ host_name.stdout }}" - monitor_friendly_name: "CPU Temperature: {{ host_name.stdout }}" - uptime_kuma_monitor_group: "{{ host_name.stdout }} - infra" - - - name: Create Uptime Kuma CPU temperature monitor setup script - copy: - dest: /tmp/setup_uptime_kuma_cpu_temp_monitor.py - content: | - #!/usr/bin/env python3 - import sys - import json - from uptime_kuma_api import UptimeKumaApi - - def main(): - api_url = sys.argv[1] - username = sys.argv[2] - password = sys.argv[3] - group_name = sys.argv[4] - monitor_name = sys.argv[5] - monitor_description = sys.argv[6] - interval = int(sys.argv[7]) - ntfy_topic = sys.argv[8] if len(sys.argv) > 8 else "alerts" - - api = UptimeKumaApi(api_url, timeout=60, wait_events=2.0) - api.login(username, password) - - monitors = api.get_monitors() - notifications = api.get_notifications() - - ntfy_notification = next((n for n in notifications if n.get('name') == f'ntfy ({ntfy_topic})'), None) - notification_id_list = {} - if ntfy_notification: - notification_id_list[ntfy_notification['id']] = True - - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - if not group: - api.add_monitor(type='group', name=group_name) - monitors = api.get_monitors() - group = next((m for m in monitors if m.get('name') == group_name and m.get('type') == 'group'), None) - - existing_monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - monitor_data = { - 'type': 'push', - 'name': monitor_name, - 'parent': group['id'], - 'interval': interval, - 'upsideDown': True, - 'description': monitor_description, - 'notificationIDList': notification_id_list - } - - if existing_monitor: - api.edit_monitor(existing_monitor['id'], **monitor_data) - else: - api.add_monitor(**monitor_data) - - monitors = api.get_monitors() - monitor = next((m for m in monitors if m.get('name') == monitor_name), None) - - result = { - 'monitor_id': monitor['id'], - 'push_token': monitor['pushToken'], - 'group_name': group_name, - 'group_id': group['id'], - 'monitor_name': monitor_name - } - print(json.dumps(result)) - - api.disconnect() - - if __name__ == '__main__': - main() - mode: '0755' - delegate_to: localhost - become: no - - - name: Run Uptime Kuma monitor setup script - command: > - {{ ansible_playbook_python }} - /tmp/setup_uptime_kuma_cpu_temp_monitor.py - "{{ uptime_kuma_api_url }}" - "{{ uptime_kuma_username }}" - "{{ uptime_kuma_password }}" - "{{ uptime_kuma_monitor_group }}" - "{{ monitor_name }}" - "{{ monitor_friendly_name }} - Alerts when temperature exceeds {{ temp_threshold_celsius }}°C" - "{{ (temp_check_interval_minutes * 60) + 60 }}" - "{{ ntfy_topic }}" - register: monitor_setup_result - delegate_to: localhost - become: no - changed_when: false - - - name: Parse monitor setup result - set_fact: - monitor_info_parsed: "{{ monitor_setup_result.stdout | from_json }}" - - - name: Set push URL and monitor ID as facts - set_fact: - uptime_kuma_cpu_temp_push_url: "{{ uptime_kuma_api_url }}/api/push/{{ monitor_info_parsed.push_token }}" - uptime_kuma_monitor_id: "{{ monitor_info_parsed.monitor_id }}" - - - name: Install required packages for temperature monitoring - package: - name: - - lm-sensors - - curl - - jq - - bc - state: present - - - name: Create monitoring script directory - file: - path: "{{ monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create CPU temperature monitoring script - copy: - dest: "{{ monitoring_script_path }}" - content: | - #!/bin/bash - - # CPU Temperature Monitoring Script - # Monitors CPU temperature and sends alerts to Uptime Kuma - - LOG_FILE="{{ log_file }}" - TEMP_THRESHOLD="{{ temp_threshold_celsius }}" - UPTIME_KUMA_URL="{{ uptime_kuma_cpu_temp_push_url }}" - - log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" - } - - get_cpu_temp() { - local temp="" - - if command -v sensors >/dev/null 2>&1; then - temp=$(sensors 2>/dev/null | grep -E "Core 0|Package id 0|Tdie|Tctl" | head -1 | grep -oE '[0-9]+\.[0-9]+°C' | grep -oE '[0-9]+\.[0-9]+') - fi - - if [ -z "$temp" ] && [ -f /sys/class/thermal/thermal_zone0/temp ]; then - temp=$(cat /sys/class/thermal/thermal_zone0/temp) - temp=$(echo "scale=1; $temp/1000" | bc -l 2>/dev/null || echo "$temp") - fi - - if [ -z "$temp" ] && command -v acpi >/dev/null 2>&1; then - temp=$(acpi -t 2>/dev/null | grep -oE '[0-9]+\.[0-9]+' | head -1) - fi - - echo "$temp" - } - - send_uptime_kuma_alert() { - local temp="$1" - local message="CPU Temperature Alert: ${temp}°C (Threshold: ${TEMP_THRESHOLD}°C)" - - log_message "ALERT: $message" - - encoded_message=$(printf '%s\n' "$message" | sed 's/ /%20/g; s/°/%C2%B0/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g') - response=$(curl -s -w "\n%{http_code}" "$UPTIME_KUMA_URL?status=up&msg=$encoded_message" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Alert sent successfully to Uptime Kuma (HTTP $http_code)" - else - log_message "ERROR: Failed to send alert to Uptime Kuma (HTTP $http_code)" - fi - } - - main() { - log_message "Starting CPU temperature check" - - current_temp=$(get_cpu_temp) - - if [ -z "$current_temp" ]; then - log_message "ERROR: Could not read CPU temperature" - exit 1 - fi - - log_message "Current CPU temperature: ${current_temp}°C" - - if (( $(echo "$current_temp > $TEMP_THRESHOLD" | bc -l) )); then - log_message "WARNING: CPU temperature ${current_temp}°C exceeds threshold ${TEMP_THRESHOLD}°C" - send_uptime_kuma_alert "$current_temp" - else - log_message "CPU temperature is within normal range" - fi - } - - main - owner: root - group: root - mode: '0755' - - - name: Create systemd service for CPU temperature monitoring - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.service" - content: | - [Unit] - Description=CPU Temperature Monitor - After=network.target - - [Service] - Type=oneshot - ExecStart={{ monitoring_script_path }} - User=root - StandardOutput=journal - StandardError=journal - - [Install] - WantedBy=multi-user.target - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for CPU temperature monitoring - copy: - dest: "/etc/systemd/system/{{ systemd_service_name }}.timer" - content: | - [Unit] - Description=Run CPU Temperature Monitor every {{ temp_check_interval_minutes }} minute(s) - Requires={{ systemd_service_name }}.service - - [Timer] - OnBootSec={{ temp_check_interval_minutes }}min - OnUnitActiveSec={{ temp_check_interval_minutes }}min - Persistent=true - - [Install] - WantedBy=timers.target - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start CPU temperature monitoring timer - systemd: - name: "{{ systemd_service_name }}.timer" - enabled: yes - state: started - - - name: Test CPU temperature monitoring script - command: "{{ monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "CPU temperature monitoring script failed to execute properly" - - - name: Display monitoring configuration - debug: - msg: - - "CPU Temperature Monitoring configured successfully" - - "Temperature threshold: {{ temp_threshold_celsius }}°C" - - "Check interval: {{ temp_check_interval_minutes }} minute(s)" - - "Monitor Name: {{ monitor_friendly_name }}" - - "Monitor Group: {{ uptime_kuma_monitor_group }}" - - "Uptime Kuma Push URL: {{ uptime_kuma_cpu_temp_push_url }}" - - "Monitoring script: {{ monitoring_script_path }}" - - "Systemd Service: {{ systemd_service_name }}.service" - - "Systemd Timer: {{ systemd_service_name }}.timer" - - - name: Clean up temporary Uptime Kuma setup script - file: - path: /tmp/setup_uptime_kuma_cpu_temp_monitor.py - state: absent - delegate_to: localhost - become: no - diff --git a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml index 1aad2e7..2efbf8c 100644 --- a/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml +++ b/ansible/infra/nodito/32_zfs_pool_setup_playbook.yml @@ -170,141 +170,55 @@ when: "'ONLINE' not in final_zfs_status.stdout" # ───────────────────────────────────────────────────────────────────────────── -# ZFS health monitoring and monthly scrub. +# The monthly scrub. # -# The check script decides healthy/unhealthy and says so in its exit code, which -# systemd keeps: systemctl is-failed zfs-health-monitor.service +# The ZFS HEALTH CHECK that used to share this play is gone: it is now the +# zfs-health check in infra/400_host_monitoring.yml, which carries the same five +# conditions - pool state, device states, resilver in progress, read/write/ +# checksum errors, and errors from the last scan - but reports to Gatus like +# every other host check instead of owning its own push plumbing. # -# Reporting anywhere else is optional. Set `healthcheck_push_url` and the script -# will also GET it with ?status=up|down; leave it empty and the exit code is -# still the whole answer. Today that URL points at Uptime Kuma, which is still -# running on watchtower but is no longer deployed by Ansible - the credentials -# were retired, the service was not. If it is ever replaced, `healthcheck_push_url` -# is the only thing that needs to change here. +# The scrub itself stays here, because it is not monitoring: it is the +# maintenance that gives the health check something true to report. A pool that +# is never scrubbed has no idea whether it is healthy. # ───────────────────────────────────────────────────────────────────────────── -- name: Setup ZFS Pool Health Monitoring and Monthly Scrubs +- name: Schedule the monthly ZFS scrub hosts: hypervisor become: true + vars_files: + - ../../infra_vars.yml vars: - zfs_check_interval_seconds: 86400 # 24 hours - zfs_check_timeout_seconds: 90000 # ~25 hours (interval + buffer) - zfs_check_retries: 1 - zfs_monitoring_script_dir: /opt/zfs-monitoring - zfs_monitoring_script_path: "{{ zfs_monitoring_script_dir }}/zfs_health_monitor.sh" - zfs_log_file: "{{ zfs_monitoring_script_dir }}/zfs_health_monitor.log" - zfs_systemd_health_service_name: zfs-health-monitor zfs_systemd_scrub_service_name: zfs-monthly-scrub - # Optional. Empty is fine and is not an error - see the banner above. - healthcheck_push_url: "{{ healthcheck_push_urls.zfs_health | default('') }}" tasks: - - name: Install required packages for ZFS monitoring - package: - name: - - curl - - jq - state: present - - - name: Create monitoring script directory - file: - path: "{{ zfs_monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create ZFS health monitoring script - template: - dest: "{{ zfs_monitoring_script_path }}" - src: templates/zfs_health_monitor.sh.j2 - owner: root - group: root - mode: '0755' - - - name: Create systemd service for ZFS health monitoring - template: - dest: "/etc/systemd/system/{{ zfs_systemd_health_service_name }}.service" - src: templates/zfs-health-monitor.service.j2 - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for daily ZFS health monitoring - template: - dest: "/etc/systemd/system/{{ zfs_systemd_health_service_name }}.timer" - src: templates/zfs-health-monitor.timer.j2 - owner: root - group: root - mode: '0644' - - name: Create systemd service for ZFS monthly scrub template: - dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.service" src: templates/zfs-monthly-scrub.service.j2 + dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.service" owner: root group: root mode: '0644' - name: Create systemd timer for monthly ZFS scrub template: - dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.timer" src: templates/zfs-monthly-scrub.timer.j2 + dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.timer" owner: root group: root mode: '0644' - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start ZFS health monitoring timer - systemd: - name: "{{ zfs_systemd_health_service_name }}.timer" - enabled: yes - state: started - - - name: Enable and start ZFS monthly scrub timer + - name: Enable and start the monthly scrub timer systemd: name: "{{ zfs_systemd_scrub_service_name }}.timer" enabled: yes state: started + daemon_reload: yes - - name: Test ZFS health monitoring script - command: "{{ zfs_monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "ZFS health monitoring script failed - check pool health" - - - name: Display monitoring configuration + - name: Report the scrub schedule debug: - msg: | - ✓ ZFS Pool Health Monitoring deployed successfully! - - Pool Name: {{ zfs_pool_name }} - Push URL: {{ healthcheck_push_url | default('', true) | ternary('set', 'not set - exit code only') }} - - Health Check: - - Frequency: Every {{ zfs_check_interval_seconds }} seconds (24 hours) - - Timeout: {{ zfs_check_timeout_seconds }} seconds (~25 hours) - - Script: {{ zfs_monitoring_script_path }} - - Log: {{ zfs_log_file }} - - Service: {{ zfs_systemd_health_service_name }}.service - - Timer: {{ zfs_systemd_health_service_name }}.timer - - Monthly Scrub: - - Schedule: Last day of month at 4:00 AM - - Service: {{ zfs_systemd_scrub_service_name }}.service - - Timer: {{ zfs_systemd_scrub_service_name }}.timer - - Conditions monitored: - - Pool state (must be ONLINE) - - Device states (no DEGRADED/FAULTED/OFFLINE/UNAVAIL) - - Resilver status (alerts if resilvering) - - Read/Write/Checksum errors - - Scrub errors + msg: >- + Monthly scrub of {{ zfs_pool_name }}: + last day of each month at 04:00. + Health is reported separately by the zfs-health check + (infra/400_host_monitoring.yml). diff --git a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml index 8e188b6..12fe720 100644 --- a/ansible/infra/nodito/34_nut_ups_setup_playbook.yml +++ b/ansible/infra/nodito/34_nut_ups_setup_playbook.yml @@ -225,121 +225,11 @@ - nut-server - nut-monitor - -# ───────────────────────────────────────────────────────────────────────────── -# UPS heartbeat monitoring. +# The UPS heartbeat play that used to live here is gone. What it deployed - +# /opt/ups-monitoring plus a ups-heartbeat timer - is now the ups-status check +# in infra/400_host_monitoring.yml, which reports to Gatus like every other +# host check instead of carrying its own push plumbing. # -# The script decides on-mains/on-battery and says so in its exit code, which -# systemd keeps: systemctl is-failed ups-heartbeat.service -# -# Reporting anywhere else is optional. Set `healthcheck_push_url` and the script -# will also GET it with ?status=up|down; leave it empty and the exit code is -# still the whole answer. Today that URL points at Uptime Kuma, which is still -# running on watchtower but is no longer deployed by Ansible - the credentials -# were retired, the service was not. If it is ever replaced, `healthcheck_push_url` -# is the only thing that needs to change here. -# -# NOTE: this play has never actually been applied to nodito. /opt/ups-monitoring -# does not exist and there is no ups-heartbeat timer. What is on the box is a -# hand-written /usr/local/bin/ups-heartbeat.sh - mode 0644, not executable, and -# referenced by no unit and no cron entry, so nothing has ever run it. Its push -# token (uLmCPkLLO4) belongs to a monitor that DOES still exist and answer, so -# that monitor has had no heartbeat since January 2026. It is now -# healthcheck_push_urls.ups in the vault, and running this play is what will -# finally start feeding it. -# ───────────────────────────────────────────────────────────────────────────── -- name: Setup UPS Heartbeat Monitoring - hosts: hypervisor - become: true - - vars: - ups_heartbeat_interval_seconds: 60 - ups_heartbeat_timeout_seconds: 120 - ups_heartbeat_retries: 1 - ups_monitoring_script_dir: /opt/ups-monitoring - ups_monitoring_script_path: "{{ ups_monitoring_script_dir }}/ups_heartbeat.sh" - ups_log_file: "{{ ups_monitoring_script_dir }}/ups_heartbeat.log" - ups_systemd_service_name: ups-heartbeat - # Optional. Empty is fine and is not an error - see the banner above. - healthcheck_push_url: "{{ healthcheck_push_urls.ups | default('') }}" - - tasks: - - name: Install required packages for UPS monitoring - package: - name: - - curl - state: present - - - name: Create monitoring script directory - file: - path: "{{ ups_monitoring_script_dir }}" - state: directory - owner: root - group: root - mode: '0755' - - - name: Create UPS heartbeat monitoring script - template: - dest: "{{ ups_monitoring_script_path }}" - src: templates/ups_heartbeat.sh.j2 - owner: root - group: root - mode: '0755' - - - name: Create systemd service for UPS heartbeat - template: - dest: "/etc/systemd/system/{{ ups_systemd_service_name }}.service" - src: templates/ups-heartbeat.service.j2 - owner: root - group: root - mode: '0644' - - - name: Create systemd timer for UPS heartbeat - template: - dest: "/etc/systemd/system/{{ ups_systemd_service_name }}.timer" - src: templates/ups-heartbeat.timer.j2 - owner: root - group: root - mode: '0644' - - - name: Reload systemd daemon - systemd: - daemon_reload: yes - - - name: Enable and start UPS heartbeat timer - systemd: - name: "{{ ups_systemd_service_name }}.timer" - enabled: yes - state: started - - - name: Test UPS heartbeat script - command: "{{ ups_monitoring_script_path }}" - register: script_test - changed_when: false - - - name: Verify script execution - assert: - that: - - script_test.rc == 0 - fail_msg: "UPS heartbeat script failed - check UPS status and communication" - - - name: Display monitoring configuration - debug: - msg: - - "UPS Monitoring configured successfully" - - "" - - "NUT Configuration:" - - " UPS Name: {{ ups_name }}" - - " UPS Description: {{ ups_desc }}" - - " Off Delay: {{ ups_offdelay }}s (time after shutdown before UPS cuts power)" - - " On Delay: {{ ups_ondelay }}s (time after mains returns before UPS restores power)" - - "" - - "Health reporting:" - - " Check interval: {{ ups_heartbeat_interval_seconds }}s" - - " Push URL: {{ healthcheck_push_url | default('', true) | ternary('set', 'not set - exit code only') }}" - - "" - - "Scripts and Services:" - - " Script: {{ ups_monitoring_script_path }}" - - " Log: {{ ups_log_file }}" - - " Service: {{ ups_systemd_service_name }}.service" - - " Timer: {{ ups_systemd_service_name }}.timer" +# This playbook is now purely NUT setup: the driver, upsd, upsmon and the +# shutdown behaviour. Monitoring whether the UPS is on mains is a separate +# concern and belongs with the other host checks. diff --git a/ansible/infra/nodito/templates/ups-heartbeat.service.j2 b/ansible/infra/nodito/templates/ups-heartbeat.service.j2 deleted file mode 100644 index 71484c1..0000000 --- a/ansible/infra/nodito/templates/ups-heartbeat.service.j2 +++ /dev/null @@ -1,14 +0,0 @@ -[Unit] -Description=UPS Heartbeat Monitor -After=network.target nut-monitor.service - -[Service] -Type=oneshot -ExecStart={{ ups_monitoring_script_path }} -User=root -Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} -StandardOutput=journal -StandardError=journal - -[Install] -WantedBy=multi-user.target diff --git a/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 b/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 deleted file mode 100644 index 925998e..0000000 --- a/ansible/infra/nodito/templates/ups-heartbeat.timer.j2 +++ /dev/null @@ -1,11 +0,0 @@ -[Unit] -Description=Run UPS Heartbeat Monitor every {{ ups_heartbeat_interval_seconds }} seconds -Requires={{ ups_systemd_service_name }}.service - -[Timer] -OnBootSec=1min -OnUnitActiveSec={{ ups_heartbeat_interval_seconds }}sec -Persistent=true - -[Install] -WantedBy=timers.target diff --git a/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 b/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 deleted file mode 100644 index c8d5d0d..0000000 --- a/ansible/infra/nodito/templates/ups_heartbeat.sh.j2 +++ /dev/null @@ -1,68 +0,0 @@ -#!/bin/bash - -# UPS heartbeat check - managed by Ansible (infra/nodito/34_nut_ups_setup_playbook.yml) -# -# The exit code is the answer and systemd keeps it: -# systemctl is-failed {{ ups_systemd_service_name }}.service -# Reporting anywhere else is optional and generic. - -LOG_FILE="{{ ups_log_file }}" -UPS_NAME="{{ ups_name }}" -PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" - -log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" -} - -report() { - local status="$1" - local message="$2" - - # No push URL is normal, not an error: the exit code below is still a - # complete answer for anything reading unit state. - [ -n "$PUSH_URL" ] || return 0 - - local encoded_message - encoded_message=$(printf '%s\n' "$message" | sed 's/%/%25/g; s/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') - - local response http_code - response=$(curl -s --max-time 10 --retry 2 -w "\n%{http_code}" "${PUSH_URL}?status=${status}&msg=${encoded_message}&ping=" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Reported ${status}: $message (HTTP $http_code)" - return 0 - else - log_message "ERROR: Failed to report ${status} (HTTP $http_code)" - return 1 - fi -} - -main() { - local status charge runtime load - - status=$(upsc ${UPS_NAME}@localhost ups.status 2>/dev/null) - - if [ -z "$status" ]; then - log_message "ERROR: Cannot communicate with UPS" - report "down" "cannot communicate with UPS ${UPS_NAME}" - exit 1 - fi - - charge=$(upsc ${UPS_NAME}@localhost battery.charge 2>/dev/null) - runtime=$(upsc ${UPS_NAME}@localhost battery.runtime 2>/dev/null) - load=$(upsc ${UPS_NAME}@localhost ups.load 2>/dev/null) - - if [[ "$status" == *"OL"* ]]; then - local message="UPS on mains (charge=${charge}% runtime=${runtime}s load=${load}%)" - log_message "$message" - report "up" "$message" - exit 0 - else - log_message "UPS not on mains power (status=$status)" - report "down" "UPS not on mains (status=${status} charge=${charge}%)" - exit 1 - fi -} - -main diff --git a/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 b/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 deleted file mode 100644 index 104d974..0000000 --- a/ansible/infra/nodito/templates/zfs-health-monitor.service.j2 +++ /dev/null @@ -1,14 +0,0 @@ -[Unit] -Description=ZFS Pool Health Monitor -After=zfs.target network.target - -[Service] -Type=oneshot -ExecStart={{ zfs_monitoring_script_path }} -User=root -Environment=HEALTHCHECK_PUSH_URL={{ healthcheck_push_url }} -StandardOutput=journal -StandardError=journal - -[Install] -WantedBy=multi-user.target diff --git a/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 b/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 deleted file mode 100644 index 7878ee3..0000000 --- a/ansible/infra/nodito/templates/zfs-health-monitor.timer.j2 +++ /dev/null @@ -1,11 +0,0 @@ -[Unit] -Description=Run ZFS Pool Health Monitor daily -Requires={{ zfs_systemd_health_service_name }}.service - -[Timer] -OnBootSec=5min -OnUnitActiveSec={{ zfs_check_interval_seconds }}sec -Persistent=true - -[Install] -WantedBy=timers.target diff --git a/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 b/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 deleted file mode 100644 index 3ff4c41..0000000 --- a/ansible/infra/nodito/templates/zfs_health_monitor.sh.j2 +++ /dev/null @@ -1,181 +0,0 @@ -#!/bin/bash - -# ZFS pool health check - managed by Ansible (infra/nodito/32_zfs_pool_setup_playbook.yml) -# -# The exit code is the answer and systemd keeps it: -# systemctl is-failed {{ zfs_systemd_health_service_name }}.service -# Reporting anywhere else is optional and generic. - -LOG_FILE="{{ zfs_log_file }}" -POOL_NAME="{{ zfs_pool_name }}" -PUSH_URL="${HEALTHCHECK_PUSH_URL:-}" -HOSTNAME=$(hostname) - -# Function to log messages -log_message() { - echo "$(date '+%Y-%m-%d %H:%M:%S') - $1" >> "$LOG_FILE" -} - -# Function to check pool health using JSON output -check_pool_health() { - local pool="$1" - local issues_found=0 - - # Get pool status as JSON - local pool_json - pool_json=$(zpool status -j "$pool" 2>&1) - - if [ $? -ne 0 ]; then - log_message "ERROR: Failed to get pool status for $pool" - log_message " -> $pool_json" - return 1 - fi - - # Check 1: Pool state must be ONLINE - local pool_state - pool_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].state') - - if [ "$pool_state" != "ONLINE" ]; then - log_message "ISSUE: Pool state is $pool_state (expected ONLINE)" - issues_found=1 - else - log_message "OK: Pool state is ONLINE" - fi - - # Check 2: Check all vdevs and devices for non-ONLINE states - local bad_states - bad_states=$(echo "$pool_json" | jq -r --arg pool "$pool" ' - .pools[$pool].vdevs[] | - .. | objects | - select(.state? and .state != "ONLINE") | - "\(.name // "unknown"): \(.state)" - ' 2>/dev/null) - - if [ -n "$bad_states" ]; then - log_message "ISSUE: Found devices not in ONLINE state:" - echo "$bad_states" | while read -r line; do - log_message " -> $line" - done - issues_found=1 - else - log_message "OK: All devices are ONLINE" - fi - - # Check 3: Check for resilvering in progress - local scan_function scan_state - scan_function=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.function // "NONE"') - scan_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.state // "NONE"') - - if [ "$scan_function" = "RESILVER" ] && [ "$scan_state" = "SCANNING" ]; then - local resilver_progress - resilver_progress=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.issued // "unknown"') - log_message "ISSUE: Pool is currently resilvering (disk reconstruction in progress) - ${resilver_progress} processed" - issues_found=1 - fi - - # Check 4: Check for read/write/checksum errors on all devices - # Note: ZFS JSON output has error counts as strings, so convert to numbers for comparison - local devices_with_errors - devices_with_errors=$(echo "$pool_json" | jq -r --arg pool "$pool" ' - .pools[$pool].vdevs[] | - .. | objects | - select(.name? and ((.read_errors // "0" | tonumber) > 0 or (.write_errors // "0" | tonumber) > 0 or (.checksum_errors // "0" | tonumber) > 0)) | - "\(.name): read=\(.read_errors // 0) write=\(.write_errors // 0) cksum=\(.checksum_errors // 0)" - ' 2>/dev/null) - - if [ -n "$devices_with_errors" ]; then - log_message "ISSUE: Found devices with I/O errors:" - echo "$devices_with_errors" | while read -r line; do - log_message " -> $line" - done - issues_found=1 - else - log_message "OK: No read/write/checksum errors detected" - fi - - # Check 5: Check for scan errors (from last scrub/resilver) - local scan_errors - scan_errors=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.errors // "0"') - - if [ "$scan_errors" != "0" ] && [ "$scan_errors" != "null" ] && [ -n "$scan_errors" ]; then - log_message "ISSUE: Last scan reported $scan_errors errors" - issues_found=1 - else - log_message "OK: No scan errors" - fi - - return $issues_found -} - -# Function to get last scrub info for status message -get_scrub_info() { - local pool="$1" - local pool_json - pool_json=$(zpool status -j "$pool" 2>/dev/null) - - local scan_func scan_state scan_start - scan_func=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.function // "NONE"') - scan_state=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.state // "NONE"') - scan_start=$(echo "$pool_json" | jq -r --arg pool "$pool" '.pools[$pool].scan_stats.start_time // ""') - - if [ "$scan_func" = "SCRUB" ] && [ "$scan_state" = "SCANNING" ]; then - echo "scrub in progress (started $scan_start)" - elif [ "$scan_func" = "SCRUB" ] && [ -n "$scan_start" ]; then - echo "last scrub: $scan_start" - else - echo "no scrub history" - fi -} - -# Optional reporting to whatever is watching. No push URL is normal, not an -# error: the script's exit code is still a complete answer for anything reading -# unit state. -report() { - local status="$1" - local message="$2" - - [ -n "$PUSH_URL" ] || return 0 - - log_message "Reporting ${status}: $message" - - # URL encode the message - local encoded_message - encoded_message=$(printf '%s\n' "$message" | sed 's/%/%25/g; s/ /%20/g; s/(/%28/g; s/)/%29/g; s/:/%3A/g; s/\//%2F/g') - - local response http_code - response=$(curl -s --max-time 10 --retry 2 -w "\n%{http_code}" "${PUSH_URL}?status=${status}&msg=${encoded_message}&ping=" 2>&1) - http_code=$(echo "$response" | tail -n1) - - if [ "$http_code" = "200" ] || [ "$http_code" = "201" ]; then - log_message "Report sent successfully (HTTP $http_code)" - return 0 - else - log_message "ERROR: Failed to report (HTTP $http_code)" - return 1 - fi -} - -# Main health check logic -main() { - log_message "==========================================" - log_message "Starting ZFS health check for pool: $POOL_NAME on $HOSTNAME" - - # Run all health checks - if check_pool_health "$POOL_NAME"; then - local scrub_info - scrub_info=$(get_scrub_info "$POOL_NAME") - - local message="Pool $POOL_NAME healthy ($scrub_info)" - report "up" "$message" - - log_message "Health check completed: ALL OK" - exit 0 - else - log_message "Health check completed: ISSUES DETECTED" - report "down" "Pool $POOL_NAME unhealthy - see $LOG_FILE" - exit 1 - fi -} - -# Run main function -main diff --git a/ansible/roles/healthcheck/defaults/main.yml b/ansible/roles/healthcheck/defaults/main.yml index 526347a..51534c2 100644 --- a/ansible/roles/healthcheck/defaults/main.yml +++ b/ansible/roles/healthcheck/defaults/main.yml @@ -46,4 +46,9 @@ healthcheck_log_dir: /var/log/healthchecks healthcheck_disk_threshold: 85 # percent healthcheck_cpu_temp_threshold: 80 # celsius healthcheck_zfs_pool: "" +# ZFS degrades badly once a pool passes roughly 80% - allocation gets slow and +# fragmentation becomes hard to undo, and unlike a normal filesystem you cannot +# simply delete your way back to good performance. So this alarms well before +# the pool is actually out of space. +healthcheck_zfs_capacity_threshold: 80 healthcheck_ups_name: "" diff --git a/ansible/roles/healthcheck/tasks/main.yml b/ansible/roles/healthcheck/tasks/main.yml index fb0257b..f37a0d5 100644 --- a/ansible/roles/healthcheck/tasks/main.yml +++ b/ansible/roles/healthcheck/tasks/main.yml @@ -11,10 +11,26 @@ or healthcheck_command, and a token whenever a push URL is set. A push URL with no token would report to Gatus and be rejected 401 on every run. +# Deduplicated across the whole play run. This role is included once PER CHECK, +# and a host with several checks was otherwise running apt several times to +# install a curl that was already there - 29 apt transactions estate-wide, and +# the slowest thing in the deploy by a wide margin. The fact below remembers +# what has already been ensured on this host. - name: Install healthcheck dependencies ansible.builtin.package: - name: "{{ healthcheck_packages | default(['curl']) }}" + name: "{{ healthcheck_wanted_packages }}" state: present + vars: + healthcheck_wanted_packages: >- + {{ (healthcheck_packages | default(['curl'])) + | difference(healthcheck_installed_packages | default([])) }} + when: healthcheck_wanted_packages | length > 0 + +- name: Remember which dependencies this host already has + ansible.builtin.set_fact: + healthcheck_installed_packages: >- + {{ (healthcheck_installed_packages | default([])) + | union(healthcheck_packages | default(['curl'])) }} - name: Create the healthcheck log directory ansible.builtin.file: @@ -49,10 +65,6 @@ group: root mode: "0644" -- name: Reload systemd - ansible.builtin.systemd: - daemon_reload: yes - # `restarted`, not `started`: started is a no-op on an already-active timer, so # a changed interval or a stuck timer would never be picked up. - name: "Enable and start the {{ healthcheck_name }} timer" diff --git a/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 b/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 index b6c7dae..ebe88b9 100644 --- a/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 +++ b/ansible/roles/healthcheck/templates/checks/zfs-health.sh.j2 @@ -44,9 +44,20 @@ issues="${issues}${issues:+; }scan errors ${scan_err}" fi + # 6. Capacity. Not an error condition in `zpool status` - a 95% full pool is + # reported perfectly ONLINE - so it has to be read separately, and it is + # the failure you get warning of rather than the one you discover. + local capacity + capacity=$(zpool list -H -o capacity "$pool" 2>/dev/null | tr -dc '0-9') + if [ -z "$capacity" ]; then + issues="${issues}${issues:+; }cannot read capacity" + elif [ "$capacity" -ge {{ healthcheck_zfs_capacity_threshold }} ]; then + issues="${issues}${issues:+; }pool ${capacity}% full (>={{ healthcheck_zfs_capacity_threshold }}%)" + fi + if [ -n "$issues" ]; then MESSAGE="$issues"; return 1; fi local scrub scrub=$(echo "$json" | jq -r --arg p "$pool" '.pools[$p].scan_stats.start_time // "never"') - MESSAGE="${pool} ONLINE, last scrub ${scrub}" + MESSAGE="${pool} ONLINE, ${capacity}% full, last scrub ${scrub}" return 0 diff --git a/ansible/site.yml b/ansible/site.yml index fd042ce..a5f8576 100644 --- a/ansible/site.yml +++ b/ansible/site.yml @@ -18,6 +18,17 @@ - import_playbook: infra/02_firewall_and_fail2ban_playbook.yml - import_playbook: infra/900_install_rsync.yml - import_playbook: infra/920_join_headscale_mesh.yml +# Idempotent and kept permanently: guarantees a rebuilt or restored host cannot +# quietly bring the Uptime-Kuma-era monitoring back. +- import_playbook: infra/409_remove_legacy_monitoring.yml + +# ── Monitoring ────────────────────────────────────────────────────────────── +# Gatus first: the three plays below register endpoints with it, and registering +# against a host that is not serving yet would simply fail. +- import_playbook: services/gatus/deploy_gatus_playbook.yml +- import_playbook: infra/400_host_monitoring.yml +- import_playbook: infra/401_service_monitoring.yml +- import_playbook: infra/402_public_monitoring.yml # 910_docker says `hosts: managed`, but only 5 of 11 managed hosts have or need # Docker. Left out until it has a [docker] group — see the note in PLAN_7. @@ -56,13 +67,6 @@ # Deliberately not here. Every playbook in the repo is either imported above or # listed below, so this file accounts for all of them: # -# infra/410_disk_usage_alerts.yml assert on the Uptime Kuma credentials that -# infra/420_system_healthcheck.yml were removed from the vault, so they fail -# infra/430_cpu_temp_alerts.yml before doing anything. What they deploy IS -# running on the boxes - same state the two -# nodito playbooks were in before 0f03c50. -# Add them once they are de-Kuma'd. -# # infra/910_docker_playbook.yml says `hosts: managed`, but Docker is on 5 # of 11 managed hosts and those 5 are exactly # the ones that need it. Running it would @@ -74,7 +78,7 @@ # # services/ntfy/setup_ntfy_uptime_ creates a notification channel INSIDE # kuma_notification.yml Uptime Kuma. Kuma-specific tooling, not -# a deployment. +# a deployment, and Kuma is being retired. # # services/vaultwarden/disable_ deliberate manual actions, not convergence # vaultwarden_sign_ups_playbook.yml