- name: Setup ZFS RAID 1 Pool for Proxmox Storage hosts: hypervisor become: true tasks: - name: Verify Proxmox VE is running command: pveversion register: pve_version_check changed_when: false failed_when: pve_version_check.rc != 0 - name: Update package cache apt: update_cache: yes cache_valid_time: 3600 - name: Install ZFS utilities package: name: - zfsutils-linux - zfs-initramfs state: present - name: Load ZFS kernel module modprobe: name: zfs - name: Ensure ZFS module loads at boot lineinfile: path: /etc/modules line: zfs state: present - name: Check if ZFS pool already exists command: zpool list {{ zfs_pool_name }} register: zfs_pool_exists failed_when: false changed_when: false - name: Check if disks are in use shell: | for disk in {{ zfs_disk_1 }} {{ zfs_disk_2 }}; do if mount | grep -q "^$disk"; then echo "ERROR: $disk is mounted" exit 1 fi if lsblk -n -o MOUNTPOINT "$disk" | grep -v "^$" | grep -q .; then echo "ERROR: $disk has mounted partitions" exit 1 fi done register: disk_usage_check failed_when: disk_usage_check.rc != 0 changed_when: false - name: Create ZFS RAID 1 pool with optimized settings command: > zpool create {{ zfs_pool_name }} -o ashift=12 -O mountpoint=none mirror {{ zfs_disk_1 }} {{ zfs_disk_2 }} when: zfs_pool_exists.rc != 0 register: zfs_pool_create_result - name: Check if ZFS dataset already exists command: zfs list {{ zfs_pool_name }}/vm-storage register: zfs_dataset_exists failed_when: false changed_when: false - name: Create ZFS dataset for Proxmox storage command: zfs create {{ zfs_pool_name }}/vm-storage when: zfs_dataset_exists.rc != 0 register: zfs_dataset_create_result - name: Set ZFS dataset properties for Proxmox command: zfs set {{ item.property }}={{ item.value }} {{ zfs_pool_name }}/vm-storage loop: - { property: "mountpoint", value: "{{ zfs_pool_mountpoint }}" } - { property: "compression", value: "lz4" } - { property: "atime", value: "off" } - { property: "xattr", value: "sa" } - { property: "acltype", value: "posixacl" } - { property: "dnodesize", value: "auto" } when: zfs_dataset_exists.rc != 0 - name: Set ZFS pool properties for Proxmox command: zpool set autotrim=off {{ zfs_pool_name }} when: zfs_pool_exists.rc != 0 - name: Set ZFS pool mountpoint for Proxmox command: zfs set mountpoint={{ zfs_pool_mountpoint }} {{ zfs_pool_name }} when: zfs_pool_exists.rc == 0 - name: Export and re-import ZFS pool for Proxmox compatibility shell: | zpool export {{ zfs_pool_name }} zpool import {{ zfs_pool_name }} when: zfs_pool_exists.rc != 0 register: zfs_pool_import_result - name: Ensure ZFS services are enabled systemd: name: "{{ item }}" enabled: yes state: started loop: - zfs-import-cache - zfs-import-scan - zfs-mount - zfs-share - zfs-zed - name: Check if ZFS pool storage already exists in Proxmox config stat: path: /etc/pve/storage.cfg register: storage_cfg_file - name: Check if storage name exists in Proxmox config shell: "grep -q '^zfspool: {{ zfs_pool_name }}' /etc/pve/storage.cfg" register: storage_exists_check failed_when: false changed_when: false when: storage_cfg_file.stat.exists - name: Set storage not configured when config file doesn't exist set_fact: storage_exists_check: rc: 1 when: not storage_cfg_file.stat.exists - name: Debug storage configuration status debug: msg: | Config file exists: {{ storage_cfg_file.stat.exists }} Storage check result: {{ storage_exists_check.rc }} Pool exists: {{ zfs_pool_exists.rc == 0 }} Will add storage: {{ zfs_pool_exists.rc == 0 and storage_exists_check.rc != 0 }} # Registration is add-only on purpose. There used to be a "Remove existing # storage if it exists" task here that ran `pvesm remove` whenever the # storage WAS present, paired with an add that only ran when it was ABSENT. # The two conditions are mutually exclusive, so a real run against a # correctly-configured hypervisor removed the storage entry backing every VM # and never put it back. It would also have dropped `mountpoint /var/lib/vz`, # which the live entry has and which `pvesm add` below does not set. # # If the storage entry ever needs its options changed, edit # /etc/pve/storage.cfg or use `pvesm set` - do not re-register it from here. - name: Add ZFS pool storage to Proxmox using pvesm command: > pvesm add zfspool {{ zfs_pool_name }} --pool {{ zfs_pool_name }} --content rootdir,images --sparse 1 when: - zfs_pool_exists.rc == 0 - storage_exists_check.rc != 0 register: pvesm_add_result - name: Verify ZFS pool is healthy command: zpool status {{ zfs_pool_name }} register: final_zfs_status changed_when: false - name: Fail if ZFS pool is not healthy fail: msg: "ZFS pool {{ zfs_pool_name }} is not in a healthy state" when: "'ONLINE' not in final_zfs_status.stdout" # ───────────────────────────────────────────────────────────────────────────── # ZFS health monitoring and monthly scrub. # # The check script decides healthy/unhealthy and says so in its exit code, which # systemd keeps: systemctl is-failed zfs-health-monitor.service # # Reporting anywhere else is optional. Set `healthcheck_push_url` and the script # will also GET it with ?status=up|down; leave it empty and the exit code is # still the whole answer. Today that URL points at Uptime Kuma, which is still # running on watchtower but is no longer deployed by Ansible - the credentials # were retired, the service was not. If it is ever replaced, `healthcheck_push_url` # is the only thing that needs to change here. # ───────────────────────────────────────────────────────────────────────────── - name: Setup ZFS Pool Health Monitoring and Monthly Scrubs hosts: hypervisor become: true vars: zfs_check_interval_seconds: 86400 # 24 hours zfs_check_timeout_seconds: 90000 # ~25 hours (interval + buffer) zfs_check_retries: 1 zfs_monitoring_script_dir: /opt/zfs-monitoring zfs_monitoring_script_path: "{{ zfs_monitoring_script_dir }}/zfs_health_monitor.sh" zfs_log_file: "{{ zfs_monitoring_script_dir }}/zfs_health_monitor.log" zfs_systemd_health_service_name: zfs-health-monitor zfs_systemd_scrub_service_name: zfs-monthly-scrub # Optional. Empty is fine and is not an error - see the banner above. healthcheck_push_url: "{{ healthcheck_push_urls.zfs_health | default('') }}" tasks: - name: Install required packages for ZFS monitoring package: name: - curl - jq state: present - name: Create monitoring script directory file: path: "{{ zfs_monitoring_script_dir }}" state: directory owner: root group: root mode: '0755' - name: Create ZFS health monitoring script template: dest: "{{ zfs_monitoring_script_path }}" src: templates/zfs_health_monitor.sh.j2 owner: root group: root mode: '0755' - name: Create systemd service for ZFS health monitoring template: dest: "/etc/systemd/system/{{ zfs_systemd_health_service_name }}.service" src: templates/zfs-health-monitor.service.j2 owner: root group: root mode: '0644' - name: Create systemd timer for daily ZFS health monitoring template: dest: "/etc/systemd/system/{{ zfs_systemd_health_service_name }}.timer" src: templates/zfs-health-monitor.timer.j2 owner: root group: root mode: '0644' - name: Create systemd service for ZFS monthly scrub template: dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.service" src: templates/zfs-monthly-scrub.service.j2 owner: root group: root mode: '0644' - name: Create systemd timer for monthly ZFS scrub template: dest: "/etc/systemd/system/{{ zfs_systemd_scrub_service_name }}.timer" src: templates/zfs-monthly-scrub.timer.j2 owner: root group: root mode: '0644' - name: Reload systemd daemon systemd: daemon_reload: yes - name: Enable and start ZFS health monitoring timer systemd: name: "{{ zfs_systemd_health_service_name }}.timer" enabled: yes state: started - name: Enable and start ZFS monthly scrub timer systemd: name: "{{ zfs_systemd_scrub_service_name }}.timer" enabled: yes state: started - name: Test ZFS health monitoring script command: "{{ zfs_monitoring_script_path }}" register: script_test changed_when: false - name: Verify script execution assert: that: - script_test.rc == 0 fail_msg: "ZFS health monitoring script failed - check pool health" - name: Display monitoring configuration debug: msg: | ✓ ZFS Pool Health Monitoring deployed successfully! Pool Name: {{ zfs_pool_name }} Push URL: {{ healthcheck_push_url | default('', true) | ternary('set', 'not set - exit code only') }} Health Check: - Frequency: Every {{ zfs_check_interval_seconds }} seconds (24 hours) - Timeout: {{ zfs_check_timeout_seconds }} seconds (~25 hours) - Script: {{ zfs_monitoring_script_path }} - Log: {{ zfs_log_file }} - Service: {{ zfs_systemd_health_service_name }}.service - Timer: {{ zfs_systemd_health_service_name }}.timer Monthly Scrub: - Schedule: Last day of month at 4:00 AM - Service: {{ zfs_systemd_scrub_service_name }}.service - Timer: {{ zfs_systemd_scrub_service_name }}.timer Conditions monitored: - Pool state (must be ONLINE) - Device states (no DEGRADED/FAULTED/OFFLINE/UNAVAIL) - Resilver status (alerts if resilvering) - Read/Write/Checksum errors - Scrub errors