#!/usr/bin/env bash # {{ backup_source_description }} backup — managed by Ansible (roles/backup_source) # # Dumps to stdout, encrypts with age, writes {{ backup_source_dir }}. # The host holds only the age PUBLIC key, so it cannot read its own backups. set -euo pipefail umask 077 BACKUP_DIR="{{ backup_source_dir }}" RETENTION_DAYS={{ backup_source_retention_days }} RECIPIENT="{{ backup_source_recipient }}" SUFFIX="{{ backup_source_artifact_suffix }}" NAME="{{ backup_source_name }}" {% if backup_source_stop_service or backup_source_stop_command %} STOP_CMD={{ (backup_source_stop_command or ('systemctl stop ' ~ backup_source_stop_service)) | quote }} START_CMD={{ (backup_source_start_command or ('systemctl start ' ~ backup_source_stop_service)) | quote }} SERVICE="{{ backup_source_stop_service or backup_source_description }}" # label for the log only {% endif %} TIMESTAMP=$(date +%Y%m%d_%H%M%S) ARTIFACT="${BACKUP_DIR}/${NAME}_${TIMESTAMP}.${SUFFIX}" die() { echo "FATAL: $*" >&2; exit 1; } log() { echo "$(date '+%Y-%m-%d %H:%M:%S') $*"; } # --- Pre-flight --- [[ -n "$RECIPIENT" ]] || die "no age recipient configured" command -v age >/dev/null || die "age is not installed" # Mode must agree with what the role sets, or each undoes the other every run. mkdir -p "$BACKUP_DIR" {% if backup_source_pull_user %} chown root:{{ backup_source_pull_user }} "$BACKUP_DIR" chmod 750 "$BACKUP_DIR" {% else %} chmod 700 "$BACKUP_DIR" {% endif %} # A run that died mid-dump leaves a .partial. It is not a backup, and the prune # glob below cannot match it (it ends .partial, not .${SUFFIX}), so clear them # here or they accumulate forever. rm -f "${BACKUP_DIR}/${NAME}_"*.partial # --- Reporting ------------------------------------------------------------- # A dump that exits non-zero, or that produces a zero-byte artefact, is a failed # backup even though the script "finished". Both are reported as failures. PUSH_URL="${BACKUP_PUSH_URL:-}" PUSH_TOKEN="${BACKUP_PUSH_TOKEN:-}" report() { local success="$1" message="$2" [ -n "$PUSH_URL" ] || return 0 local encoded encoded=$(printf '%s' "$message" | sed 's/%/%25/g; s/ /%20/g; s/&/%26/g; s/+/%2B/g; s/#/%23/g') curl -s -o /dev/null --max-time 15 --retry 2 --retry-delay 3 -X POST \ -H "Authorization: Bearer ${PUSH_TOKEN}" \ "${PUSH_URL}?success=${success}&error=${encoded}" 2>/dev/null || true } # Reports on ANY exit path, so a dump that dies halfway still reports rather # than going quiet. The size of the FINISHED artefact decides success, not # merely reaching the end of the script. # # This is called FROM the single EXIT trap below - it must never register an # EXIT trap of its own. `trap ... EXIT` REPLACES the existing handler rather # than adding to it, so a second trap here silently discards the one that # restarts the service, and a backup run leaves the service stopped. That is # precisely the failure the restart trap exists to prevent. report_outcome() { local rc="$1" if [ "$rc" -ne 0 ]; then report "false" "${NAME} dump exited ${rc}" elif [ ! -s "$ARTIFACT" ]; then report "false" "${NAME} produced no artefact at ${ARTIFACT}" else report "true" "${NAME} $(du -h "$ARTIFACT" | cut -f1)" fi } # --- One EXIT handler, doing both jobs ------------------------------------- # bash keeps exactly ONE EXIT trap: `trap ... EXIT` REPLACES the previous # handler rather than adding to it. Registering a second one here would # silently discard the service restart and leave the service stopped after # every backup - which is the exact bug the restart exists to prevent, and it # is invisible until someone notices the service is down. on_exit() { local rc=$? {% if backup_source_stop_service or backup_source_stop_command %} log "Restarting ${SERVICE}..." eval "$START_CMD" || true {% endif %} report_outcome "$rc" } trap on_exit EXIT {% if backup_source_stop_service or backup_source_stop_command %} # --- Stop the service; the trap above guarantees it comes back ------------- # The trap is the point: without it a failed dump leaves the service down until # the next timer fires. Every hand-written script this replaced had that bug. # It is armed BEFORE the stop, so even a failure during the stop restarts. log "Stopping ${SERVICE}..." eval "$STOP_CMD" {% endif %} # --- Dump straight into age; plaintext never touches the disk --- log "Writing ${ARTIFACT}..." {{ backup_source_dump_command }} | age -r "$RECIPIENT" -o "${ARTIFACT}.partial" {% if backup_source_pull_user %} # Match the final ownership immediately, so even a partial left by a later # failure is not an unreadable obstacle to the pull. chown root:{{ backup_source_pull_user }} "${ARTIFACT}.partial" chmod 640 "${ARTIFACT}.partial" {% endif %} mv "${ARTIFACT}.partial" "$ARTIFACT" {% if backup_source_pull_user %} # Readable by the pull account and nobody else. The contents are age-encrypted # regardless, so this is depth rather than the actual protection. chown root:{{ backup_source_pull_user }} "$ARTIFACT" chmod 640 "$ARTIFACT" {% else %} chmod 600 "$ARTIFACT" {% endif %} log "Wrote ${ARTIFACT} ($(du -h "$ARTIFACT" | cut -f1))" # --- Prune --- log "Pruning local artefacts older than ${RETENTION_DAYS} days..." find "$BACKUP_DIR" -maxdepth 1 -type f -name "${NAME}_*.${SUFFIX}" -mtime +"${RETENTION_DAYS}" -delete log "Done."