Files
Hermes 1cc112f1e4 DC-038: backup trigger.json + result.json in dashcaddy-update.sh
The host-side updater only backed up code + data/, leaving trigger.json and
result.json unarchived. After a failed update, operators had to reconstruct
'what was being attempted' by joining timestamps across files. Now the
backup captures both files into a 'update-state/' subdir alongside code +
data backups, keyed by from-version.

- New `backup_update_state()` function in dashcaddy-update.sh: idempotent,
  tolerates absent files (cleans up empty subdir), tolerates chattr +i
  (unlock/copy/relock).
- Wired into main() right after `backup_data_dir`, before `cleanup_old_backups`.
- Deliberately does NOT auto-restore trigger.json on rollback — the rollback
  handler reads a fresh trigger.json written by the operator/container;
  restoring the previous attempt's trigger would clobber the active rollback
  request. Backups are read-only forensic evidence.
- New `dashcaddy-api/scripts/test-dashcaddy-update-backup.sh` (14 assertions,
  5 test groups): both-files-present, partial-present, no-files-present,
  idempotency, main() flow ordering. All 14 pass.
- Synced the duplicate at `dashcaddy-api/scripts/dashcaddy-update.sh`
  (md5-identical to scripts/dashcaddy-update.sh).

Tests: 1214/1214 pass (zero change). Lint: 150 warnings, all pre-existing
in untouched files (zero new warnings introduced).
2026-07-13 13:27:38 -07:00

523 lines
20 KiB
Bash
Executable File

#!/usr/bin/env bash
# DashCaddy Host-Side Updater
# Triggered by systemd path unit when the container writes trigger.json.
# Reads the trigger, backs up current API + data/, copies new files, rebuilds container.
# Writes result.json so the new container knows the outcome.
#
# This runs on the HOST, outside the container.
#
# Channel selection: by default only "stable" releases are applied. Set
# ALLOW_PRERELEASE=true in /opt/dashcaddy/updates/channel.conf to opt in to
# prerelease/beta/rc channels. Useful for staging hosts, not production.
set -euo pipefail
readonly UPDATES_DIR="/opt/dashcaddy/updates"
readonly TRIGGER_FILE="${UPDATES_DIR}/trigger.json"
readonly RESULT_FILE="${UPDATES_DIR}/result.json"
readonly BACKUPS_DIR="${UPDATES_DIR}/backups"
readonly CONTAINER_NAME="dashcaddy-api"
readonly IMAGE_TAG="dashcaddy-dashcaddy-api:latest"
readonly MAX_BACKUPS=3
readonly HEALTH_TIMEOUT=60
readonly CHANNEL_CONF="${UPDATES_DIR}/channel.conf"
# Data directory backup — stored alongside code backups so everything rolls back together
readonly DATA_SOURCE_DIR="/opt/dashcaddy/dashcaddy-api/data"
readonly DATA_BACKUP_PREFIX="data-backup"
# Updater state (trigger.json / result.json) backup — keeps the audit trail
# (what version we were attempting, what the previous update's outcome was) tied
# to the same versioned backup directory as code + data. After a failed update,
# operators can inspect what was attempted without correlating timestamps, and
# rollback tooling can reconstruct a "what just happened" view of the update
# state machine. NOTE: we do NOT auto-restore trigger.json on rollback — the
# rollback handler reads a fresh trigger.json written by the operator/container;
# restoring the previous attempt's trigger would clobber the active rollback
# request. Backups here are read-only forensic evidence.
readonly UPDATE_STATE_BACKUP_PREFIX="update-state"
readonly TRIGGER_PROCESSING="${TRIGGER_FILE}.processing"
log() { echo "[dashcaddy-update] $(date '+%Y-%m-%d %H:%M:%S') $*"; }
# Decide if a given release channel is acceptable on this host.
# Returns 0 (accept) or 1 (reject) and logs the reason.
channel_allowed() {
local channel="$1"
local allow_prerelease="false"
if [[ -f "$CHANNEL_CONF" ]]; then
# shellcheck disable=SC1090
source "$CHANNEL_CONF"
allow_prerelease="${ALLOW_PRERELEASE:-false}"
fi
case "${channel,,}" in
stable|"")
return 0
;;
prerelease|beta|rc|alpha)
if [[ "${allow_prerelease,,}" == "true" ]]; then
log "Channel '${channel}' accepted (ALLOW_PRERELEASE=true in ${CHANNEL_CONF})"
return 0
else
log "Channel '${channel}' rejected — set ALLOW_PRERELEASE=true in ${CHANNEL_CONF} to accept"
return 1
fi
;;
*)
log "Channel '${channel}' rejected — unknown channel"
return 1
;;
esac
}
write_result() {
local success="$1" version="$2" duration="$3"
shift 3
local error="${1:-}"
if [[ "$success" == "true" ]]; then
cat > "$RESULT_FILE" <<EOF
{
"success": true,
"version": "${version}",
"duration": ${duration},
"timestamp": "$(date -u +%Y-%m-%dT%H:%M:%SZ)"
}
EOF
else
cat > "$RESULT_FILE" <<EOF
{
"success": false,
"version": "${version}",
"duration": ${duration},
"error": "${error}",
"timestamp": "$(date -u +%Y-%m-%dT%H:%M:%SZ)"
}
EOF
fi
}
cleanup_old_backups() {
local count
count=$(find "$BACKUPS_DIR" -maxdepth 1 -mindepth 1 -type d 2>/dev/null | wc -l)
if (( count > MAX_BACKUPS )); then
log "Cleaning old backups (${count} > ${MAX_BACKUPS})"
find "$BACKUPS_DIR" -maxdepth 1 -mindepth 1 -type d -printf '%T+ %p\n' \
| sort | head -n $(( count - MAX_BACKUPS )) | cut -d' ' -f2- \
| xargs rm -rf
fi
}
# ── Data backup (rsync for efficiency + permissions) ──────────────────────────
backup_data_dir() {
local backup_dir="$1"
if [[ -d "$DATA_SOURCE_DIR" ]]; then
log "Backing up data/ to ${backup_dir}/${DATA_BACKUP_PREFIX}/"
mkdir -p "${backup_dir}/${DATA_BACKUP_PREFIX}"
rsync -a --delete "$DATA_SOURCE_DIR/" "${backup_dir}/${DATA_BACKUP_PREFIX}/" 2>/dev/null \
|| cp -a "$DATA_SOURCE_DIR" "${backup_dir}/${DATA_BACKUP_PREFIX}"
log "Data backup complete ($(du -sh "${backup_dir}/${DATA_BACKUP_PREFIX}" 2>/dev/null | cut -f1))"
else
log "WARNING: Data source dir $DATA_SOURCE_DIR not found — skipping data backup"
fi
}
# ── Updater state backup (trigger.json.processing + result.json) ─────────────
# Captures what was being attempted + the last result so post-mortem can answer
# "why did this fail" without joining timestamps across files. Tolerates absent
# files (first-ever run) and locked files (chattr +i). Idempotent — re-running
# overwrites the previous backup.
backup_update_state() {
local backup_dir="$1"
local state_dir="${backup_dir}/${UPDATE_STATE_BACKUP_PREFIX}"
mkdir -p "$state_dir"
local copied=0
for src in "$TRIGGER_PROCESSING" "$RESULT_FILE"; do
if [[ -f "$src" ]]; then
# Unlock temporarily if immutable, copy, re-lock.
local was_locked=false
if lsattr -d "$src" 2>/dev/null | awk '{exit !($1 ~ /i/)}'; then
was_locked=true
chattr -i "$src" 2>/dev/null || true
fi
cp -f "$src" "${state_dir}/$(basename "$src")" 2>/dev/null && copied=$(( copied + 1 ))
if [[ "$was_locked" == "true" ]]; then
chattr +i "$src" 2>/dev/null || true
fi
fi
done
if (( copied > 0 )); then
log "Update-state backup: ${copied} file(s) -> ${state_dir}"
else
log "Update-state backup: nothing to back up (no trigger/result files)"
rmdir "$state_dir" 2>/dev/null || true
fi
}
# ── Data restore ──────────────────────────────────────────────────────────────
restore_data_dir() {
local backup_dir="$1"
local data_backup="${backup_dir}/${DATA_BACKUP_PREFIX}"
if [[ -d "$data_backup" ]]; then
log "Restoring data/ from backup..."
rsync -a --delete "$data_backup/" "$DATA_SOURCE_DIR/" 2>/dev/null \
|| cp -a "$data_backup" "$DATA_SOURCE_DIR"
log "Data restored successfully"
else
log "WARNING: No data backup found at ${data_backup} — data/ not restored"
fi
}
wait_for_health() {
local port="${1:-3001}"
local timeout="$HEALTH_TIMEOUT"
local elapsed=0
log "Waiting for health check (timeout: ${timeout}s)..."
while (( elapsed < timeout )); do
if curl -fsSL --max-time 3 "http://localhost:${port}/health" &>/dev/null; then
log "Health check passed after ${elapsed}s"
return 0
fi
sleep 2
elapsed=$(( elapsed + 2 ))
done
log "Health check FAILED after ${timeout}s"
return 1
}
# ── Shared rollback: restore code + data ────────────────────────────────────
rollback_restore() {
local backup_dir="$1"
log "Rolling back: restoring code files..."
for item in "$backup_dir"/*.js "$backup_dir"/package.json "$backup_dir"/package-lock.json "$backup_dir"/Dockerfile "$backup_dir"/openapi.yaml "$backup_dir"/VERSION; do
[[ -f "$item" ]] && cp -f "$item" "$api_source_dir/" 2>/dev/null || true
done
if [[ -d "$backup_dir/routes" ]]; then
rm -rf "$api_source_dir/routes"
cp -rf "$backup_dir/routes" "$api_source_dir/routes"
fi
if [[ -d "$backup_dir/src" ]]; then
rm -rf "$api_source_dir/src"
cp -rf "$backup_dir/src" "$api_source_dir/src"
fi
if [[ -d "$backup_dir/dns-providers" ]]; then
rm -rf "$api_source_dir/dns-providers"
cp -rf "$backup_dir/dns-providers" "$api_source_dir/dns-providers"
fi
restore_data_dir "$backup_dir"
}
# ── Deployment mode ───────────────────────────────────────────────────────────
# Reproduce the SAME container the install created so an auto-update keeps every
# volume + env var (docker socket, Caddyfile, config/credentials, updates mount),
# not a minimal subset. Standard installs use docker-compose (compose file in the
# api source dir); the publish/dev host uses /opt/dashcaddy/start.sh; otherwise a
# bare docker run is the last resort. build_image() and restart_container() both
# honor the detected mode so build and run stay consistent.
deploy_mode() {
if [[ -f "$api_source_dir/docker-compose.yml" || -f "$api_source_dir/compose.yml" || -f "$api_source_dir/compose.yaml" ]]; then
echo compose
elif [[ -x /opt/dashcaddy/start.sh ]]; then
echo startsh
else
echo run
fi
}
# Build the API image using whatever the install is wired for. Returns the build
# command's exit status so callers can detect failure.
build_image() {
cd "$api_source_dir" || return 1
case "$(deploy_mode)" in
compose) docker compose build 2>&1 || docker-compose build 2>&1 ;;
*) docker build -t "$IMAGE_TAG" . 2>&1 ;;
esac
}
# ── Shared container restart — recreate with the full, install-defined spec ───
# Recreates (rm + run / compose up) so new code AND new env vars take effect.
restart_container() {
cd "$api_source_dir" 2>/dev/null || true
case "$(deploy_mode)" in
compose)
log "Recreating container via docker compose (full compose spec)..."
docker compose up -d 2>&1 || docker-compose up -d 2>&1
;;
startsh)
log "Recreating container via /opt/dashcaddy/start.sh (full container spec)..."
bash /opt/dashcaddy/start.sh
;;
*)
log "Recreating container via minimal docker run (fallback)..."
docker rm -f "$CONTAINER_NAME" 2>/dev/null || true
docker run -d --restart unless-stopped --name "$CONTAINER_NAME" \
-p 127.0.0.1:3001:3001 \
-v /opt/dashcaddy/dashcaddy-api/data:/app/data \
-e SERVICES_FILE=/app/data/services.json \
"$IMAGE_TAG"
;;
esac
log "Container recreated"
}
# ── Code-only restore (used after failed build when data hasn't changed yet) ──
code_restore() {
local backup_dir="$1"
log "Restoring code files..."
for item in "$backup_dir"/*.js "$backup_dir"/package.json "$backup_dir"/package-lock.json "$backup_dir"/Dockerfile "$backup_dir"/openapi.yaml "$backup_dir"/VERSION; do
[[ -f "$item" ]] && cp -f "$item" "$api_source_dir/" 2>/dev/null || true
done
if [[ -d "$backup_dir/routes" ]]; then
rm -rf "$api_source_dir/routes"
cp -rf "$backup_dir/routes" "$api_source_dir/routes"
fi
if [[ -d "$backup_dir/src" ]]; then
rm -rf "$api_source_dir/src"
cp -rf "$backup_dir/src" "$api_source_dir/src"
fi
if [[ -d "$backup_dir/dns-providers" ]]; then
rm -rf "$api_source_dir/dns-providers"
cp -rf "$backup_dir/dns-providers" "$api_source_dir/dns-providers"
fi
}
main() {
local start_time
start_time=$(date +%s)
# 1. Read trigger
if [[ ! -f "$TRIGGER_FILE" ]]; then
log "No trigger file found — nothing to do"
exit 0
fi
# Parse trigger.json (uses python3 which is available on all supported distros)
local action version from_version staging_dir api_source_dir commit channel
local frontend_staging_dir frontend_target_dir
action=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}'))['action'])")
version=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}'))['version'])")
from_version=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}'))['fromVersion'])")
staging_dir=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}'))['stagingDir'])")
api_source_dir=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}'))['apiSourceDir'])")
commit=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}')).get('commit') or '')")
frontend_staging_dir=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}')).get('frontendStagingDir') or '')")
frontend_target_dir=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}')).get('frontendTargetDir') or '')")
channel=$(python3 -c "import json; print(json.load(open('${TRIGGER_FILE}')).get('channel') or 'stable')")
# Handle action=rollback (no new version to deploy)
local to_version="${version}"
log "=== ${action^^}: v${from_version} -> v${to_version} (channel: ${channel}) ==="
log "Staging: ${staging_dir}"
log "API source: ${api_source_dir}"
# Consume the trigger immediately so we don't re-process on failure
mv "$TRIGGER_FILE" "${TRIGGER_FILE}.processing"
# Channel gate: refuse to apply prereleases unless explicitly opted-in.
# Rollbacks always allowed (no new release channel involved).
if [[ "${action}" != "rollback" ]] && ! channel_allowed "${channel}"; then
write_result "false" "$to_version" "0" "Channel '${channel}' not allowed on this host"
rm -f "${TRIGGER_FILE}.processing"
exit 1
fi
# ── Handle rollback ────────────────────────────────────────────────────────
if [[ "$action" == "rollback" ]]; then
local backup_dir="${BACKUPS_DIR}/${version}"
if [[ ! -d "$backup_dir" ]]; then
log "ERROR: No backup found for version ${version}"
write_result "false" "$version" "$(( $(date +%s) - start_time ))" "No backup found for version ${version}"
rm -f "${TRIGGER_FILE}.processing"
exit 1
fi
log "Performing rollback to v${version}..."
rollback_restore "$backup_dir"
# Rebuild old code
log "Rebuilding container..."
build_image 2>&1 | tail -3 || true
restart_container
wait_for_health || log "WARNING: Health check failed after rollback"
write_result "true" "$version" "$(( $(date +%s) - start_time ))"
rm -f "${TRIGGER_FILE}.processing"
log "=== Rollback complete ==="
exit 0
fi
# ── Handle update ───────────────────────────────────────────────────────────
if [[ ! -d "$staging_dir" ]]; then
log "ERROR: Staging directory not found: ${staging_dir}"
write_result "false" "$to_version" "$(( $(date +%s) - start_time ))" "Staging directory not found"
rm -f "${TRIGGER_FILE}.processing"
exit 1
fi
# 2. Backup current API code + data/
local backup_dir="${BACKUPS_DIR}/${from_version}"
mkdir -p "$backup_dir"
log "Backing up current API files to ${backup_dir}"
for item in "$api_source_dir"/*.js "$api_source_dir"/package.json "$api_source_dir"/package-lock.json "$api_source_dir"/Dockerfile "$api_source_dir"/openapi.yaml "$api_source_dir"/VERSION; do
[[ -f "$item" ]] && cp -f "$item" "$backup_dir/" 2>/dev/null || true
done
[[ -d "$api_source_dir/routes" ]] && cp -rf "$api_source_dir/routes" "$backup_dir/"
[[ -d "$api_source_dir/src" ]] && cp -rf "$api_source_dir/src" "$backup_dir/"
[[ -d "$api_source_dir/dns-providers" ]] && cp -rf "$api_source_dir/dns-providers" "$backup_dir/"
# Backup data/ directory (services.json, config.json, credentials, etc.)
backup_data_dir "$backup_dir"
# Backup updater state (trigger.json.processing + result.json) so post-mortem
# has a forensic trail tied to this exact version's backup.
backup_update_state "$backup_dir"
cleanup_old_backups
# 3. Copy new files from staging to API source
log "Deploying new API files..."
for item in "$staging_dir"/*.js "$staging_dir"/package.json "$staging_dir"/package-lock.json "$staging_dir"/Dockerfile "$staging_dir"/openapi.yaml "$staging_dir"/VERSION; do
[[ -f "$item" ]] && cp -f "$item" "$api_source_dir/" 2>/dev/null || true
done
# Safety: only replace routes/src if staging has the dir AND it's non-empty.
# An empty or partial staging dir used to cause live routes/src to be wiped
# when a prior update cycle was interrupted. We also handle locked files
# (chattr +i) by temporarily unlocking before replace and re-locking after.
deploy_tree() {
local rel="$1" # e.g. "routes"
local src="${staging_dir}/${rel}"
local dst="${api_source_dir}/${rel}"
if [[ ! -d "$src" ]] || [[ -z "$(ls -A "$src" 2>/dev/null)" ]]; then
[[ -d "$src" ]] && log "WARNING: staging ${rel}/ exists but is empty — leaving live ${rel}/ untouched"
return 0
fi
# Collect any locked files (chattr +i) in the destination. lsattr's
# first field is the attribute flags ("i" at position 5 = immutable);
# the second field is the filename. We unlock before rm -rf and re-lock
# after so the locked state survives the update.
local locked_files=()
if [[ -d "$dst" ]]; then
while IFS= read -r lf; do
[[ -n "$lf" ]] && locked_files+=("$lf")
done < <(find "$dst" -type f \( -name "*.js" -o -name "*.json" -o -name "*.sh" \) -print0 2>/dev/null \
| xargs -0 lsattr -a 2>/dev/null \
| awk '$1 ~ /i/ { print $2 }')
fi
for lf in "${locked_files[@]:-}"; do
[[ -n "$lf" ]] && chattr -i "$lf" 2>/dev/null || true
done
rm -rf "$dst"
cp -rf "$src" "$dst"
local file_count
file_count=$(find "$dst" -type f 2>/dev/null | wc -l)
log "${rel}/ deployed (${file_count} files)"
for lf in "${locked_files[@]:-}"; do
[[ -n "$lf" ]] && [[ -f "$lf" ]] && chattr +i "$lf" 2>/dev/null || true
done
}
deploy_tree "routes"
deploy_tree "src"
deploy_tree "dns-providers"
if [[ -n "$commit" ]]; then
echo "$commit" > "$api_source_dir/VERSION"
fi
# 3a. Apply post-deploy patches — fix upstream bugs in released tarballs
# (e.g. v1.14.4 has broken require paths and missing license-keygen module).
# Runs AFTER staging copy, BEFORE docker build. Idempotent.
local patch_script="/opt/dashcaddy/scripts/dashcaddy-post-deploy-patches.sh"
if [[ -x "$patch_script" ]]; then
log "Applying post-deploy patches..."
if "$patch_script" "$api_source_dir"; then
log "Post-deploy patches applied successfully"
else
log "WARNING: Post-deploy patches exited non-zero — continuing build anyway"
fi
else
log "NOTE: $patch_script not found or not executable — skipping post-deploy patches"
fi
# 3b. Sync frontend
if [[ -z "$frontend_staging_dir" ]]; then
parent_staging=$(dirname "$staging_dir")
[[ -d "$parent_staging/status" ]] && frontend_staging_dir="$parent_staging/status"
fi
if [[ -z "$frontend_target_dir" ]]; then
for candidate in /var/www/dashcaddy-status /etc/dashcaddy/sites/status; do
[[ -d "$candidate" ]] && frontend_target_dir="$candidate" && break
done
fi
if [[ -n "$frontend_staging_dir" && -n "$frontend_target_dir" && -d "$frontend_staging_dir" ]]; then
log "Syncing frontend: $frontend_staging_dir -> $frontend_target_dir"
mkdir -p "$frontend_target_dir"
[[ -f "$frontend_staging_dir/index.html" ]] && cp -f "$frontend_staging_dir/index.html" "$frontend_target_dir/index.html"
[[ -f "$frontend_staging_dir/sw.js" ]] && cp -f "$frontend_staging_dir/sw.js" "$frontend_target_dir/sw.js"
for sub in dist css vendor js; do
if [[ -d "$frontend_staging_dir/$sub" ]]; then
mkdir -p "$frontend_target_dir/$sub"
cp -rf "$frontend_staging_dir/$sub/"* "$frontend_target_dir/$sub/" 2>/dev/null || true
fi
done
if [[ -d "$frontend_staging_dir/assets" ]]; then
mkdir -p "$frontend_target_dir/assets"
cp -rf "$frontend_staging_dir/assets/"* "$frontend_target_dir/assets/" 2>/dev/null || true
fi
fi
# 4. Rebuild container
log "Rebuilding container..."
local build_ok=false
if build_image; then
build_ok=true
fi
if [[ "$build_ok" != "true" ]]; then
log "ERROR: Docker build failed — rolling back code + data"
code_restore "$backup_dir"
build_image 2>&1 | tail -3 || true
restart_container
wait_for_health || true
write_result "false" "$to_version" "$(( $(date +%s) - start_time ))" "Docker build failed"
rm -f "${TRIGGER_FILE}.processing"
exit 1
fi
# 5. Restart container (recreate so new code + env vars take effect)
restart_container
# 6. Health check
if wait_for_health; then
local duration=$(( $(date +%s) - start_time ))
log "=== Update successful: v${to_version} in ${duration}s ==="
write_result "true" "$to_version" "$duration"
else
local duration=$(( $(date +%s) - start_time ))
log "ERROR: Health check failed after update — rolling back code + data"
rollback_restore "$backup_dir"
build_image 2>&1 | tail -3 || true
restart_container
wait_for_health || log "WARNING: Rollback health check also failed"
write_result "false" "$to_version" "$duration" "Health check failed after update"
fi
# 7. Cleanup
rm -f "${TRIGGER_FILE}.processing"
rm -rf "${UPDATES_DIR}/staging" 2>/dev/null || true
log "=== Update process complete ==="
}
main "$@"