545 lines
26 KiB
Bash
Executable File
545 lines
26 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
#
|
|
# CluPilot — pull the latest code and apply it.
|
|
#
|
|
# The checkout is bind-mounted into every container, so a `git merge` swaps the
|
|
# running code instantly. There is no atomic release directory to switch to;
|
|
# what there is, is maintenance mode. So the order is:
|
|
#
|
|
# down → merge → dependencies → migrate → assets → restart → up
|
|
#
|
|
# That trades a short, deliberate outage for never serving code whose schema
|
|
# does not exist yet. If any step fails the site STAYS down: coming back up with
|
|
# new code on an old schema is worse than staying dark until someone looks.
|
|
set -euo pipefail
|
|
|
|
cd "$(dirname "$0")/.."
|
|
# shellcheck source=deploy/lib/release.sh
|
|
. deploy/lib/release.sh
|
|
|
|
BRANCH="${BRANCH:-main}"
|
|
# RELEASE=v1.2.0 moves a pinned server to that tag. Without it, the mode the
|
|
# server was installed in is kept — a production box pinned to a release does
|
|
# not quietly start following main because someone re-ran the updater.
|
|
RELEASE="${RELEASE:-}"
|
|
|
|
# Never as root: the checkout belongs to the service account, and running this
|
|
# as root would leave root-owned files behind that the app cannot write.
|
|
if [[ $EUID -eq 0 ]]; then
|
|
echo "Run this as the service account, not as root:" >&2
|
|
echo " sudo -u clupilot bash $0" >&2
|
|
exit 1
|
|
fi
|
|
STATE_FILE="storage/app/deployed-commit"
|
|
# The root-owned half of an update, and the version of what it can do. Both are
|
|
# fixed by deploy/install-agent.sh; this script only asks. Raising HOST_STEP_NEEDS
|
|
# here is what makes an update tell the operator to run the installer again.
|
|
HOST_STEP=/usr/local/sbin/clupilot-host-step
|
|
HOST_STEP_NEEDS=1
|
|
# Which step is running, for the console to show. Written as a KEY, not as the
|
|
# sentence below it: the console is translated and this script is not, so an
|
|
# English line here would surface untranslated in the interface. A run that dies
|
|
# leaves the key of the step it died at, which is the one thing the operator
|
|
# needs and the log alone makes them hunt for.
|
|
PHASE_FILE="storage/app/deploy/update-phase"
|
|
|
|
log() { printf '\n\033[1;34m==>\033[0m %s\n' "$*"; }
|
|
warn() { printf '\033[1;33m !\033[0m %s\n' "$*"; }
|
|
|
|
# A step, announced to the terminal and recorded for the console.
|
|
phase() {
|
|
local key="$1"; shift
|
|
mkdir -p "$(dirname "$PHASE_FILE")"
|
|
# Not atomic on purpose: this is a hint for a progress line, and a torn read
|
|
# costs a single poll. Writing a temp file per step would be more moving
|
|
# parts than the thing is worth.
|
|
printf '%s\t%s\n' "$key" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" > "$PHASE_FILE" 2>/dev/null || true
|
|
log "$*"
|
|
}
|
|
|
|
# As the application's own user, not as root.
|
|
#
|
|
# `docker compose exec` runs as root unless told otherwise, and this did — so
|
|
# when `artisan optimize` failed during a deployment it wrote its error into
|
|
# storage/logs/laravel.log AS ROOT. From that moment the application could not
|
|
# append to its own log: Monolog threw on every attempt, and a throw while
|
|
# logging is a 500 on every page that logs, with nothing in the log to say why.
|
|
# It surfaced as a 500 on the VPN config download — one of the few pages that
|
|
# writes a log line on the way through — and looked for days like a fault in
|
|
# the VPN code.
|
|
#
|
|
# docker/entrypoint.sh already drops to www-data for exactly these commands.
|
|
# The deployment simply never did.
|
|
in_app() { docker compose exec -T -u www-data app "$@"; }
|
|
|
|
# Hand back anything an earlier run left owned by root.
|
|
#
|
|
# Not tidiness. That damage is silent and permanent — nothing repairs a
|
|
# root-owned log file on its own, and the symptom surfaces on whichever page
|
|
# happens to log first, which is nowhere near the deployment that caused it. A
|
|
# server already carrying it heals on its next deployment instead of needing
|
|
# somebody to work out what happened.
|
|
#
|
|
# And it is now load-bearing: with in_app unprivileged, a vendor/ or
|
|
# node_modules/ left owned by root would make the very next composer or npm
|
|
# step fail outright.
|
|
normalise_ownership() {
|
|
# find, not `stat` on the directory itself. The first version of this
|
|
# sampled the top-level owner and skipped the recursion when it matched —
|
|
# and node_modules was owned by www-data while node_modules/.vite-temp,
|
|
# left by an earlier root build, was not. The deployment then failed at
|
|
# `npm run build` with EACCES, in maintenance mode, on the very release
|
|
# that was supposed to prevent it. A directory's owner says nothing about
|
|
# its contents.
|
|
#
|
|
# One walk that changes only what is wrong, rather than a detection pass
|
|
# and then a blanket chown -R over thirty thousand files.
|
|
docker compose exec -T -u root app sh -c '
|
|
for d in storage bootstrap/cache vendor node_modules public/build; do
|
|
[ -d "$d" ] || continue
|
|
find "$d" ! -user www-data -exec chown www-data:www-data {} + 2>/dev/null || true
|
|
done
|
|
' >/dev/null 2>&1 || true
|
|
}
|
|
|
|
down=0
|
|
finish() {
|
|
local code=$?
|
|
if [[ $code -ne 0 && $down -eq 1 ]]; then
|
|
warn "Update failed. The site is STILL in maintenance mode on purpose."
|
|
warn "Fix the problem and re-run this script, or force it back up with:"
|
|
warn " docker compose exec -u www-data app php artisan up"
|
|
fi
|
|
exit $code
|
|
}
|
|
trap finish EXIT
|
|
|
|
# Which line this server follows, and what it is being moved to.
|
|
mode="$(release_mode)"
|
|
[[ -n "$RELEASE" ]] && mode="release"
|
|
|
|
if [[ "$mode" == "release" ]]; then
|
|
# A pinned server without an explicit RELEASE has nothing to do: staying on
|
|
# the tag is the whole point of being pinned.
|
|
if [[ -z "$RELEASE" ]]; then
|
|
RELEASE="$(release_source)"
|
|
RELEASE="${RELEASE#refs/tags/}"
|
|
[[ -n "$RELEASE" ]] || { echo "Pinned to a release, but no tag recorded. Pass RELEASE=vX.Y.Z." >&2; exit 1; }
|
|
fi
|
|
|
|
phase fetch "Fetching release $RELEASE"
|
|
git fetch --quiet --tags --force origin
|
|
git rev-parse -q --verify "refs/tags/${RELEASE}^{commit}" >/dev/null \
|
|
|| { echo "No such release tag: ${RELEASE}" >&2; exit 1; }
|
|
|
|
target="$(git rev-parse "refs/tags/${RELEASE}^{commit}")"
|
|
source_ref="refs/tags/${RELEASE}"
|
|
else
|
|
phase fetch "Fetching $BRANCH"
|
|
git fetch --quiet origin "$BRANCH"
|
|
target="$(git rev-parse "origin/$BRANCH")"
|
|
source_ref="$BRANCH"
|
|
fi
|
|
|
|
before="$(git rev-parse HEAD)"
|
|
|
|
# Backwards is not an update. The database has already been migrated forward,
|
|
# and older code against a newer schema is the one failure this script exists to
|
|
# prevent — with the added cruelty that the migrations needed to roll back are
|
|
# not in the older checkout at all. A genuine downgrade means restoring the
|
|
# pre-upgrade database snapshot first, deliberately, by hand.
|
|
if release_would_go_backwards "$before" "$target"; then
|
|
cat >&2 <<BACKWARDS
|
|
Refusing to move backwards: ${source_ref} is behind what is deployed.
|
|
|
|
deployed $(git rev-parse --short "$before")
|
|
requested $(git rev-parse --short "$target")
|
|
|
|
The schema has already moved forward, and older code against a newer schema is
|
|
what maintenance mode exists to avoid. To go back: restore the database
|
|
snapshot taken before the upgrade, then install the older release.
|
|
BACKWARDS
|
|
exit 1
|
|
fi
|
|
# What was last deployed successfully — not what is merely checked out. A run
|
|
# that died halfway leaves these different, and the next run finishes the job
|
|
# instead of declaring "already up to date".
|
|
deployed="$(cat "$STATE_FILE" 2>/dev/null || echo '')"
|
|
|
|
# Readiness is re-established after every start, never carried over. `up -d`
|
|
# succeeds once a container is CREATED, so one that starts and exits — an
|
|
# unreadable certificate, an address it cannot bind — would leave a previous
|
|
# VPN_READY=true in place and the application still advertising a resolver that
|
|
# is no longer running.
|
|
#
|
|
# A function because there are two ways out of this script: a full deploy, and
|
|
# the shortcut taken when the checkout has not moved. Both start services; both
|
|
# have to record what actually came up.
|
|
reconcile_vpn_readiness() {
|
|
grep -qE '^COMPOSE_PROFILES=.*vpn' .env 2>/dev/null || return 0
|
|
|
|
local host vpn_ready=false
|
|
host="$(sed -n 's/^VPN_INTERNAL_HOST=//p' .env | tail -1)"
|
|
|
|
# ASKED, not assumed. A container can be up and running while the process
|
|
# inside it listens in a network namespace that was torn down underneath it
|
|
# — nothing errors, the address simply refuses connections, and a
|
|
# container-state check reports everything fine.
|
|
#
|
|
# Asked at the address the gateway actually serves. It binds to the hub
|
|
# address alone (see docker/caddy/vpn.Caddyfile), so nothing has ever
|
|
# listened on 127.0.0.1 — the previous probe could only ever fail, and it
|
|
# did: VPN_READY stayed false on a perfectly healthy tunnel, and because of
|
|
# that the application withheld the resolver from every client config it
|
|
# issued. The console was then unreachable over the VPN for weeks, with the
|
|
# deployment printing a warning that read like a gateway fault.
|
|
#
|
|
# Over plain HTTP on the health port, not HTTPS: the TLS site matches on the
|
|
# console's hostname, so a request to the bare address presents no SNI, gets
|
|
# no certificate and fails the handshake — which looks exactly like the
|
|
# outage this is meant to detect. Caddy refuses to start at all if the
|
|
# certificate is unreadable, so a health port that answers still proves the
|
|
# whole gateway loaded.
|
|
# The same variable compose hands the gateway, so the two cannot drift.
|
|
local hub port health
|
|
hub="$(sed -n 's/^CLUPILOT_WG_HUB_ADDRESS=//p' .env | tail -1)"
|
|
hub="${hub:-10.66.0.1}"
|
|
port="$(sed -n 's/^VPN_HEALTH_PORT=//p' .env | tail -1)"
|
|
health="http://${hub}:${port:-8081}/healthz"
|
|
|
|
# Long enough for a restart to finish. These containers are restarted twice
|
|
# in a deployment — once by `up -d`, then again after the hub they live
|
|
# inside — and `docker compose exec` into one that is still coming up fails
|
|
# outright rather than waiting. Twenty seconds was not enough for that, and
|
|
# the run then reported a healthy gateway as down.
|
|
for _ in $(seq 1 30); do
|
|
if docker compose exec -T vpn-gateway sh -c \
|
|
"wget -q --spider '$health' 2>/dev/null || wget -q -O /dev/null '$health' 2>/dev/null" >/dev/null 2>&1; then
|
|
vpn_ready=true
|
|
break
|
|
fi
|
|
sleep 2
|
|
done
|
|
|
|
if [[ "$vpn_ready" != "true" ]]; then
|
|
printf '\033[1;33m !\033[0m %s\n' "The tunnel gateway is not answering on ${host} — clients will not be given its resolver."
|
|
fi
|
|
|
|
if [[ "$(sed -n 's/^VPN_READY=//p' .env | tail -1)" != "$vpn_ready" ]]; then
|
|
sed -i '/^VPN_READY=/d' .env
|
|
printf 'VPN_READY=%s\n' "$vpn_ready" >> .env
|
|
in_app php artisan config:clear >/dev/null 2>&1 || true
|
|
in_app php artisan config:cache >/dev/null 2>&1 || true
|
|
fi
|
|
}
|
|
|
|
# The VPN gateway and resolver live behind a compose profile, and behind a
|
|
# certificate. Enabling the profile on the hostname alone would start a Caddy
|
|
# with an empty tls directive — it fails to load, the internal console never
|
|
# comes up, and the application is meanwhile handing out client configs that
|
|
# point at a resolver nobody started. So: both, or neither, and BEFORE the
|
|
# services are brought up rather than after.
|
|
if ! grep -qE '^VPN_INTERNAL_HOST=..' .env 2>/dev/null; then
|
|
# Cleared. Leaving the profile in place keeps both services running against
|
|
# a placeholder hostname and stale certificate paths, so "empty disables it"
|
|
# would not be true.
|
|
if grep -qE '^COMPOSE_PROFILES=.*vpn' .env 2>/dev/null; then
|
|
# Rebuilt from the remaining entries rather than cut out in place:
|
|
# deleting "vpn" from "ci,vpn" by substitution leaves "ci," and Compose
|
|
# reads the empty entry as a profile.
|
|
profiles="$(sed -n 's/^COMPOSE_PROFILES=//p' .env | tail -1)"
|
|
# `|| true`: with only "vpn" in the list, grep finds nothing and exits
|
|
# 1 — which under `set -e` aborts the update mid-flight, in maintenance
|
|
# mode, with the services it was about to stop still running.
|
|
profiles="$(printf '%s' "$profiles" | tr ',' '\n' | grep -vx 'vpn' | grep -v '^$' | paste -sd, - || true)"
|
|
sed -i '/^COMPOSE_PROFILES=/d' .env
|
|
[[ -n "$profiles" ]] && printf 'COMPOSE_PROFILES=%s\n' "$profiles" >> .env
|
|
docker compose --profile vpn stop vpn-dns vpn-gateway >/dev/null 2>&1 || true
|
|
docker compose --profile vpn rm -f vpn-dns vpn-gateway >/dev/null 2>&1 || true
|
|
# The application caches its configuration, so clearing .env alone
|
|
# would leave it issuing configs that name a resolver just stopped.
|
|
in_app php artisan config:clear >/dev/null 2>&1 || true
|
|
log "Disabled the vpn compose profile — VPN_INTERNAL_HOST is empty"
|
|
fi
|
|
elif grep -qE '^VPN_INTERNAL_HOST=..' .env 2>/dev/null; then
|
|
vpn_cert="$(sed -n 's/^VPN_CERT_PATH=//p' .env | tail -1)"
|
|
vpn_key="$(sed -n 's/^VPN_KEY_PATH=//p' .env | tail -1)"
|
|
|
|
vpn_host="$(sed -n 's/^VPN_INTERNAL_HOST=//p' .env | tail -1)"
|
|
|
|
# The certificate has to belong to the CURRENT hostname. After a rename the
|
|
# old paths are still set and still exist, and the gateway would start
|
|
# serving the previous name's certificate to clients resolving the new one.
|
|
if [[ -n "$vpn_cert" && "$(basename "$vpn_cert")" != "${vpn_host}.crt" ]]; then
|
|
sed -i '/^VPN_CERT_PATH=/d;/^VPN_KEY_PATH=/d;/^VPN_READY=/d' .env
|
|
vpn_cert=''; vpn_key=''
|
|
in_app php artisan config:clear >/dev/null 2>&1 || true
|
|
printf '\033[1;33m !\033[0m %s\n' "The tunnel certificate does not match VPN_INTERNAL_HOST — cleared."
|
|
fi
|
|
|
|
if [[ -z "$vpn_cert" || -z "$vpn_key" ]]; then
|
|
# Both or neither: leaving the profile on with empty tls paths starts a
|
|
# Caddy that cannot load its configuration and crashes forever.
|
|
if grep -qE '^COMPOSE_PROFILES=.*vpn' .env 2>/dev/null; then
|
|
docker compose --profile vpn stop vpn-dns vpn-gateway >/dev/null 2>&1 || true
|
|
profiles="$(sed -n 's/^COMPOSE_PROFILES=//p' .env | tail -1)"
|
|
profiles="$(printf '%s' "$profiles" | tr ',' '\n' | grep -vx 'vpn' | grep -v '^$' | paste -sd, - || true)"
|
|
sed -i '/^COMPOSE_PROFILES=/d' .env
|
|
[[ -n "$profiles" ]] && printf 'COMPOSE_PROFILES=%s\n' "$profiles" >> .env
|
|
sed -i '/^VPN_READY=/d' .env
|
|
fi
|
|
fi
|
|
|
|
if [[ -n "$vpn_cert" && -n "$vpn_key" ]]; then
|
|
if ! grep -qE '^COMPOSE_PROFILES=.*vpn' .env; then
|
|
if grep -qE '^COMPOSE_PROFILES=' .env; then
|
|
sed -i 's/^COMPOSE_PROFILES=\(.*\)$/COMPOSE_PROFILES=\1,vpn/' .env
|
|
else
|
|
printf 'COMPOSE_PROFILES=vpn\n' >> .env
|
|
fi
|
|
log "Enabled the vpn compose profile"
|
|
# Started here, not left to the deploy below: the update may exit
|
|
# early as already deployed, and the profile would then be marked
|
|
# on with nothing running behind it.
|
|
docker compose --profile vpn up -d vpn-dns vpn-gateway >/dev/null 2>&1 || true
|
|
reconcile_vpn_readiness
|
|
fi
|
|
else
|
|
printf '\033[1;33m !\033[0m %s\n' "VPN_INTERNAL_HOST is set but the certificate paths are not."
|
|
printf ' %s\n' "The console is not reachable inside the tunnel. Run: sudo bash $(pwd)/deploy/install-agent.sh"
|
|
fi
|
|
fi
|
|
|
|
if [[ "$before" == "$target" && "$deployed" == "$target" ]]; then
|
|
# Nothing to deploy, but possibly something to decide: pinning a server to
|
|
# a release that happens to be the commit it is already on is a real and
|
|
# sensible thing to do. Taking this exit without recording it would leave
|
|
# the machine looking pinned while the next plain update quietly walks it
|
|
# back onto the branch.
|
|
# The manifest is checked too, not just the mode: if writing it failed at
|
|
# the end of the last run — a full disk is enough — nothing else would ever
|
|
# repair it, and the console would report the previous release forever
|
|
# while every later update declared itself already done.
|
|
if [[ "$(release_mode)" != "$mode" || "$(release_source)" != "$source_ref" \
|
|
|| "$(release_manifest_commit)" != "$before" ]]; then
|
|
log "Already on this commit — recording it as $mode ($source_ref)"
|
|
# And actually pin it. A checkout still attached to the branch is not
|
|
# pinned, whatever the manifest claims: the next thing that touches git
|
|
# moves it, and the machine drifts off the release it is supposed to be
|
|
# nailed to.
|
|
if [[ "$mode" == "release" ]]; then
|
|
git checkout --quiet --detach "$target"
|
|
fi
|
|
release_remember "$mode" "$source_ref"
|
|
release_write_manifest "$before" "$source_ref" "$mode"
|
|
else
|
|
log "Already up to date ($(git rev-parse --short HEAD), ${source_ref})"
|
|
fi
|
|
exit 0
|
|
fi
|
|
|
|
# What the change checks below compare against. After a failed run the checkout
|
|
# is already at the target, so diffing against it would compare a commit with
|
|
# itself and skip the very steps that did not finish — the base has to be the
|
|
# last commit we actually deployed.
|
|
base="$before"
|
|
if [[ -n "$deployed" ]] && git cat-file -e "${deployed}^{commit}" 2>/dev/null; then
|
|
base="$deployed"
|
|
fi
|
|
|
|
if [[ "$before" != "$target" ]]; then
|
|
log "Updating $(git rev-parse --short "$before") → $(git rev-parse --short "$target")"
|
|
git --no-pager log --oneline "$before..$target" | sed 's/^/ /'
|
|
else
|
|
warn "Code is current but the last update did not finish — repeating the steps."
|
|
fi
|
|
|
|
# Before the first unprivileged command, not after: if an earlier run left the
|
|
# log or the vendor directory owned by root, everything below fails on it.
|
|
normalise_ownership
|
|
|
|
phase maintenance_on "Enabling maintenance mode"
|
|
in_app php artisan down --retry=60 >/dev/null || warn "Could not enable maintenance mode (continuing)"
|
|
down=1
|
|
|
|
if [[ "$mode" == "release" ]]; then
|
|
# Detached on purpose: a release is a fixed point, not a line to follow.
|
|
# `git merge --ff-only` would be meaningless here, and on a detached HEAD it
|
|
# is how a pinned server silently rejoins main.
|
|
phase checkout "Checking out $source_ref"
|
|
git checkout --quiet --detach "$target"
|
|
else
|
|
phase checkout "Checking out $BRANCH"
|
|
git merge --quiet --ff-only "origin/$BRANCH"
|
|
fi
|
|
after="$(git rev-parse HEAD)"
|
|
|
|
# The image is only rebuilt when its definition changed — minutes versus seconds.
|
|
if ! git diff --quiet "$base" "$after" -- docker/ 2>/dev/null; then
|
|
phase image "Rebuilding the image"
|
|
docker compose build --quiet app
|
|
# Recreate now, not at the end: everything below runs INSIDE this container,
|
|
# and an update that changes the PHP runtime would otherwise install and
|
|
# migrate under the old one.
|
|
docker compose up -d --force-recreate app
|
|
for _ in $(seq 1 30); do
|
|
in_app php -v >/dev/null 2>&1 && break
|
|
sleep 2
|
|
done
|
|
fi
|
|
|
|
# vendor/ and node_modules/ live in the bind mount, so they shadow whatever the
|
|
# image contains: rebuilding the image does NOT update them. Install explicitly
|
|
# whenever a lockfile moved, or the migration below runs against stale packages.
|
|
if ! git diff --quiet "$base" "$after" -- composer.json composer.lock 2>/dev/null || [[ ! -d vendor ]]; then
|
|
phase composer "Installing PHP dependencies"
|
|
in_app composer install --no-interaction --no-dev --prefer-dist --no-progress --optimize-autoloader
|
|
fi
|
|
|
|
if ! git diff --quiet "$base" "$after" -- package.json package-lock.json 2>/dev/null || [[ ! -d node_modules ]]; then
|
|
phase npm "Installing JS dependencies"
|
|
in_app npm ci --no-fund --no-audit
|
|
fi
|
|
|
|
phase migrate "Applying migrations"
|
|
in_app php artisan migrate --force
|
|
|
|
phase assets "Rebuilding assets"
|
|
in_app npm run build
|
|
|
|
# Before the restarts, not after: a service that starts while the old cache is
|
|
# still on disk loads it and keeps those values for the life of its process.
|
|
phase caches "Rebuilding caches"
|
|
# optimize:clear first, then optimize: a half-warm cache from before the update
|
|
# is what produces a page styled with assets that no longer exist.
|
|
in_app php artisan optimize:clear >/dev/null
|
|
in_app php artisan optimize >/dev/null
|
|
|
|
# Existing installations keep whatever .env already said, so safer defaults in
|
|
# compose protect new installs only. A backend published on 0.0.0.0 is reachable
|
|
# from the internet even with UFW closed — Docker publishes ahead of it — and
|
|
# reaching a backend directly skips every hostname and address rule the proxy
|
|
# enforces, including the one keeping the console private.
|
|
#
|
|
# Rewritten only where a reverse proxy is actually in front. A development box
|
|
# without one genuinely needs the port exposed, and silently taking it away
|
|
# would look like the machine had broken.
|
|
# ACTIVE, not merely installed. A retired package still on disk is not a proxy,
|
|
# and rebinding a development box to loopback because caddy happens to be
|
|
# installed makes it unreachable with no obvious cause.
|
|
proxy_running=0
|
|
systemctl is-active --quiet caddy 2>/dev/null && proxy_running=1
|
|
systemctl is-active --quiet nginx 2>/dev/null && proxy_running=1
|
|
# And actually holding 443 — a service that is up but proxying something else
|
|
# entirely is no reason to take this application off the network.
|
|
if [[ $proxy_running -eq 1 ]] && ! ss -ltn 2>/dev/null | grep -qE ':443\s'; then
|
|
proxy_running=0
|
|
fi
|
|
|
|
if [[ $proxy_running -eq 1 ]]; then
|
|
for var in APP_PORT REVERB_HOST_PORT; do
|
|
value="$(sed -n "s/^${var}=//p" .env | tail -1)"
|
|
|
|
# A bare number: published on every interface.
|
|
if [[ "$value" =~ ^[0-9]+$ ]]; then
|
|
sed -i "s|^${var}=.*|${var}=127.0.0.1:${value}|" .env
|
|
printf '\033[1;33m !\033[0m %s\n' "${var} was published on all interfaces — bound to 127.0.0.1:${value}."
|
|
printf ' %s\n' "Reaching it directly bypassed the reverse proxy entirely. Set ${var}=${value} again if that was deliberate."
|
|
fi
|
|
done
|
|
fi
|
|
|
|
phase restart "Restarting services"
|
|
docker compose up -d
|
|
# Workers hold their PHP classes for the life of the process; without this they
|
|
# keep running the code from before the update.
|
|
docker compose restart queue queue-provisioning scheduler reverb
|
|
|
|
# AFTER the hub, always. Both VPN services live in the provisioning container's
|
|
# network namespace, and a process holds the namespace it started in — so once
|
|
# the hub is restarted they are listening inside one that no longer exists.
|
|
# Nothing errors; connections to the tunnel address are simply refused, which
|
|
# looks exactly like the gateway never having worked.
|
|
if grep -qE '^COMPOSE_PROFILES=.*vpn' .env 2>/dev/null; then
|
|
# And only once wg0 is back. The hub brings the interface up as part of its
|
|
# start command, so restarting these the instant the container is "started"
|
|
# has them binding an address that does not exist yet.
|
|
for _ in $(seq 1 30); do
|
|
docker compose exec -T queue-provisioning ip -4 addr show wg0 2>/dev/null | grep -q 'inet ' && break
|
|
sleep 2
|
|
done
|
|
docker compose --profile vpn restart vpn-dns vpn-gateway >/dev/null 2>&1 || true
|
|
fi
|
|
|
|
reconcile_vpn_readiness
|
|
|
|
# ── Host packages the update can install itself ──────────────────────────────
|
|
# This script cannot install anything on the host: it runs as the service
|
|
# account. What it can do is ask the root-owned helper that install-agent.sh put
|
|
# at a path the service account cannot write to, for one fixed command line that
|
|
# sudoers permits by name. See the long note in deploy/install-agent.sh for why
|
|
# it is built that way and not as sudo on this script.
|
|
#
|
|
# Deliberately after the restart and never fatal. rsync is needed by whoever
|
|
# collects the invoice archive, some minutes or hours from now — it is not worth
|
|
# a failed deployment, and an update that stops here would be a far bigger
|
|
# problem than a backup that has to wait for the next run.
|
|
if ! command -v rsync >/dev/null 2>&1; then
|
|
if [[ -x "$HOST_STEP" ]] && sudo -n "$HOST_STEP" ensure-rsync >/dev/null 2>&1; then
|
|
log "Installed rsync on the host (the invoice archive is collected over ssh)"
|
|
else
|
|
warn "rsync is not installed on this host — the invoice archive cannot be collected."
|
|
fi
|
|
fi
|
|
|
|
phase maintenance_off "Leaving maintenance mode"
|
|
in_app php artisan up >/dev/null
|
|
down=0
|
|
|
|
printf '%s' "$after" > "$STATE_FILE"
|
|
|
|
# Only now: the manifest is what the console reports, and it must mean "this
|
|
# came up". Written after `artisan up`, atomically, so a reader never catches it
|
|
# half-finished.
|
|
release_remember "$mode" "$source_ref"
|
|
release_write_manifest "$after" "$source_ref" "$mode"
|
|
|
|
# The panel's update button needs a host-side timer, and this script cannot
|
|
# install it: it runs as the service account and /etc/systemd is root's. Say so
|
|
# once per update rather than leaving the button disabled without explanation.
|
|
agent_hint=''
|
|
if ! systemctl list-unit-files clupilot-update-agent.timer >/dev/null 2>&1 \
|
|
|| ! systemctl is-enabled --quiet clupilot-update-agent.timer 2>/dev/null; then
|
|
agent_hint="the panel's update button stays disabled"
|
|
fi
|
|
|
|
# The proxy's console allowlist has to be wired once, as root, and an update
|
|
# never runs as root — so an existing installation would keep its hard-coded
|
|
# list and everything the owner changes in the console would do nothing.
|
|
# A host helper from before a step was added to it. Nothing is broken, but the
|
|
# new step silently does nothing — which is worse than an error, because the
|
|
# next person to look assumes it ran.
|
|
if [[ -z "$agent_hint" ]]; then
|
|
# `|| true` on the read, not on the whole test: an installation that has
|
|
# never had the helper reports 0 and lands in the same hint.
|
|
host_step_have="$( [[ -x "$HOST_STEP" ]] && "$HOST_STEP" contract 2>/dev/null || true )"
|
|
[[ "$host_step_have" =~ ^[0-9]+$ ]] || host_step_have=0
|
|
|
|
if (( host_step_have < HOST_STEP_NEEDS )); then
|
|
agent_hint="the update cannot install host packages such as rsync by itself"
|
|
fi
|
|
fi
|
|
|
|
if [[ -z "$agent_hint" ]] && command -v caddy >/dev/null 2>&1 \
|
|
&& [[ -f /etc/caddy/Caddyfile ]] \
|
|
&& ! grep -q 'clupilot-console-allow.conf' /etc/caddy/Caddyfile 2>/dev/null; then
|
|
agent_hint="the console's access list does not reach the reverse proxy"
|
|
fi
|
|
|
|
if [[ -n "$agent_hint" ]]; then
|
|
printf '\033[1;33m !\033[0m %s\n' "One-time setup missing — $agent_hint."
|
|
printf ' %s\n' "Run once: sudo bash $(pwd)/deploy/install-agent.sh"
|
|
fi
|
|
|
|
log "Done — $(release_version) on $(git rev-parse --short HEAD) (${source_ref})"
|