#!/usr/bin/env bash # # Installs the update agent's systemd timer. Root, once per machine, idempotent. # # Separate from install.sh because the installed base never runs install.sh # again — existing servers upgrade through deploy/update.sh, which runs as the # service account and cannot write to /etc/systemd. Without this as its own # entry point, every server that already exists would show the update button # permanently disabled. # # sudo bash deploy/install-agent.sh # set -euo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" APP_USER="${APP_USER:-clupilot}" [[ $EUID -eq 0 ]] || { echo "Please run as root: sudo bash deploy/install-agent.sh" >&2; exit 1; } id -u "$APP_USER" >/dev/null 2>&1 || { echo "No such account: $APP_USER" >&2; exit 1; } install -o "$APP_USER" -g "$APP_USER" -d "$ROOT/storage/app/deploy" chmod +x "$ROOT/deploy/update-agent.sh" cat > /etc/systemd/system/clupilot-update-agent.service < /etc/systemd/system/clupilot-update-agent.timer <<'EOF' [Unit] Description=Check for CluPilot updates and run requested ones [Timer] # Every minute: a `git fetch` against one remote is not a nuisance at this # rate, and the panel now shows a live countdown to the next tick — a wait # that reads as "obviously nothing is happening" at five minutes reads as # "obviously working" at one. OnBootSec=3min OnUnitActiveSec=1min # Tightened along with the interval above: at five minutes a 30s fuzz was a # rounding error; at one minute it is half the interval, and the panel's # countdown would sit at 0:00 for as long as systemd felt like waiting. AccuracySec=5s [Install] WantedBy=timers.target EOF # ── The proxy's console allowlist ──────────────────────────────────────────── # The proxy has its own allowlist and it runs BEFORE the application, so # everything the owner adds in the console has no effect until the proxy is # told. The agent regenerates this fragment from the console's own list. # # Entirely optional: an installation may sit behind nginx, Zoraxy, or nothing at # all, and none of that is a reason for the update agent itself to fail to # install. CADDYFILE="${CADDYFILE:-/etc/caddy/Caddyfile}" ALLOWFILE="/etc/caddy/clupilot-console-allow.conf" if command -v caddy >/dev/null 2>&1 && [[ -f "$CADDYFILE" ]]; then echo " Wiring the console allowlist into Caddy" # Seed it before anything imports it: an import of a missing file makes the # whole proxy config invalid, and that takes the public site down too. ALLOWLIST_READY=0 if [[ -s "$ALLOWFILE" ]] && grep -q '@allowed remote_ip .' "$ALLOWFILE"; then ALLOWLIST_READY=1 else # Compose discovers its own file from the working directory — naming it # here guessed wrong once. if ( cd "$ROOT" && runuser -u "$APP_USER" -- docker compose exec -T app \ php artisan clupilot:console-access caddy ) > "$ALLOWFILE.new" 2>/dev/null \ && grep -q '@allowed remote_ip .' "$ALLOWFILE.new"; then mv -f "$ALLOWFILE.new" "$ALLOWFILE" ALLOWLIST_READY=1 else # NO fallback matcher here. A loopback-only list looks like a valid # allowlist, passes validation, and locks out every remote operator # the moment Caddy reloads — because the application happened to be # unavailable for the ten seconds this ran. rm -f "$ALLOWFILE.new" echo " ! Could not read the console allowlist from the application." echo " Leaving the proxy configuration untouched." fi fi [[ -f "$ALLOWFILE" ]] && { chown "$APP_USER":root "$ALLOWFILE"; chmod 0644 "$ALLOWFILE"; } # Replace a hard-coded matcher with the import, once. Backed up and # validated first — a broken Caddyfile is an outage of everything, not just # of the console. if [[ "$ALLOWLIST_READY" == "1" ]] && ! grep -q "import $ALLOWFILE" "$CADDYFILE"; then # Exactly one, or none of them. An unqualified substitution would # rewrite a matcher belonging to some other site — and the result would # still validate, so the first sign of it would be that site's access # control quietly changing. matches="$(grep -cE '^[[:space:]]*@allowed remote_ip ' "$CADDYFILE" || true)" if [[ "$matches" == "1" ]]; then cp -a "$CADDYFILE" "$CADDYFILE.clupilot.bak" sed -i "s|^\([[:space:]]*\)@allowed remote_ip .*|\1import $ALLOWFILE|" "$CADDYFILE" if caddy validate --config "$CADDYFILE" --adapter caddyfile >/dev/null 2>&1; then if systemctl reload caddy >/dev/null 2>&1; then echo " Console allowlist is now managed from the console." else # Handed to the agent. Without this marker the file already # matches what the agent would generate, so nothing would # ever trigger another reload and the proxy would keep the # old hard-coded list indefinitely. : > "$ROOT/storage/app/deploy/.caddy-reload-pending" chown "$APP_USER":"$APP_USER" "$ROOT/storage/app/deploy/.caddy-reload-pending" echo " ! Caddy did not reload — the agent retries within a minute." fi else echo " ! Caddy rejected the change — restoring the previous config." mv -f "$CADDYFILE.clupilot.bak" "$CADDYFILE" fi else # Said out loud rather than assumed. Nothing to replace, or more # than one candidate and no way to tell which belongs to the # console — either way, carrying on silently would leave the # console's allowlist a decoration while the proxy kept its own. echo " ! Found $matches single-line '@allowed remote_ip' entries in $CADDYFILE (expected exactly 1)." echo " Put this inside the console's site block yourself, then reload Caddy:" echo " import $ALLOWFILE" fi fi # Narrow on purpose: one command, no arguments, nothing else. cat > /etc/sudoers.d/clupilot-caddy-reload </dev/null || rm -f /etc/sudoers.d/clupilot-caddy-reload fi # ── The console inside the tunnel ──────────────────────────────────────────── # The gateway serves the console on the WireGuard hub address using the # certificate the public Caddy already obtains for that same hostname — a # certificate is bound to the NAME, not to the address serving it. Finding it # has to happen here, as root: the storage is Caddy's and the paths contain the # ACME directory, which is not something to ask an operator to type. VPN_HOST="$(sed -n 's/^VPN_INTERNAL_HOST=//p' "$ROOT/.env" 2>/dev/null | tail -1)" if [[ -z "$VPN_HOST" ]]; then # Cleared. Reconciled here too, because the updater exits early when the # checkout has not moved — so without this there may be no path at all that # stops the services, and clients would keep being handed a resolver that # is no longer meant to exist. runuser -u "$APP_USER" -- sh -c " cd '$ROOT' if grep -qE '^COMPOSE_PROFILES=.*vpn' .env 2>/dev/null || grep -qE '^VPN_READY=' .env 2>/dev/null; then docker compose --profile vpn stop vpn-dns vpn-gateway >/dev/null 2>&1 || true docker compose --profile vpn rm -f vpn-dns vpn-gateway >/dev/null 2>&1 || true sed -i '/^VPN_CERT_PATH=/d;/^VPN_KEY_PATH=/d;/^VPN_READY=/d' .env profiles=\"\$(sed -n 's/^COMPOSE_PROFILES=//p' .env | tail -1)\" profiles=\"\$(printf '%s' \"\$profiles\" | tr ',' '\n' | grep -vx 'vpn' | grep -v '^\$' | paste -sd, - || true)\" sed -i '/^COMPOSE_PROFILES=/d' .env [ -n \"\$profiles\" ] && printf 'COMPOSE_PROFILES=%s\n' \"\$profiles\" >> .env docker compose exec -T app php artisan config:clear >/dev/null 2>&1 || true echo ' Tunnel gateway disabled — VPN_INTERNAL_HOST is empty.' fi " || true fi if [[ -n "$VPN_HOST" ]]; then # From the project's .env, because that is the file Compose reads for the # mount. Looking only at the default would report "no certificate" on an # installation whose storage is somewhere else. CADDY_DATA="$(sed -n 's/^CADDY_DATA_DIR=//p' "$ROOT/.env" 2>/dev/null | tail -1)" CADDY_DATA="${CADDY_DATA:-${CADDY_DATA_DIR:-/var/lib/caddy/.local/share/caddy}}" # -print -quit, not | head -1: under `set -o pipefail` a find that is still # producing output when head exits returns SIGPIPE, and the whole agent # installation aborts — on the perfectly ordinary case of one certificate # existing under two issuer directories. cert="$(find "$CADDY_DATA/certificates" -name "${VPN_HOST}.crt" -print -quit 2>/dev/null)" key="${cert%.crt}.key" if [[ -n "$cert" && -f "$key" ]]; then # Paths as the gateway container sees them: the storage is mounted at /certs. rel="${cert#$CADDY_DATA/}" runuser -u "$APP_USER" -- sh -c " cd '$ROOT' sed -i '/^VPN_CERT_PATH=/d;/^VPN_KEY_PATH=/d' .env printf 'VPN_CERT_PATH=/certs/%s\n' '$rel' >> .env printf 'VPN_KEY_PATH=/certs/%s\n' '${rel%.crt}.key' >> .env " # No chmod. The gateway container runs as root and reads these through # the read-only mount already; making the private key world-readable # would let every local account copy the console's TLS identity. # Enabled AND started here. The updater exits early when the checkout # has not moved, so leaving this to "the next deploy" leaves the # gateway stopped and the success message a lie. runuser -u "$APP_USER" -- sh -c " cd '$ROOT' grep -qE '^COMPOSE_PROFILES=.*vpn' .env \ || { grep -qE '^COMPOSE_PROFILES=' .env \ && sed -i 's/^COMPOSE_PROFILES=\\(.*\\)\$/COMPOSE_PROFILES=\\1,vpn/' .env \ || printf 'COMPOSE_PROFILES=vpn\n' >> .env; } docker compose --profile vpn up -d vpn-dns vpn-gateway " >/dev/null 2>&1 || true # `up -d` succeeds once the containers are CREATED. One that starts and # exits — Caddy unable to read the certificate, or to bind the hub # address — still counts as success there, so the state is checked. sleep 3 vpn_started=0 if runuser -u "$APP_USER" -- sh -c "cd '$ROOT' && docker compose --profile vpn ps --status running --services" 2>/dev/null \ | grep -q '^vpn-gateway$' \ && runuser -u "$APP_USER" -- sh -c "cd '$ROOT' && docker compose --profile vpn ps --status running --services" 2>/dev/null \ | grep -q '^vpn-dns$'; then vpn_started=1 fi # The marker, and only on success. Certificate paths alone would report # the tunnel ready after a failed start, and the application would hand # out configs naming a resolver that never came up. runuser -u "$APP_USER" -- sh -c " cd '$ROOT' sed -i '/^VPN_READY=/d' .env printf 'VPN_READY=%s\n' '$([[ $vpn_started -eq 1 ]] && echo true || echo false)' >> .env " if [[ $vpn_started -ne 1 ]]; then echo " ! Could not start the tunnel gateway — check: docker compose --profile vpn up -d" fi # The application caches its configuration. Without rebuilding it here, # vpn_ready stays false and client configs keep omitting the tunnel # resolver — the gateway running but nothing pointed at it. runuser -u "$APP_USER" -- sh -c " cd '$ROOT' docker compose exec -T app php artisan config:clear >/dev/null 2>&1 docker compose exec -T app php artisan config:cache >/dev/null 2>&1 " || true # Only when it is true. Announcing success while the gateway failed to # start is worse than silence: it sends the operator looking somewhere # else entirely. if [[ $vpn_started -eq 1 ]]; then echo " Console reachable inside the tunnel at https://$VPN_HOST" fi else # Nothing usable for the CURRENT hostname. Leaving the old paths, the # profile and the readiness marker in place would have the application # advertise the renamed endpoint as ready while the gateway still # served the previous name. runuser -u "$APP_USER" -- sh -c " cd '$ROOT' sed -i '/^VPN_CERT_PATH=/d;/^VPN_KEY_PATH=/d;/^VPN_READY=/d' .env docker compose --profile vpn stop vpn-dns vpn-gateway >/dev/null 2>&1 || true # And the profile with it, or the next ordinary `up -d` starts a # Caddy with empty tls paths that can only crash. profiles=\"\$(sed -n 's/^COMPOSE_PROFILES=//p' .env | tail -1)\" profiles=\"\$(printf '%s' \"\$profiles\" | tr ',' '\n' | grep -vx 'vpn' | grep -v '^\$' | paste -sd, - || true)\" sed -i '/^COMPOSE_PROFILES=/d' .env [ -n \"\$profiles\" ] && printf 'COMPOSE_PROFILES=%s\n' \"\$profiles\" >> .env docker compose exec -T app php artisan config:clear >/dev/null 2>&1 || true " || true echo " ! No certificate for $VPN_HOST in $CADDY_DATA yet." echo " Reach https://$VPN_HOST publicly once so Caddy obtains it, then run this again." fi fi systemctl daemon-reload systemctl enable --now clupilot-update-agent.timer >/dev/null # Run it once now, so the panel has a status to show instead of "never reported # in" for the first minute. systemctl start clupilot-update-agent.service || true echo "Update agent installed and running (clupilot-update-agent.timer)."