From 5301ce4309be5d9864aa885924620d2f651e7187 Mon Sep 17 00:00:00 2001 From: nexxo Date: Sat, 25 Jul 2026 20:04:58 +0200 Subject: [PATCH] feat(monitoring): Uptime Kuma support via a REST bridge + resilient monitoring policy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Kuma has no write REST API (monitor CRUD is Socket.IO only), so docker/kuma-bridge translates the exact generic REST contract HttpMonitoringClient already speaks — no PHP client change needed. Opt-in compose profile 'monitoring'. Verified end-to-end against a real Uptime Kuma 1.x: create -> id, create again -> same id (idempotent), list, status, delete and PHP MonitoringClient -> bridge -> Kuma round-trip. The e2e run caught a real bug: add_monitor takes 'maxretries', not 'retries'. Monitoring is now observability, not a delivery gate: - RegisterMonitoring retries MONITORING_ATTEMPTS times on an outage, then continues degraded with a visible 'info' event - RunAcceptanceChecks no longer fails when monitoring isn't green (a fresh Kuma monitor legitimately reports 'pending' — confirmed in the e2e run); it records the gap instead. MONITORING_REQUIRED=true restores strict gating. - docs/monitoring-uptime-kuma.md documents setup, failure behaviour, alternatives Co-Authored-By: Claude Opus 4.8 --- ..env.swp | Bin 1024 -> 1024 bytes .env.example | 28 ++- .../Steps/Customer/RegisterMonitoring.php | 28 ++- .../Steps/Customer/RunAcceptanceChecks.php | 23 +- .../Monitoring/FakeMonitoringClient.php | 7 + config/provisioning.php | 8 + docker-compose.yml | 17 ++ docker/kuma-bridge/Dockerfile | 24 +++ docker/kuma-bridge/app.py | 196 ++++++++++++++++++ docker/kuma-bridge/requirements.txt | 5 + docs/monitoring-uptime-kuma.md | 93 +++++++++ .../Provisioning/CustomerStepsTest.php | 37 +++- 12 files changed, 456 insertions(+), 10 deletions(-) create mode 100644 docker/kuma-bridge/Dockerfile create mode 100644 docker/kuma-bridge/app.py create mode 100644 docker/kuma-bridge/requirements.txt create mode 100644 docs/monitoring-uptime-kuma.md diff --git a/..env.swp b/..env.swp index b1c79c4012353a59d358c11227b7e09c9b15faf6..cae078b6acf015e60c48cd2c174df43019ffaddc 100644 GIT binary patch delta 31 ZcmZqRXyBM2!E}7;MyadJf}sp(004T}1t0(b delta 16 XcmZqRXyBM2!IV_LQR*u5L<2zpD*^=( diff --git a/.env.example b/.env.example index 9f7ab00..35a56b9 100644 --- a/.env.example +++ b/.env.example @@ -131,14 +131,32 @@ CLUPILOT_SSH_COMMAND_TIMEOUT=2000 # ── Monitoring ─────────────────────────────────────────────────────────── # The built-in client speaks a GENERIC REST API: # GET/POST /monitors, GET/DELETE /monitors/{id} -# Uptime Kuma's own REST API is READ-ONLY (badges/status page/push) — creating -# monitors goes through Socket.IO, so Kuma needs a small REST bridge in front -# of it (or a dedicated client). Leave blank to disable monitoring entirely: -# provisioning then records a stable breadcrumb and its own health checks -# (occ status, TLS, admin user) still gate acceptance. +# For Uptime Kuma use the bundled bridge below (Kuma's own REST API is +# read-only — monitor CRUD goes through Socket.IO): +# MONITORING_API_URL=http://kuma-bridge:8080 +# Leave blank to disable monitoring entirely: provisioning records a stable +# breadcrumb and its own health checks (occ status, TLS, admin user) still +# gate acceptance. MONITORING_API_URL= MONITORING_API_TOKEN= +# ── Uptime Kuma (über die mitgelieferte Bridge) ─────────────────────────── +# Kuma kann Monitore nur über Socket.IO anlegen, daher die Bridge: +# docker compose --profile monitoring up -d --build kuma-bridge +# Danach hier MONITORING_API_URL auf die Bridge zeigen lassen: +# MONITORING_API_URL=http://kuma-bridge:8080 +# MONITORING_API_TOKEN ist gleichzeitig das Bridge-Token (frei wählbar, lang). +KUMA_URL= +KUMA_USERNAME= +KUMA_PASSWORD= +KUMA_TOTP= + +# Monitoring-Verhalten bei Ausfall: +# MONITORING_REQUIRED=false -> Bereitstellung läuft weiter (Warn-Event), Standard +# MONITORING_REQUIRED=true -> Lauf schlägt fehl, wenn Monitoring nicht erreichbar +MONITORING_REQUIRED=false +MONITORING_ATTEMPTS=2 + # ── Nextcloud blueprint template ───────────────────────────────────────── # NOT an env var: set the Proxmox template VMID per plan in # config/provisioning.php → plans.*.template_vmid (default 9000). diff --git a/app/Provisioning/Steps/Customer/RegisterMonitoring.php b/app/Provisioning/Steps/Customer/RegisterMonitoring.php index 67f0ae3..b4b0a1a 100644 --- a/app/Provisioning/Steps/Customer/RegisterMonitoring.php +++ b/app/Provisioning/Steps/Customer/RegisterMonitoring.php @@ -30,7 +30,33 @@ class RegisterMonitoring extends CustomerStep $url = 'https://'.$instance->subdomain.'.'.config('provisioning.dns.zone').'/status.php'; // Register with the monitoring service [E] before recording it. - $targetId = $this->monitoring->registerTarget($instance->subdomain, $url); + // Monitoring is observability, not the product: a paid customer's cloud + // must not fail to be delivered because Kuma/the bridge is down. Retry a + // few times, then continue DEGRADED with a visible event — unless the + // operator declares monitoring mandatory. + try { + $targetId = $this->monitoring->registerTarget($instance->subdomain, $url); + } catch (\Throwable $e) { + if (config('provisioning.monitoring.required', false)) { + throw $e; // operator wants monitoring to gate delivery + } + + $allowed = (int) config('provisioning.monitoring.attempts', 2); + if ($run->attempt < $allowed) { + return StepResult::retry(30, 'monitoring unavailable: '.$e->getMessage()); + } + + // Give up on monitoring, keep the cloud. Recorded so it is visible in + // the admin console rather than silently missing. + $run->events()->create([ + 'step' => $this->key(), + 'attempt' => $run->attempt, + 'outcome' => 'info', + 'message' => 'Monitoring übersprungen (Dienst nicht erreichbar): '.$e->getMessage(), + ]); + + return StepResult::advance(); + } // Local row BEFORE the breadcrumb (the short-circuit guard). $instance->monitoringTargets()->firstOrCreate( diff --git a/app/Provisioning/Steps/Customer/RunAcceptanceChecks.php b/app/Provisioning/Steps/Customer/RunAcceptanceChecks.php index 58a3e17..e352821 100644 --- a/app/Provisioning/Steps/Customer/RunAcceptanceChecks.php +++ b/app/Provisioning/Steps/Customer/RunAcceptanceChecks.php @@ -65,10 +65,27 @@ class RunAcceptanceChecks extends CustomerStep return StepResult::fail('acceptance_failed:backup'); } - // Monitoring target exists and reports healthy. + // Monitoring is a SECONDARY signal: it depends on an external poller, and + // a freshly created monitor legitimately has no heartbeat yet. The four + // checks above already prove the service itself works, so only gate on + // monitoring when the operator declared it required — otherwise record it + // and deliver (consistent with RegisterMonitoring's degrade-on-outage). $target = $instance->monitoringTargets()->first(); - if ($target === null || ! $this->monitoring->isHealthy($target->external_id)) { - return StepResult::fail('acceptance_failed:monitoring'); + $monitoringGreen = $target !== null && $this->monitoring->isHealthy($target->external_id); + + if (! $monitoringGreen) { + if (config('provisioning.monitoring.required', false)) { + return StepResult::fail('acceptance_failed:monitoring'); + } + + $run->events()->create([ + 'step' => $this->key(), + 'attempt' => $run->attempt, + 'outcome' => 'info', + 'message' => $target === null + ? 'Abnahme ohne Monitoring: kein Monitor registriert.' + : 'Abnahme ohne Monitoring-Bestätigung: Monitor noch nicht grün.', + ]); } return StepResult::advance(); diff --git a/app/Services/Monitoring/FakeMonitoringClient.php b/app/Services/Monitoring/FakeMonitoringClient.php index f86651e..e86334b 100644 --- a/app/Services/Monitoring/FakeMonitoringClient.php +++ b/app/Services/Monitoring/FakeMonitoringClient.php @@ -7,8 +7,15 @@ class FakeMonitoringClient implements MonitoringClient /** @var array id => url */ public array $targets = []; + /** Set to simulate the monitoring service being unreachable. */ + public ?string $failWith = null; + public function registerTarget(string $name, string $url): string { + if ($this->failWith !== null) { + throw new \RuntimeException($this->failWith); + } + // Idempotent by URL — a retry after a crash returns the same target id. $existing = array_search($url, $this->targets, true); if ($existing !== false) { diff --git a/config/provisioning.php b/config/provisioning.php index 5adb90d..c5f9b48 100644 --- a/config/provisioning.php +++ b/config/provisioning.php @@ -108,6 +108,14 @@ return [ 'dynamic_path' => env('TRAEFIK_DYNAMIC_PATH', '/etc/traefik/dynamic'), ], + 'monitoring' => [ + // false: a monitoring outage does NOT block delivering a customer's cloud + // (the step retries, then continues degraded with a visible event). + // true: monitoring registration must succeed or the run fails. + 'required' => (bool) env('MONITORING_REQUIRED', false), + 'attempts' => (int) env('MONITORING_ATTEMPTS', 2), + ], + // CluPilot VM acts as the WireGuard hub; hosts join it during onboarding. 'wireguard' => [ 'subnet' => $wgSubnet, diff --git a/docker-compose.yml b/docker-compose.yml index a9c00f2..6412510 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -117,6 +117,23 @@ services: volumes: - redis-data:/data + # Uptime Kuma bridge: Kuma has no write REST API (monitor CRUD is Socket.IO), + # so this translates the REST contract CluPilot speaks. Opt-in: + # docker compose --profile monitoring up -d kuma-bridge + # Then set MONITORING_API_URL=http://kuma-bridge:8080 in .env. + kuma-bridge: + build: + context: ./docker/kuma-bridge + image: clupilot-kuma-bridge:dev + restart: unless-stopped + profiles: ["monitoring"] + environment: + KUMA_URL: "${KUMA_URL:-}" + KUMA_USERNAME: "${KUMA_USERNAME:-}" + KUMA_PASSWORD: "${KUMA_PASSWORD:-}" + KUMA_TOTP: "${KUMA_TOTP:-}" + BRIDGE_TOKEN: "${MONITORING_API_TOKEN:-}" + volumes: db-data: redis-data: diff --git a/docker/kuma-bridge/Dockerfile b/docker/kuma-bridge/Dockerfile new file mode 100644 index 0000000..18e1215 --- /dev/null +++ b/docker/kuma-bridge/Dockerfile @@ -0,0 +1,24 @@ +# CluPilot ⇄ Uptime Kuma bridge: exposes the generic REST contract CluPilot's +# MonitoringClient speaks and translates it to Kuma's Socket.IO API. +FROM python:3.12-slim + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 + +WORKDIR /app + +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY app.py . + +# Run unprivileged. +RUN useradd --create-home --uid 10001 bridge +USER bridge + +EXPOSE 8080 + +HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \ + CMD python -c "import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://127.0.0.1:8080/health', timeout=4).status==200 else 1)" + +CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8080"] diff --git a/docker/kuma-bridge/app.py b/docker/kuma-bridge/app.py new file mode 100644 index 0000000..72a699b --- /dev/null +++ b/docker/kuma-bridge/app.py @@ -0,0 +1,196 @@ +""" +CluPilot ⇄ Uptime Kuma bridge. + +Uptime Kuma has no official REST API for creating or deleting monitors — all +write operations go through its Socket.IO interface. This tiny service exposes +exactly the REST contract that CluPilot's HttpMonitoringClient already speaks, +and translates it to Socket.IO via the uptime-kuma-api library: + + GET /monitors -> {"monitors": [{"id", "url", "friendly_name"}]} + POST /monitors <- {"friendly_name", "url", "type"} + -> {"monitor": {"id": }} + GET /monitors/{id} -> {"monitor": {"id", "status": "up"|"down"|...}} + DELETE /monitors/{id} -> {"ok": true} + GET /health -> bridge + Kuma reachability (no auth) + +All monitor routes require `Authorization: Bearer $BRIDGE_TOKEN`. +Kuma's Socket.IO API is internal and may change between Kuma releases — pin the +Kuma version you tested against. +""" + +import os +import threading +from typing import Any + +from fastapi import Depends, FastAPI, Header, HTTPException, Response +from pydantic import BaseModel +from uptime_kuma_api import MonitorType, UptimeKumaApi, UptimeKumaException + +KUMA_URL = os.environ.get("KUMA_URL", "").rstrip("/") +KUMA_USERNAME = os.environ.get("KUMA_USERNAME", "") +KUMA_PASSWORD = os.environ.get("KUMA_PASSWORD", "") +KUMA_TOTP = os.environ.get("KUMA_TOTP", "") # optional, if 2FA is enabled +BRIDGE_TOKEN = os.environ.get("BRIDGE_TOKEN", "") + +app = FastAPI(title="CluPilot Uptime Kuma bridge", docs_url=None, redoc_url=None) + +# One shared, lazily-created Kuma session. Socket.IO connections are stateful, +# so guard it with a lock and re-establish it if a call fails. +_lock = threading.Lock() +_api: UptimeKumaApi | None = None + + +def _connect() -> UptimeKumaApi: + global _api + if not KUMA_URL: + raise HTTPException(status_code=503, detail="KUMA_URL is not configured") + + if _api is None: + api = UptimeKumaApi(KUMA_URL, timeout=15) + try: + if KUMA_TOTP: + api.login(KUMA_USERNAME, KUMA_PASSWORD, KUMA_TOTP) + else: + api.login(KUMA_USERNAME, KUMA_PASSWORD) + except Exception as exc: # noqa: BLE001 — surface any connect/login problem as 503 + try: + api.disconnect() + except Exception: # noqa: BLE001, S110 + pass + raise HTTPException(status_code=503, detail=f"Kuma unreachable: {exc}") from exc + _api = api + + return _api + + +def _reset() -> None: + """Drop the session so the next call reconnects (Kuma restart, token expiry…).""" + global _api + if _api is not None: + try: + _api.disconnect() + except Exception: # noqa: BLE001, S110 + pass + _api = None + + +def _call(fn, *args, **kwargs) -> Any: + """ + Run a Kuma call under the lock, retrying once on a stale connection. + Any Kuma-side failure becomes a 502/503 — never an unhandled 500, so the + caller always gets a meaningful status it can act on. + """ + with _lock: + try: + return fn(_connect(), *args, **kwargs) + except HTTPException: + raise + except Exception as first: # noqa: BLE001 — includes UptimeKumaException, OSError, socket errors + _reset() + try: + return fn(_connect(), *args, **kwargs) + except HTTPException: + raise + except Exception as exc: # noqa: BLE001 + raise HTTPException(status_code=502, detail=f"Kuma call failed: {exc}") from first + + +def require_token(authorization: str = Header(default="")) -> None: + if not BRIDGE_TOKEN: + raise HTTPException(status_code=503, detail="BRIDGE_TOKEN is not configured") + if authorization != f"Bearer {BRIDGE_TOKEN}": + raise HTTPException(status_code=401, detail="Invalid bridge token") + + +class MonitorIn(BaseModel): + url: str + friendly_name: str | None = None + type: str | None = "https" # CluPilot sends "https"; Kuma type is "http" + interval: int | None = 60 + retries: int | None = 1 # mapped to Kuma's `maxretries` + + +def _shape(monitor: dict) -> dict: + """Kuma monitor -> the field names CluPilot's client expects.""" + return { + "id": monitor.get("id"), + "url": monitor.get("url"), + "friendly_name": monitor.get("name"), + "active": monitor.get("active", True), + } + + +@app.get("/health") +def health() -> dict: + """Unauthenticated: is the bridge up, and can it reach Kuma?""" + if not KUMA_URL: + return {"bridge": "up", "kuma": "unconfigured"} + try: + _call(lambda api: api.get_monitors()) + return {"bridge": "up", "kuma": "up"} + except HTTPException as exc: + return {"bridge": "up", "kuma": "down", "detail": str(exc.detail)} + except Exception as exc: # noqa: BLE001 — health must never 500 + return {"bridge": "up", "kuma": "down", "detail": str(exc)} + + +@app.get("/monitors", dependencies=[Depends(require_token)]) +def list_monitors() -> dict: + monitors = _call(lambda api: api.get_monitors()) + return {"monitors": [_shape(m) for m in monitors]} + + +@app.post("/monitors", dependencies=[Depends(require_token)]) +def create_monitor(payload: MonitorIn) -> dict: + name = payload.friendly_name or payload.url + + # Idempotent: reuse an existing monitor for this exact URL. CluPilot also + # guards this, but a crash between its POST and its bookkeeping must not + # leave duplicates behind. + existing = next( + (m for m in _call(lambda api: api.get_monitors()) if m.get("url") == payload.url), + None, + ) + if existing is not None: + return {"monitor": {"id": existing.get("id")}} + + created = _call( + lambda api: api.add_monitor( + type=MonitorType.HTTP, + name=name, + url=payload.url, + interval=payload.interval or 60, + maxretries=payload.retries or 1, # Kuma's own field name + ) + ) + monitor_id = created.get("monitorID") if isinstance(created, dict) else None + if monitor_id is None: + raise HTTPException(status_code=502, detail=f"Kuma returned no monitor id: {created}") + + return {"monitor": {"id": monitor_id}} + + +@app.get("/monitors/{monitor_id}", dependencies=[Depends(require_token)]) +def get_monitor(monitor_id: int) -> dict: + monitor = _call(lambda api: api.get_monitor(monitor_id)) + + # Kuma heartbeat status: 0 down, 1 up, 2 pending, 3 maintenance. + status = "unknown" + try: + beats = _call(lambda api: api.get_monitor_beats(monitor_id, 1)) or [] + if beats: + status = {0: "down", 1: "up", 2: "pending", 3: "maintenance"}.get( + beats[-1].get("status"), "unknown" + ) + except HTTPException: + status = "unknown" # monitor exists but no heartbeat history yet + + shaped = _shape(monitor) + shaped["status"] = status + return {"monitor": shaped} + + +@app.delete("/monitors/{monitor_id}", dependencies=[Depends(require_token)]) +def delete_monitor(monitor_id: int) -> Response: + _call(lambda api: api.delete_monitor(monitor_id)) + return Response(status_code=200, content='{"ok":true}', media_type="application/json") diff --git a/docker/kuma-bridge/requirements.txt b/docker/kuma-bridge/requirements.txt new file mode 100644 index 0000000..438c1d6 --- /dev/null +++ b/docker/kuma-bridge/requirements.txt @@ -0,0 +1,5 @@ +# Ranges (not exact pins) so the image keeps building; uptime-kuma-api wraps +# Kuma's internal Socket.IO API, so re-smoke-test after a Kuma upgrade. +uptime-kuma-api>=1.2,<2 +fastapi>=0.110,<1 +uvicorn[standard]>=0.27,<1 diff --git a/docs/monitoring-uptime-kuma.md b/docs/monitoring-uptime-kuma.md new file mode 100644 index 0000000..df67748 --- /dev/null +++ b/docs/monitoring-uptime-kuma.md @@ -0,0 +1,93 @@ +# Monitoring mit Uptime Kuma + +CluPilot registriert für jede Kundeninstanz ein Uptime-Ziel +(`https:///status.php`) und fragt dessen Gesundheit während der +Abnahmeprüfung ab. + +## Warum eine Bridge? + +Uptime Kuma hat **keine schreibende REST-API**. Monitore anlegen/löschen läuft +ausschließlich über **Socket.IO** — offiziell nur für Kumas eigenes Frontend +gedacht. Statt Socket.IO in PHP nachzubauen, läuft ein kleiner Dienst mit: + +``` +CluPilot (PHP) ──REST──▶ kuma-bridge ──Socket.IO──▶ Uptime Kuma +``` + +Die Bridge implementiert **exakt** den Vertrag, den `HttpMonitoringClient` +ohnehin spricht — am PHP-Code musste nichts geändert werden: + +| Route | Zweck | +|---|---| +| `GET /monitors` | Liste (für Idempotenz: existiert die URL schon?) | +| `POST /monitors` | Monitor anlegen (`{friendly_name, url, type}`) | +| `GET /monitors/{id}` | Status lesen (`up` / `down` / `pending` / `maintenance`) | +| `DELETE /monitors/{id}` | Monitor entfernen (bei Kündigung) | +| `GET /health` | Bridge + Kuma-Erreichbarkeit (ohne Token) | + +Alle Monitor-Routen erfordern `Authorization: Bearer $BRIDGE_TOKEN`. + +## Einrichtung + +1. **In Uptime Kuma** einen eigenen Benutzer für CluPilot anlegen (nicht den + persönlichen Admin verwenden). + +2. **`.env` ausfüllen:** + + ```dotenv + KUMA_URL=http://kuma.dein-host.tld:3001 + KUMA_USERNAME=clupilot + KUMA_PASSWORD= + KUMA_TOTP= # nur falls 2FA aktiv + + MONITORING_API_URL=http://kuma-bridge:8080 + MONITORING_API_TOKEN= # = Bridge-Token + ``` + +3. **Bridge starten** (eigenes Compose-Profil, läuft nicht im normalen `up`): + + ```bash + docker compose --profile monitoring up -d --build kuma-bridge + ``` + +4. **Prüfen:** + + ```bash + docker compose exec app curl -s http://kuma-bridge:8080/health + # {"bridge":"up","kuma":"up"} + ``` + +5. Config-Cache leeren, falls gesetzt: `clupilot artisan config:clear` + +## Verhalten bei Ausfall + +Monitoring ist Beobachtung, nicht das Produkt. Fällt Kuma oder die Bridge aus, +darf die Bereitstellung eines **bezahlten** Kunden nicht scheitern: + +- Der Schritt `register_monitoring` versucht es `MONITORING_ATTEMPTS` mal (Standard 2), +- danach läuft die Bereitstellung **degradiert** weiter und schreibt ein + sichtbares `info`-Ereignis in die Admin-Konsole, +- mit `MONITORING_REQUIRED=true` wird Monitoring stattdessen erzwungen (der Lauf + schlägt fehl, wenn es nicht erreichbar ist). + +Bleibt `MONITORING_API_URL` leer, ist Monitoring komplett aus: die Bereitstellung +notiert eine stabile Kennung und die **eigenen** Gesundheitsprüfungen +(`occ status`, TLS-Zertifikat, Admin-User) entscheiden weiterhin über die Abnahme. + +## Wartung + +Die Bridge nutzt Kumas **interne** Socket.IO-API (über `uptime-kuma-api`). Diese +kann sich zwischen Kuma-Releases ändern: + +- Nach einem Kuma-Upgrade: `GET /health` und einen Testkunden prüfen. +- Kuma-Version notieren, gegen die getestet wurde. +- Bei Bruch ist der Ausfall dank obigem Verhalten **nicht** kundenwirksam — + es fehlen nur neue Monitore, sichtbar als `info`-Ereignis. + +## Alternativen (falls Kuma zu fragil wird) + +- **Gatus** — Config-as-Code: eine YAML-Datei pro Instanz in einem gemergten + Verzeichnis, Hot-Reload, eingebaute Statusseite. Passt zum Muster, mit dem + CluPilot schon Traefik-Routen schreibt. Keine API nötig. +- **Kuvasz** — laut Doku vollständige REST-API; der bestehende + `HttpMonitoringClient` wäre dann ohne Bridge nutzbar (evtl. Feld-Mapping). diff --git a/tests/Feature/Provisioning/CustomerStepsTest.php b/tests/Feature/Provisioning/CustomerStepsTest.php index f9c00ac..563e403 100644 --- a/tests/Feature/Provisioning/CustomerStepsTest.php +++ b/tests/Feature/Provisioning/CustomerStepsTest.php @@ -326,9 +326,15 @@ it('passes acceptance only when everything is genuinely green', function () { expect(app(RunAcceptanceChecks::class)->execute($run)->type)->toBe('advance'); - // monitoring unhealthy → fail + // Monitoring is a secondary signal: not green → still deliver, but record it. $s['monitoring']->healthy = false; + expect(app(RunAcceptanceChecks::class)->execute($run->fresh())->type)->toBe('advance') + ->and($run->events()->where('outcome', 'info')->where('step', 'run_acceptance_checks')->exists())->toBeTrue(); + + // ...unless the operator declared monitoring required. + config()->set('provisioning.monitoring.required', true); expect(app(RunAcceptanceChecks::class)->execute($run->fresh())->type)->toBe('fail'); + config()->set('provisioning.monitoring.required', false); $s['monitoring']->healthy = true; // Nextcloud not installed → fail @@ -358,3 +364,32 @@ it('completes: activates instance+order, seeds onboarding, delivers credentials' app(CompleteProvisioning::class)->execute($run->fresh()); Notification::assertSentOnDemandTimes(CloudReady::class, 1); }); + +// 13b. Monitoring resilience — observability must not block delivery. +it('keeps provisioning when monitoring is unreachable (retry, then degrade)', function () { + $s = fakeServices(); + $s['monitoring']->failWith = 'connection refused'; + ['run' => $run, 'instance' => $instance] = reservedRun(); + + // Early attempts retry — the outage may be transient. + $first = app(RegisterMonitoring::class)->execute($run->fresh()); + expect($first->type)->toBe('retry')->and($first->reason)->toContain('connection refused'); + + // Once the allowance is used up, continue DEGRADED with a visible event + // instead of failing a paid customer's provisioning. + $run->update(['attempt' => (int) config('provisioning.monitoring.attempts', 2)]); + expect(app(RegisterMonitoring::class)->execute($run->fresh())->type)->toBe('advance'); + + expect($instance->fresh()->monitoringTargets()->count())->toBe(0) + ->and($run->events()->where('outcome', 'info')->where('step', 'register_monitoring')->exists())->toBeTrue(); +}); + +it('fails the run on a monitoring outage when monitoring is declared required', function () { + config()->set('provisioning.monitoring.required', true); + $s = fakeServices(); + $s['monitoring']->failWith = 'connection refused'; + ['run' => $run] = reservedRun(); + + expect(fn () => app(RegisterMonitoring::class)->execute($run->fresh())) + ->toThrow(RuntimeException::class); +});