From ec08d361df1e89b73e609b0c7c2e3690f29dabb1 Mon Sep 17 00:00:00 2001 From: Matthias Linhuber Date: Fri, 28 Aug 2026 18:53:54 +0200 Subject: [PATCH 1/3] feat: configure alerting per cluster and document what is monitored Plumbs the eduide-cluster chart's new alerting through to the clusters. spec.alerting names the channels but never the webhook URLs: those are credentials and come from ALERT_WEBHOOK_SLACK and ALERT_WEBHOOK_DISCORD on the cluster's GitHub Environment, written by bootstrap into a values file rather than a --set, which would put them in the process list and in Actions debug logs. A channel's secretKey prefix decides which secret is read, so it has to match the channel type; the schema and test-deploy-logic.sh both enforce that. Alerting on with no channels, or with a channel whose webhook was never supplied, fails rather than firing into nowhere - alerting that notifies nobody is worse than none, because it reads as covered. spec.monitorCertManager opts a cluster into scraping cert-manager. Nothing watches certificate expiry today, including the webview wildcard that is renewed by hand once a year and takes every preview on the cluster with it when it lapses. Also drops monitoring.sessionNamespaces, whose PodMonitor is gone. docs/monitoring-setup.md is rewritten. It previously described session pods as exporting metrics, which is the belief that produced a PodMonitor that scraped them for a year and collected nothing. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_019qeiQRFu8xAMRYWPdZewjG --- .github/workflows/bootstrap-cluster.yml | 85 ++++++++- AGENTS.md | 32 +++- clusters/eduide.yaml | 19 ++ clusters/tum-production.yaml | 19 ++ clusters/tum-student.yaml | 19 ++ docs/monitoring-setup.md | 235 +++++++++++++++++------- schemas/cluster.schema.json | 83 +++++++++ scripts/test-deploy-logic.sh | 50 +++++ 8 files changed, 472 insertions(+), 70 deletions(-) diff --git a/.github/workflows/bootstrap-cluster.yml b/.github/workflows/bootstrap-cluster.yml index 02959e4..afbbf67 100644 --- a/.github/workflows/bootstrap-cluster.yml +++ b/.github/workflows/bootstrap-cluster.yml @@ -279,9 +279,32 @@ jobs: echo " enabled: true" echo " targetNamespaces:" sed 's/^/ - /' namespaces.txt - echo " sessionNamespaces:" - sed 's/^/ - /' namespaces.txt } >> listeners.yaml + # cert-manager exports certificate expiry but ships no + # ServiceMonitor, so by default nothing watches it. Opt in per + # cluster: the webview wildcard is renewed by hand once a year and + # has never had anything warning about it. + if [[ "$(yq -r '.spec.monitorCertManager // false' "clusters/${CLUSTER}.yaml")" == "true" ]]; then + printf ' certManager:\n enabled: true\n' >> listeners.yaml + fi + # Alerting. The channel list is in the manifest; the webhook URLs + # are not - they are credentials and come from the environment's + # secrets, written to a separate file below. + if [[ "$(yq -r '.spec.alerting.enabled // false' "clusters/${CLUSTER}.yaml")" == "true" ]]; then + { + echo " alerting:" + echo " enabled: true" + echo " minSeverity: $(yq -r '.spec.alerting.minSeverity // "warning"' "clusters/${CLUSTER}.yaml")" + if [[ "$(yq -r '.spec.alerting.grafanaUrl // ""' "clusters/${CLUSTER}.yaml")" != "" ]]; then + echo " grafanaUrl: $(yq -r '.spec.alerting.grafanaUrl' "clusters/${CLUSTER}.yaml")" + fi + # Passed through as YAML rather than rebuilt field by field: the + # manifest's channel keys are exactly the chart's, so there is + # nothing to translate and nothing to forget when a key is added. + echo " channels:" + yq -r '.spec.alerting.channels' "clusters/${CLUSTER}.yaml" | sed 's/^/ /' + } >> listeners.yaml + fi else echo "::warning::no environment on ${CLUSTER} opts into monitoring" printf 'monitoring:\n enabled: false\n' >> listeners.yaml @@ -360,12 +383,66 @@ jobs: key: "$(printf '%s' '${{ secrets.THEIA_WILDCARD_CERTIFICATE_KEY }}' | base64 | tr -d '\n')" EOF + - name: Collect the alert webhook URLs + env: + ALERT_WEBHOOK_SLACK: ${{ secrets.ALERT_WEBHOOK_SLACK }} + ALERT_WEBHOOK_DISCORD: ${{ secrets.ALERT_WEBHOOK_DISCORD }} + run: | + set -euo pipefail + # A Slack or Discord webhook URL is a credential: anyone holding it can + # post into the channel. It goes in a values file read from a secret, + # never on a `--set`, which would put it in the process list and in + # Actions debug logs. + # + # Always written, even when alerting is off, so the later helm calls + # can name the file unconditionally. + install -m 600 /dev/null alert-secrets.yaml + if [[ "$(yq -r '.spec.alerting.enabled // false' "clusters/${CLUSTER}.yaml")" != "true" ]]; then + echo "{}" > alert-secrets.yaml + echo "alerting is off for ${CLUSTER}" + exit 0 + fi + # Every channel names the key it reads. A channel whose secret is not + # set would render an AlertmanagerConfig that notifies nobody and + # reports no error, so fail here instead. + MISSING=() + { + echo "monitoring:" + echo " alerting:" + echo " webhookSecret:" + echo " create: true" + echo " data:" + } > alert-secrets.yaml + while read -r key; do + [[ -n "$key" ]] || continue + case "$key" in + slack-*) value="$ALERT_WEBHOOK_SLACK" ;; + discord-*) value="$ALERT_WEBHOOK_DISCORD" ;; + *) + echo "::error::channel secretKey '${key}' must start with 'slack-' or 'discord-'" + exit 1 + ;; + esac + if [[ -z "$value" ]]; then + MISSING+=("$key") + continue + fi + echo " ${key}: \"$(printf '%s' "$value" | base64 | tr -d '\n')\"" >> alert-secrets.yaml + done < <(yq -r '.spec.alerting.channels[]?.secretKey' "clusters/${CLUSTER}.yaml") + if (( ${#MISSING[@]} > 0 )); then + echo "::error::alerting is enabled on ${CLUSTER} but no webhook URL was supplied for: ${MISSING[*]}" + echo "::error::set ALERT_WEBHOOK_SLACK and/or ALERT_WEBHOOK_DISCORD on the '${{ needs.resolve.outputs.environment }}' environment." + echo "::error::See docs/monitoring.md." + exit 1 + fi + echo "collected $(yq -r '.spec.alerting.channels | length' "clusters/${CLUSTER}.yaml") webhook(s)" + - name: Preview run: | set -euo pipefail helm diff upgrade eduide-cluster oci://ghcr.io/eduide/charts/eduide-cluster \ --version "${{ inputs.chart_version }}" \ - -n eduide-system -f listeners.yaml -f gw-secrets.yaml \ + -n eduide-system -f listeners.yaml -f gw-secrets.yaml -f alert-secrets.yaml \ --allow-unreleased --no-color > gw.diff 2>&1 || true { echo "
Cluster chart pending change" @@ -379,7 +456,7 @@ jobs: helm upgrade --install eduide-cluster oci://ghcr.io/eduide/charts/eduide-cluster \ --version "${{ inputs.chart_version }}" \ -n eduide-system --create-namespace \ - -f listeners.yaml -f gw-secrets.yaml --wait --timeout 10m + -f listeners.yaml -f gw-secrets.yaml -f alert-secrets.yaml --wait --timeout 10m - name: Stamp this cluster with its name if: ${{ !inputs.dry_run }} diff --git a/AGENTS.md b/AGENTS.md index 6d4b3f3..904fe7d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -76,12 +76,38 @@ both, with no clue why. An installation with no identity provider yet sets `keycloak.allowUnauthenticated: true` and says so; Bonn does. **Monitoring is on by default and opted out per environment** with -`monitoring.enabled: false`. The PodMonitors themselves are in the cluster -chart - they must be created in Rancher's namespace to be discovered, and one +`monitoring.enabled: false`. The PodMonitor itself is in the cluster +chart - it must be created in Rancher's namespace to be discovered, and one per tenant would collide on names - so the flag decides whether the -environment's namespace is in the list they watch. Do not confuse it with +environment's namespace is in the list it watches. Do not confuse it with `monitor.enable`, the operator's session activity tracker. +**Session pods export no metrics, and nothing should try to scrape them.** They +are Theia IDEs serving HTML; a PodMonitor pointed at them collects nothing and +reports every target as down. There was one, for a year, and it never produced +a sample. Everything the session dashboards show comes from cAdvisor, kubelet +and kube-state-metrics, which need no PodMonitor at all. The REST service is +the only EduIDE component that is scraped, and it exports exactly one custom +metric - session startup latency, registered lazily on the first session, so it +does not exist on a freshly restarted service. + +**Alert `namespace` labels are a routing artifact; silence on +`eduide_namespace`.** Every EduIDE alert claims `namespace: eduide-system` +whatever environment it concerns, because the Prometheus Operator prepends a +`namespace = ` matcher to every route it +generates and the alert would otherwise reach no receiver. The real environment +is in `eduide_namespace`, and grouping and inhibition must use that one too. +Webhook URLs never appear in a manifest: `spec.alerting.channels` names a +`secretKey`, whose `slack-`/`discord-` prefix tells bootstrap which GitHub +Environment secret to read. See `docs/monitoring-setup.md`. + +**A selector that matches nothing looks exactly like a healthy platform.** Four +dashboard panels selected on `service=~"theia-.*"`, a label that stopped +existing in 2.0.0, and rendered empty for months; the namespace picker offered +`theia` and `test1` long after both namespaces were gone. Neither could fail a +render test. Check new expressions against a live Prometheus before shipping +them, not just `helm template`. + **Keycloak is per environment, including `authUrl`.** TUM installations share a server and differ only by realm; Bonn and Mannheim bring their own. Nothing about the identity provider belongs in `_base.yaml`, and secrets never go in a diff --git a/clusters/eduide.yaml b/clusters/eduide.yaml index 20942d3..24ca3e1 100644 --- a/clusters/eduide.yaml +++ b/clusters/eduide.yaml @@ -89,3 +89,22 @@ spec: runner: ubuntu-latest bootstrapEnvironment: cluster-eduide + + # Scrape cert-manager so certificates can be alerted on before they lapse. + # cert-manager exports expiry but ships no ServiceMonitor, so nothing watches + # it by default. The webview wildcard is renewed by hand once a year. + monitorCertManager: true + + # Where alerts go. The channel list is here; the webhook URLs are not - they + # are credentials and live in the cluster's GitHub Environment as + # ALERT_WEBHOOK_SLACK and ALERT_WEBHOOK_DISCORD. A channel's `secretKey` must + # start with `slack-` or `discord-`, which is how the workflow knows which + # secret to read. + # + # `minSeverity` is what reaches the channels, not what fires. Everything below + # it still fires and is visible in Alertmanager and on the dashboards; it just + # does not page anyone. See docs/monitoring.md. + alerting: + enabled: false + minSeverity: warning + channels: [] diff --git a/clusters/tum-production.yaml b/clusters/tum-production.yaml index e4f7da9..34b6d4e 100644 --- a/clusters/tum-production.yaml +++ b/clusters/tum-production.yaml @@ -74,3 +74,22 @@ spec: runner: ubuntu-latest bootstrapEnvironment: cluster-tum-production + + # Scrape cert-manager so certificates can be alerted on before they lapse. + # cert-manager exports expiry but ships no ServiceMonitor, so nothing watches + # it by default. The webview wildcard is renewed by hand once a year. + monitorCertManager: true + + # Where alerts go. The channel list is here; the webhook URLs are not - they + # are credentials and live in the cluster's GitHub Environment as + # ALERT_WEBHOOK_SLACK and ALERT_WEBHOOK_DISCORD. A channel's `secretKey` must + # start with `slack-` or `discord-`, which is how the workflow knows which + # secret to read. + # + # `minSeverity` is what reaches the channels, not what fires. Everything below + # it still fires and is visible in Alertmanager and on the dashboards; it just + # does not page anyone. See docs/monitoring.md. + alerting: + enabled: false + minSeverity: warning + channels: [] diff --git a/clusters/tum-student.yaml b/clusters/tum-student.yaml index a12a56e..cf6fced 100644 --- a/clusters/tum-student.yaml +++ b/clusters/tum-student.yaml @@ -40,3 +40,22 @@ spec: runner: ubuntu-latest bootstrapEnvironment: cluster-tum-student + + # Scrape cert-manager so certificates can be alerted on before they lapse. + # cert-manager exports expiry but ships no ServiceMonitor, so nothing watches + # it by default. The webview wildcard is renewed by hand once a year. + monitorCertManager: true + + # Where alerts go. The channel list is here; the webhook URLs are not - they + # are credentials and live in the cluster's GitHub Environment as + # ALERT_WEBHOOK_SLACK and ALERT_WEBHOOK_DISCORD. A channel's `secretKey` must + # start with `slack-` or `discord-`, which is how the workflow knows which + # secret to read. + # + # `minSeverity` is what reaches the channels, not what fires. Everything below + # it still fires and is visible in Alertmanager and on the dashboards; it just + # does not page anyone. See docs/monitoring.md. + alerting: + enabled: false + minSeverity: warning + channels: [] diff --git a/docs/monitoring-setup.md b/docs/monitoring-setup.md index 59fec33..3d29d5d 100644 --- a/docs/monitoring-setup.md +++ b/docs/monitoring-setup.md @@ -1,72 +1,181 @@ -# Monitoring Setup +# Monitoring and alerting -This guide explains how to set up monitoring and observability for Theia Cloud deployments using Prometheus and Grafana. +What EduIDE measures, what it alerts on, and how to point those alerts at a +chat channel. -## Overview +## What is installed, and by whom -Monitoring is essential for understanding system health, resource usage, and performance. This setup is based on the [Theia Cloud Observability](https://github.com/eclipsesource/theia-cloud-observability) project and includes: +| | Who installs it | +|---|---| +| Prometheus, Alertmanager, Grafana | **not this repository**. On the TUM clusters this is Rancher's monitoring stack, installed out of band | +| PodMonitor for the REST service | `Bootstrap cluster`, from the `eduide-cluster` chart | +| ServiceMonitor for cert-manager | same, when the cluster sets `spec.monitorCertManager` | +| Four Grafana dashboards | same | +| PrometheusRule and AlertmanagerConfig | same, when the cluster sets `spec.alerting.enabled` | -- **Prometheus**: Metrics collection and storage -- **Grafana**: Visualization dashboards -- **Theia-specific dashboards**: Custom dashboards for Theia Cloud metrics -- **Kubernetes metrics**: Cluster and pod-level monitoring +The values used for the manual kube-prometheus-stack install are preserved at +[`reference/kube-prometheus-stack-values.yaml`](reference/kube-prometheus-stack-values.yaml) +as the only record of it. Nothing applies them and they may have drifted. -## Architecture +## Where the numbers come from +**Session pods export no metrics.** This is the single most important thing on +this page, because assuming otherwise produced a PodMonitor that scraped every +session pod for a year and never collected one sample: session pods are Theia +IDEs, they serve HTML on that port, and Prometheus rejected every response with +`unsupported Content-Type "text/html"`. That PodMonitor is gone. Everything +about sessions is observed from outside them: + +| Source | What it gives us | +|---|---| +| cAdvisor | per-container CPU, memory, throttling, network | +| kubelet | workspace volume bytes and inodes | +| kube-state-metrics | pod phase, restarts, termination reason and exit code, deployment availability, PVC phase | +| the REST service, at `/q/metrics` | session startup latency, JVM health | +| cert-manager | certificate expiry and readiness | + +Only one custom metric exists: `application_application_theiacloud_session_startup_seconds_*`. +It is **registered lazily**, on the first session the service serves. After a +restart it does not exist until somebody starts a session, so an empty startup +panel on a freshly deployed environment is expected rather than broken. + +## The dashboards + +| Dashboard | uid | For | +|---|---|---| +| EduIDE Sessions | `eduide-sessions` | who is using it, how well it serves them, what it costs | +| EduIDE Session Detail | `eduide-session-detail` | one session: is it about to be OOM-killed, why did it stop | +| Theia Cloud | `bdrjgy1fv3d34b` | the original overview | +| Session Startup Time | `bf4ha4miogutcc` | startup latency quantiles | + +The environment picker on every dashboard is a query over the namespaces the +cluster actually monitors, derived from the environments. It used to be a +hand-written list, which went stale the moment the environments were renamed: +it still offered `theia`, `theia-staging` and `test1`, none of which exist, so +every panel was empty whatever you picked. + +## The alerts + +Off by default. Switch on per cluster: + +```yaml +spec: + alerting: + enabled: true + minSeverity: warning + grafanaUrl: https://rancher.example.tum.de/.../grafana + channels: + - name: platform-slack + type: slack + secretKey: slack-platform + channel: "#eduide-alerts" +``` + +**`minSeverity` is what reaches the channels, not what fires.** Everything +fires and is visible in Alertmanager and on the dashboards; the channels get a +filtered subset. This is deliberate and it is the whole reason the split +exists: a channel that receives every capacity blip stops being read, and then +the platform is unmonitored no matter how many rules are defined. + +**critical** - a person should look now + +| Alert | Meaning | +|---|---| +| `EduIDEComponentDown` | operator, REST service, landing page or garbage collector has no available replica. The description says what each one breaks | +| `EduIDEConversionWebhookDown` | cluster-wide: no CRD read or write succeeds while it is down | +| `EduIDEComponentCrashLooping` | a platform container is restarting repeatedly | +| `EduIDEServiceScrapeDown` | the REST service is unscrapeable. Students may be fine; we are blind | +| `EduIDEWarmPoolEmpty` | every pre-warmed instance is gone, so every student now waits for a cold start | + +**warning** - worth knowing today + +| Alert | Meaning | +|---|---| +| `EduIDESessionOOMKillSpike` | sessions being killed for memory above a rate threshold | +| `EduIDESessionCrashSpike` | sessions exiting with errors above a rate threshold | +| `EduIDESessionStartupSlow` | p95 startup over the threshold, while sessions are actually starting | +| `EduIDESessionPodPending` | a session cannot be scheduled | +| `EduIDESessionImagePullFailing` | a tag that does not exist, usually a blanket image override | +| `EduIDESessionEvicted` | node ran out of ephemeral disk, typically build caches | +| `EduIDEWorkspaceVolumeFilling` | a workspace is over 85% full by bytes | +| `EduIDEWorkspaceInodesFilling` | over 85% full by inodes, which happens long before bytes do | +| `EduIDEPVCPending` | a workspace volume will not provision | +| `EduIDECertificateExpiringSoon` | under 21 days left and not renewed | +| `EduIDECertificateNotReady` | a certificate has been failing to issue for an hour | + +Session-level faults alert on **rates across an environment, never per pod**. +This platform runs student code: sessions get OOM-killed and crash during +normal exercise work, and one notification per event is how a channel becomes +noise. Exit codes 137 and 143 are excluded outright, because those are the +garbage collector stopping an idle session. + +Every threshold is a chart value. Tune one in `values.yaml` under +`monitoring.alerting.thresholds` rather than waiting on a rule change. + +## Silences: match on `eduide_namespace`, not `namespace` + +Every EduIDE alert carries `namespace: eduide-system` regardless of which +environment it is about. **That label is a routing artifact.** The Prometheus +Operator defaults `alertmanagerConfigMatcherStrategy` to `OnNamespace`, which +prepends `namespace = ` to every route +it generates, so an alert has to claim that namespace to reach any receiver. + +The environment an alert is actually about is in **`eduide_namespace`**. + +```bash +# silence maintenance on one environment +amtool silence add eduide_namespace=eduide-test1 --duration=2h --comment="upgrade" ``` -┌─────────────────────────────────────────────────────────────┐ -│ Theia Cloud Deployment │ -│ │ -│ ┌─────────────────┐ ┌─────────────────┐ │ -│ │ Theia Pods │ │ Operator Pods │ │ -│ │ (Metrics) │ │ (Metrics) │ │ -│ └────────┬────────┘ └────────┬────────┘ │ -│ │ │ │ -└───────────┼────────────────────┼────────────────────────────┘ - │ │ - │ (scrape) │ - ▼ ▼ - ┌─────────────────────────────────────┐ - │ Prometheus Server │ - │ (Collects and Stores Metrics) │ - └──────────────┬──────────────────────┘ - │ - │ (query) - ▼ - ┌─────────────────────────────────────┐ - │ Grafana │ - │ (Visualizes Metrics in Dashboards) │ - └─────────────────────────────────────┘ + +Silencing `namespace=eduide-test1` matches nothing. Silencing +`namespace=eduide-system` silences every EduIDE alert on the cluster, which is +almost certainly not what was meant. The same applies to any inhibition rule: +compare `eduide_namespace`, or one environment's outage will suppress warnings +in all the others. + +## Adding a channel + +Webhook URLs are credentials. They never go in a manifest, a values file or a +`--set`, which would put them in the process list and in Actions debug logs. + +1. Create the incoming webhook in Slack or Discord. +2. Put it on the **cluster** GitHub Environment: + + ```bash + REPO=EduIDE/EduIDE-deployment + gh secret set ALERT_WEBHOOK_SLACK --repo "$REPO" --env cluster-tum-student + gh secret set ALERT_WEBHOOK_DISCORD --repo "$REPO" --env cluster-tum-student + ``` + +3. Add the channel to `clusters/.yaml`. The `secretKey` prefix decides + which secret is read, so it must be `slack-*` for a Slack channel and + `discord-*` for a Discord one. `test-deploy-logic.sh` and the cluster schema + both check this. +4. Re-run `Bootstrap cluster`. + +Alerting enabled with no channels fails the render rather than firing into +nowhere, and a channel whose webhook secret is unset fails the bootstrap with a +message naming the key. + +Discord is a first-class receiver here, not a Slack-compatible shim: +Alertmanager 0.25 and later support `discord_configs` natively, and the +`AlertmanagerConfig` CRD on these clusters exposes it. + +## Checking it works + +```bash +# the rules are loaded +kubectl -n eduide-system get prometheusrule eduide-alerts + +# Alertmanager picked up the receivers +kubectl -n eduide-system get alertmanagerconfig eduide-alerts -o yaml + +# nothing is silently failing to scrape +kubectl -n cattle-monitoring-system port-forward svc/rancher-monitoring-prometheus 9090:9090 +# then: http://127.0.0.1:9090/targets, filter for theia-cloud ``` -## Prerequisites - -- Kubernetes cluster with Theia Cloud deployed -- kubectl configured for your cluster -- Helm 3.x installed -- Admin access to the cluster - -## Step 1: TBA - -> **Note:** this guide is unfinished. No workflow in this repository installs -> kube-prometheus-stack, so the Prometheus/Grafana stack was installed manually -> and out-of-band. The values used are preserved at -> [`reference/kube-prometheus-stack-values.yaml`](reference/kube-prometheus-stack-values.yaml) -> as the only surviving record of that install. They are not applied by any -> automation and may have drifted from what is running. -> -> Only the `theia-monitoring` chart (PodMonitors and Grafana dashboard -> ConfigMaps) is part of the `eduide-cluster` chart and is installed by -> `.github/workflows/bootstrap-cluster.yml`, once per cluster. - -## The watched namespaces are derived - -The two PodMonitors name every namespace they scrape. That list used to be -written by hand in the chart's values and had gone stale — it still named -`theia` and `theia-staging`, so some environments were scraped and others were -not, silently. - -`Bootstrap cluster` now derives it from the environments that claim the -cluster, the same way it derives the Gateway's listeners. Adding an environment -picks up monitoring with no second edit. A PodMonitor rendered with an empty -namespace list fails the template rather than watching nothing. +To prove the whole path end to end, scale the garbage collector to zero and +wait five minutes. `EduIDEComponentDown` should fire and arrive in the channel +with a description, a runbook link and, if `grafanaUrl` is set, a dashboard +link. Scale it back and confirm the resolved message. diff --git a/schemas/cluster.schema.json b/schemas/cluster.schema.json index 4b32547..6208fbb 100644 --- a/schemas/cluster.schema.json +++ b/schemas/cluster.schema.json @@ -69,6 +69,89 @@ "bootstrapEnvironment": { "type": "string" }, + "monitorCertManager": { + "type": "boolean", + "description": "Create a ServiceMonitor for cert-manager, so certificate expiry can be alerted on. cert-manager exports the metric but ships no ServiceMonitor, so nothing collects it by default." + }, + "alerting": { + "type": "object", + "additionalProperties": false, + "description": "Where alerts go. Webhook URLs are never listed here: they are credentials and live in the cluster's GitHub Environment.", + "properties": { + "enabled": { + "type": "boolean" + }, + "minSeverity": { + "type": "string", + "enum": [ + "warning", + "critical" + ], + "description": "What reaches the channels, not what fires. Everything below this still fires and is visible in Alertmanager and on the dashboards." + }, + "grafanaUrl": { + "type": "string", + "description": "Base URL of the Grafana serving the EduIDE dashboards, no trailing slash. Used to deep-link a notification to the affected session." + }, + "channels": { + "type": "array", + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "name", + "type", + "secretKey" + ], + "properties": { + "name": { + "type": "string" + }, + "type": { + "type": "string", + "enum": [ + "slack", + "discord" + ] + }, + "secretKey": { + "type": "string", + "pattern": "^(slack|discord)-", + "description": "Key in the webhook Secret. The prefix decides which GitHub Environment secret bootstrap reads, so it must match the channel type." + }, + "channel": { + "type": "string", + "description": "Slack channel override, e.g. \"#eduide-alerts\". Ignored for Discord, where the webhook itself names the channel." + } + } + } + } + }, + "allOf": [ + { + "if": { + "properties": { + "enabled": { + "const": true + } + }, + "required": [ + "enabled" + ] + }, + "then": { + "required": [ + "channels" + ], + "properties": { + "channels": { + "minItems": 1 + } + } + } + } + ] + }, "tls": { "type": "object", "description": "TLS Secret each shared-Gateway listener role terminates with. Per cluster because the clusters issue certificates differently. A listener without a secret renders an empty certificateRef: the Gateway is accepted and simply never programs TLS for that hostname.", diff --git a/scripts/test-deploy-logic.sh b/scripts/test-deploy-logic.sh index 0dbd5ca..460a092 100755 --- a/scripts/test-deploy-logic.sh +++ b/scripts/test-deploy-logic.sh @@ -251,6 +251,56 @@ for cf in "$ROOT"/clusters/*.yaml; do ok "$cluster: $n_in of $n_all environment(s) monitored" done +# --- alerting is complete where it is switched on -------------------------- +# Alerting that fires into nowhere is worse than no alerting: it reads as +# covered. The chart already fails the render on an empty channel list, but the +# failure would land mid-bootstrap on a cluster, so the same conditions are +# checked here where the feedback is a pull request comment. +# +# The secretKey prefix is load-bearing, not cosmetic. bootstrap-cluster.yml +# picks which GitHub Environment secret to read from it, so a key named +# anything else silently gets no webhook URL. +echo +echo "=== alerting configuration is complete ===" +for cf in "$ROOT"/clusters/*.yaml; do + cluster=$(yq -r '.metadata.name' "$cf") + if [[ "$(yq -r '.spec.alerting.enabled // false' "$cf")" != "true" ]]; then + ok "$cluster: alerting off" + continue + fi + n=$(yq -r '(.spec.alerting.channels // []) | length' "$cf") + if [[ "$n" == "0" ]]; then + bad "$cluster enables alerting but declares no channels" \ + "alerts would fire and notify nobody; add spec.alerting.channels" + continue + fi + sev=$(yq -r '.spec.alerting.minSeverity // "warning"' "$cf") + if [[ "$sev" != "warning" && "$sev" != "critical" ]]; then + bad "$cluster has minSeverity '$sev'" "must be 'warning' or 'critical'" + fi + bad_channel=0 + while read -r line; do + [[ -n "$line" ]] || continue + cname=${line%% *}; ctype=${line#* }; ckey=${ctype#* }; ctype=${ctype%% *} + if [[ "$ctype" != "slack" && "$ctype" != "discord" ]]; then + bad "$cluster channel '$cname' has type '$ctype'" "supported: slack, discord" + bad_channel=1 + fi + if [[ "$ckey" != slack-* && "$ckey" != discord-* ]]; then + bad "$cluster channel '$cname' has secretKey '$ckey'" \ + "must start with slack- or discord-, which is how bootstrap picks the secret" + bad_channel=1 + fi + if [[ "$ctype" == "slack" && "$ckey" != slack-* ]] \ + || [[ "$ctype" == "discord" && "$ckey" != discord-* ]]; then + bad "$cluster channel '$cname' is $ctype but reads '$ckey'" \ + "the prefix decides which webhook URL is used, so it must match the type" + bad_channel=1 + fi + done < <(yq -r '.spec.alerting.channels[]? | .name + " " + .type + " " + .secretKey' "$cf") + [[ $bad_channel -eq 0 ]] && ok "$cluster: alerting on, $n channel(s), minSeverity $sev" +done + # --- every namespace carries the eduide- prefix ---------------------------- echo echo "=== namespaces are prefixed ===" From 402516ca6b933890cb72e9183898343dc434cdb9 Mon Sep 17 00:00:00 2001 From: Matthias Linhuber Date: Fri, 28 Aug 2026 19:22:54 +0200 Subject: [PATCH 2/3] fix: keep cluster-scoped monitoring when every environment opts out Review feedback. bootstrap-cluster.yml gated the whole monitoring block on namespaces.txt, so a cluster whose environments all set monitoring.enabled: false lost its cert-manager scraping and its alerting as well. Those are cluster-scoped and have nothing to do with environments: certificate expiry is about the Gateway's secrets, not about anyone's namespace. Monitoring is now switched on when environments ask for it or when the cluster does, and targetNamespaces is emitted only when there are any. The chart already handles the empty list - the namespace regex becomes ^$ - so per-environment rules match nothing while the cluster-scoped ones still fire. The schema accepted type: slack with secretKey: discord-alerts, which would have sent Slack-formatted payloads to a Discord webhook. Added a conditional per channel type. test-deploy-logic.sh split channel fields on spaces, so a schema-valid name like "platform alerts" shifted every field along, parsed the type as "alerts" and blocked validation on a correct manifest. Reads @tsv now. secretKey was checked only for its slack-/discord- prefix, so slack-a/b passed and would then be written into a Secret data key, which Kubernetes rejects - partway through a bootstrap, after the Gateway had been reconciled. The schema, the script and the workflow now all require ^(slack|discord)-[A-Za-z0-9._-]*$ and at most 253 characters. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_019qeiQRFu8xAMRYWPdZewjG --- .github/workflows/bootstrap-cluster.yml | 40 ++++++++++++++++----- schemas/cluster.schema.json | 47 +++++++++++++++++++++++-- scripts/test-deploy-logic.sh | 27 +++++++++----- 3 files changed, 94 insertions(+), 20 deletions(-) diff --git a/.github/workflows/bootstrap-cluster.yml b/.github/workflows/bootstrap-cluster.yml index afbbf67..614655e 100644 --- a/.github/workflows/bootstrap-cluster.yml +++ b/.github/workflows/bootstrap-cluster.yml @@ -273,24 +273,40 @@ jobs: # two environments were being scraped and the rest were not. # A cluster where every environment opted out gets monitoring # switched off, rather than PodMonitors that watch nothing. - if [[ -s namespaces.txt ]]; then + # Two independent reasons to switch monitoring on: environments opted + # into it, or the cluster asked for something cluster-scoped + # (cert-manager scraping, alerting). Gating the second on the first + # would mean a cluster whose environments all opt out silently loses + # its certificate alerts, which are not about environments at all. + CERT_MANAGER=$(yq -r '.spec.monitorCertManager // false' "clusters/${CLUSTER}.yaml") + ALERTING=$(yq -r '.spec.alerting.enabled // false' "clusters/${CLUSTER}.yaml") + if [[ -s namespaces.txt || "$CERT_MANAGER" == "true" || "$ALERTING" == "true" ]]; then { echo "monitoring:" echo " enabled: true" - echo " targetNamespaces:" - sed 's/^/ - /' namespaces.txt } >> listeners.yaml + if [[ -s namespaces.txt ]]; then + { + echo " targetNamespaces:" + sed 's/^/ - /' namespaces.txt + } >> listeners.yaml + else + # Supported, and the chart handles it: the namespace regex becomes + # ^$ so per-environment alerts match nothing and the dashboard + # pickers are empty, while the cluster-scoped rules still work. + echo "::warning::no environment on ${CLUSTER} opts into monitoring; only cluster-scoped rules will fire" + fi # cert-manager exports certificate expiry but ships no # ServiceMonitor, so by default nothing watches it. Opt in per # cluster: the webview wildcard is renewed by hand once a year and # has never had anything warning about it. - if [[ "$(yq -r '.spec.monitorCertManager // false' "clusters/${CLUSTER}.yaml")" == "true" ]]; then + if [[ "$CERT_MANAGER" == "true" ]]; then printf ' certManager:\n enabled: true\n' >> listeners.yaml fi # Alerting. The channel list is in the manifest; the webhook URLs # are not - they are credentials and come from the environment's # secrets, written to a separate file below. - if [[ "$(yq -r '.spec.alerting.enabled // false' "clusters/${CLUSTER}.yaml")" == "true" ]]; then + if [[ "$ALERTING" == "true" ]]; then { echo " alerting:" echo " enabled: true" @@ -415,13 +431,19 @@ jobs: } > alert-secrets.yaml while read -r key; do [[ -n "$key" ]] || continue + # Checked here as well as in the schema, because this is the step + # that writes the key into a Secret. Anything outside what + # Kubernetes allows in a data key would be rejected by the API + # server partway through the install, after the Gateway had already + # been reconciled. + if [[ ! "$key" =~ ^(slack|discord)-[A-Za-z0-9._-]*$ ]] || (( ${#key} > 253 )); then + echo "::error::channel secretKey '${key}' must match ^(slack|discord)-[A-Za-z0-9._-]*$ and be at most 253 characters." + echo "::error::The prefix selects the webhook secret; the rest has to be a valid Kubernetes Secret data key." + exit 1 + fi case "$key" in slack-*) value="$ALERT_WEBHOOK_SLACK" ;; discord-*) value="$ALERT_WEBHOOK_DISCORD" ;; - *) - echo "::error::channel secretKey '${key}' must start with 'slack-' or 'discord-'" - exit 1 - ;; esac if [[ -z "$value" ]]; then MISSING+=("$key") diff --git a/schemas/cluster.schema.json b/schemas/cluster.schema.json index 6208fbb..a2f5af5 100644 --- a/schemas/cluster.schema.json +++ b/schemas/cluster.schema.json @@ -116,14 +116,55 @@ }, "secretKey": { "type": "string", - "pattern": "^(slack|discord)-", - "description": "Key in the webhook Secret. The prefix decides which GitHub Environment secret bootstrap reads, so it must match the channel type." + "pattern": "^(slack|discord)-[A-Za-z0-9._-]*$", + "maxLength": 253, + "description": "Key in the webhook Secret. The prefix decides which GitHub Environment secret bootstrap reads, so it must match the channel type. The rest is restricted to what Kubernetes accepts in a Secret data key: a slash or a space here would be written into the Secret and rejected by the API server at bootstrap time." }, "channel": { "type": "string", "description": "Slack channel override, e.g. \"#eduide-alerts\". Ignored for Discord, where the webhook itself names the channel." } - } + }, + "allOf": [ + { + "if": { + "properties": { + "type": { + "const": "slack" + } + }, + "required": [ + "type" + ] + }, + "then": { + "properties": { + "secretKey": { + "pattern": "^slack-" + } + } + } + }, + { + "if": { + "properties": { + "type": { + "const": "discord" + } + }, + "required": [ + "type" + ] + }, + "then": { + "properties": { + "secretKey": { + "pattern": "^discord-" + } + } + } + } + ] } } }, diff --git a/scripts/test-deploy-logic.sh b/scripts/test-deploy-logic.sh index 460a092..75e265b 100755 --- a/scripts/test-deploy-logic.sh +++ b/scripts/test-deploy-logic.sh @@ -279,25 +279,36 @@ for cf in "$ROOT"/clusters/*.yaml; do bad "$cluster has minSeverity '$sev'" "must be 'warning' or 'critical'" fi bad_channel=0 - while read -r line; do - [[ -n "$line" ]] || continue - cname=${line%% *}; ctype=${line#* }; ckey=${ctype#* }; ctype=${ctype%% *} + # Tab-separated, not space-separated. A channel name is a free-form string and + # may contain spaces; splitting on those silently shifted every field along and + # reported a bogus "unsupported type", which would block validation on a + # perfectly valid manifest. + while IFS=$'\t' read -r cname ctype ckey; do + [[ -n "$cname" ]] || continue if [[ "$ctype" != "slack" && "$ctype" != "discord" ]]; then bad "$cluster channel '$cname' has type '$ctype'" "supported: slack, discord" bad_channel=1 fi - if [[ "$ckey" != slack-* && "$ckey" != discord-* ]]; then + # The prefix picks the GitHub Environment secret; the rest has to survive + # being used as a Kubernetes Secret data key, which allows only + # alphanumerics, '-', '_' and '.'. A slash passes a prefix-only check and is + # then rejected by the API server halfway through a bootstrap. + if [[ ! "$ckey" =~ ^(slack|discord)-[A-Za-z0-9._-]*$ ]]; then bad "$cluster channel '$cname' has secretKey '$ckey'" \ - "must start with slack- or discord-, which is how bootstrap picks the secret" + "must match ^(slack|discord)-[A-Za-z0-9._-]*\$ so it is both routable and a valid Secret key" bad_channel=1 - fi - if [[ "$ctype" == "slack" && "$ckey" != slack-* ]] \ + elif [[ "$ctype" == "slack" && "$ckey" != slack-* ]] \ || [[ "$ctype" == "discord" && "$ckey" != discord-* ]]; then bad "$cluster channel '$cname' is $ctype but reads '$ckey'" \ "the prefix decides which webhook URL is used, so it must match the type" bad_channel=1 fi - done < <(yq -r '.spec.alerting.channels[]? | .name + " " + .type + " " + .secretKey' "$cf") + if (( ${#ckey} > 253 )); then + bad "$cluster channel '$cname' has a secretKey of ${#ckey} characters" \ + "Kubernetes Secret keys are limited to 253" + bad_channel=1 + fi + done < <(yq -r '.spec.alerting.channels[]? | [.name, .type, .secretKey] | @tsv' "$cf") [[ $bad_channel -eq 0 ]] && ok "$cluster: alerting on, $n channel(s), minSeverity $sev" done From 82e5f7f3aa724f127a29f4e14a6bbb8dff20770a Mon Sep 17 00:00:00 2001 From: Matthias Linhuber Date: Fri, 28 Aug 2026 19:33:17 +0200 Subject: [PATCH 3/3] feat: alert Bonn and Mannheim in their own Discord channels Both installations share the eduide cluster and belong to different people, so each gets its own Discord and neither sees the other's incidents. Cluster-scoped alerts - a certificate expiring, the conversion webhook failing - affect both and go to both. The webhook secret is now looked up by name rather than one per type: secretKey `discord-mannheim` reads ALERT_WEBHOOK_DISCORD_MANNHEIM. A single ALERT_WEBHOOK_DISCORD per cluster could not express two installations wanting different channels. GitHub expressions cannot index secrets by a computed name, so the map is passed in as JSON and one key is picked out with jq. test-deploy-logic.sh checks every scoped namespace against the environments actually on that cluster. A typo there would match nothing, and the channel would quietly receive only cluster-scoped alerts while the installation's own alerts went to whoever was unscoped - invisible at render time, since the matcher is just a regex. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_019qeiQRFu8xAMRYWPdZewjG --- .github/workflows/bootstrap-cluster.yml | 28 ++++++++----- clusters/eduide.yaml | 27 +++++++++---- clusters/tum-production.yaml | 10 ++--- clusters/tum-student.yaml | 10 ++--- docs/monitoring-setup.md | 53 ++++++++++++++++++++++--- schemas/cluster.schema.json | 8 ++++ scripts/test-deploy-logic.sh | 18 +++++++++ 7 files changed, 121 insertions(+), 33 deletions(-) diff --git a/.github/workflows/bootstrap-cluster.yml b/.github/workflows/bootstrap-cluster.yml index 614655e..aa7654b 100644 --- a/.github/workflows/bootstrap-cluster.yml +++ b/.github/workflows/bootstrap-cluster.yml @@ -401,8 +401,17 @@ jobs: - name: Collect the alert webhook URLs env: - ALERT_WEBHOOK_SLACK: ${{ secrets.ALERT_WEBHOOK_SLACK }} - ALERT_WEBHOOK_DISCORD: ${{ secrets.ALERT_WEBHOOK_DISCORD }} + # Looked up by name rather than listed one per type, because one + # cluster can host installations that belong to different people and + # want different channels: Bonn and Mannheim share the `eduide` + # cluster and each has its own Discord. A channel's secretKey + # `discord-mannheim` reads `ALERT_WEBHOOK_DISCORD_MANNHEIM`. + # + # GitHub expressions cannot index `secrets` by a computed name, so the + # whole map is passed in and one key is picked out with jq. Nothing in + # this step echoes it, and Actions masks secret values in logs + # regardless. + ALL_SECRETS: ${{ toJSON(secrets) }} run: | set -euo pipefail # A Slack or Discord webhook URL is a credential: anyone holding it can @@ -441,20 +450,19 @@ jobs: echo "::error::The prefix selects the webhook secret; the rest has to be a valid Kubernetes Secret data key." exit 1 fi - case "$key" in - slack-*) value="$ALERT_WEBHOOK_SLACK" ;; - discord-*) value="$ALERT_WEBHOOK_DISCORD" ;; - esac + # discord-mannheim -> ALERT_WEBHOOK_DISCORD_MANNHEIM + secret_name="ALERT_WEBHOOK_$(printf '%s' "$key" | tr '[:lower:]-' '[:upper:]_')" + value=$(printf '%s' "$ALL_SECRETS" | jq -r --arg n "$secret_name" '.[$n] // ""') if [[ -z "$value" ]]; then - MISSING+=("$key") + MISSING+=("${key} (expected secret ${secret_name})") continue fi echo " ${key}: \"$(printf '%s' "$value" | base64 | tr -d '\n')\"" >> alert-secrets.yaml done < <(yq -r '.spec.alerting.channels[]?.secretKey' "clusters/${CLUSTER}.yaml") if (( ${#MISSING[@]} > 0 )); then - echo "::error::alerting is enabled on ${CLUSTER} but no webhook URL was supplied for: ${MISSING[*]}" - echo "::error::set ALERT_WEBHOOK_SLACK and/or ALERT_WEBHOOK_DISCORD on the '${{ needs.resolve.outputs.environment }}' environment." - echo "::error::See docs/monitoring.md." + echo "::error::alerting is enabled on ${CLUSTER} but no webhook URL was supplied for:" + for m in "${MISSING[@]}"; do echo "::error:: ${m}"; done + echo "::error::Set them on the '${{ needs.resolve.outputs.environment }}' environment. See docs/monitoring-setup.md." exit 1 fi echo "collected $(yq -r '.spec.alerting.channels | length' "clusters/${CLUSTER}.yaml") webhook(s)" diff --git a/clusters/eduide.yaml b/clusters/eduide.yaml index 24ca3e1..0e0ff0e 100644 --- a/clusters/eduide.yaml +++ b/clusters/eduide.yaml @@ -96,15 +96,28 @@ spec: monitorCertManager: true # Where alerts go. The channel list is here; the webhook URLs are not - they - # are credentials and live in the cluster's GitHub Environment as - # ALERT_WEBHOOK_SLACK and ALERT_WEBHOOK_DISCORD. A channel's `secretKey` must - # start with `slack-` or `discord-`, which is how the workflow knows which - # secret to read. + # are credentials and live in this cluster's GitHub Environment. A channel's + # `secretKey` names the secret: `discord-mannheim` is read from + # ALERT_WEBHOOK_DISCORD_MANNHEIM, so one cluster can hold a different webhook + # per installation. # # `minSeverity` is what reaches the channels, not what fires. Everything below # it still fires and is visible in Alertmanager and on the dashboards; it just - # does not page anyone. See docs/monitoring.md. + # does not page anyone. See docs/monitoring-setup.md. alerting: - enabled: false + enabled: true minSeverity: warning - channels: [] + # Bonn and Mannheim share this cluster and belong to different people, so + # each has its own Discord and neither sees the other's incidents. Alerts + # that belong to neither namespace - a certificate expiring, the conversion + # webhook failing - are cluster-scoped and go to both, because they affect + # both installations. + channels: + - name: mannheim + type: discord + secretKey: discord-mannheim + environments: [eduide-mannheim] + - name: bonn + type: discord + secretKey: discord-bonn + environments: [eduide-bonn] diff --git a/clusters/tum-production.yaml b/clusters/tum-production.yaml index 34b6d4e..452623c 100644 --- a/clusters/tum-production.yaml +++ b/clusters/tum-production.yaml @@ -81,14 +81,14 @@ spec: monitorCertManager: true # Where alerts go. The channel list is here; the webhook URLs are not - they - # are credentials and live in the cluster's GitHub Environment as - # ALERT_WEBHOOK_SLACK and ALERT_WEBHOOK_DISCORD. A channel's `secretKey` must - # start with `slack-` or `discord-`, which is how the workflow knows which - # secret to read. + # are credentials and live in this cluster's GitHub Environment. A channel's + # `secretKey` names the secret: `discord-mannheim` is read from + # ALERT_WEBHOOK_DISCORD_MANNHEIM, so one cluster can hold a different webhook + # per installation. # # `minSeverity` is what reaches the channels, not what fires. Everything below # it still fires and is visible in Alertmanager and on the dashboards; it just - # does not page anyone. See docs/monitoring.md. + # does not page anyone. See docs/monitoring-setup.md. alerting: enabled: false minSeverity: warning diff --git a/clusters/tum-student.yaml b/clusters/tum-student.yaml index cf6fced..5e19c0c 100644 --- a/clusters/tum-student.yaml +++ b/clusters/tum-student.yaml @@ -47,14 +47,14 @@ spec: monitorCertManager: true # Where alerts go. The channel list is here; the webhook URLs are not - they - # are credentials and live in the cluster's GitHub Environment as - # ALERT_WEBHOOK_SLACK and ALERT_WEBHOOK_DISCORD. A channel's `secretKey` must - # start with `slack-` or `discord-`, which is how the workflow knows which - # secret to read. + # are credentials and live in this cluster's GitHub Environment. A channel's + # `secretKey` names the secret: `discord-mannheim` is read from + # ALERT_WEBHOOK_DISCORD_MANNHEIM, so one cluster can hold a different webhook + # per installation. # # `minSeverity` is what reaches the channels, not what fires. Everything below # it still fires and is visible in Alertmanager and on the dashboards; it just - # does not page anyone. See docs/monitoring.md. + # does not page anyone. See docs/monitoring-setup.md. alerting: enabled: false minSeverity: warning diff --git a/docs/monitoring-setup.md b/docs/monitoring-setup.md index 3d29d5d..2c1ff78 100644 --- a/docs/monitoring-setup.md +++ b/docs/monitoring-setup.md @@ -71,6 +71,36 @@ spec: channel: "#eduide-alerts" ``` +**A channel can be scoped to one installation.** One cluster may host +installations that belong to different people: Bonn and Mannheim share the +`eduide` cluster, and neither wants the other's incidents. + +```yaml + channels: + - name: mannheim + type: discord + secretKey: discord-mannheim + environments: [eduide-mannheim] + - name: bonn + type: discord + secretKey: discord-bonn + environments: [eduide-bonn] +``` + +`environments` lists **namespaces**, not hostnames, and each becomes a sub-route +matched on `eduide_namespace`. First match wins. + +**Anything no scoped channel claims goes to every channel.** That is deliberate +and it is the part worth remembering: a certificate expiring or the conversion +webhook failing is cluster-scoped, belongs to no tenant namespace, and would +otherwise be dropped for failing to match a tenant route. Those are the alerts +you least want to lose, so they go to everyone. + +A scoped channel pointing at a namespace no environment on that cluster uses +would match nothing and quietly receive only the cluster-scoped alerts, so +`test-deploy-logic.sh` checks every `environments` entry against the +environments actually on the cluster. + **`minSeverity` is what reaches the channels, not what fires.** Everything fires and is visible in Alertmanager and on the dashboards; the channels get a filtered subset. This is deliberate and it is the whole reason the split @@ -139,18 +169,29 @@ Webhook URLs are credentials. They never go in a manifest, a values file or a `--set`, which would put them in the process list and in Actions debug logs. 1. Create the incoming webhook in Slack or Discord. -2. Put it on the **cluster** GitHub Environment: +2. Put it on the **cluster** GitHub Environment. The secret is named after the + channel's `secretKey`, uppercased with hyphens as underscores and prefixed + `ALERT_WEBHOOK_`, so one cluster can hold a different webhook per + installation: + + | `secretKey` | GitHub Environment secret | + |---|---| + | `discord-mannheim` | `ALERT_WEBHOOK_DISCORD_MANNHEIM` | + | `discord-bonn` | `ALERT_WEBHOOK_DISCORD_BONN` | + | `slack-platform` | `ALERT_WEBHOOK_SLACK_PLATFORM` | ```bash REPO=EduIDE/EduIDE-deployment - gh secret set ALERT_WEBHOOK_SLACK --repo "$REPO" --env cluster-tum-student - gh secret set ALERT_WEBHOOK_DISCORD --repo "$REPO" --env cluster-tum-student + gh secret set ALERT_WEBHOOK_DISCORD_MANNHEIM --repo "$REPO" --env cluster-eduide < webhook.txt ``` + Pipe from a file or use `--body`; never paste a webhook URL into a shell you + share, and never into a manifest. + 3. Add the channel to `clusters/.yaml`. The `secretKey` prefix decides - which secret is read, so it must be `slack-*` for a Slack channel and - `discord-*` for a Discord one. `test-deploy-logic.sh` and the cluster schema - both check this. + which webhook type it is, so it must be `slack-*` for a Slack channel and + `discord-*` for a Discord one, and the rest must be a valid Kubernetes Secret + data key. `test-deploy-logic.sh` and the cluster schema both check this. 4. Re-run `Bootstrap cluster`. Alerting enabled with no channels fails the render rather than firing into diff --git a/schemas/cluster.schema.json b/schemas/cluster.schema.json index a2f5af5..a81a956 100644 --- a/schemas/cluster.schema.json +++ b/schemas/cluster.schema.json @@ -123,6 +123,14 @@ "channel": { "type": "string", "description": "Slack channel override, e.g. \"#eduide-alerts\". Ignored for Discord, where the webhook itself names the channel." + }, + "environments": { + "type": "array", + "minItems": 1, + "items": { + "type": "string" + }, + "description": "Namespaces whose alerts go to this channel instead of to all of them. Use when one cluster hosts installations belonging to different people. Alerts that match no scoped channel - including every cluster-scoped alert, such as a certificate expiring - still go to all channels." } }, "allOf": [ diff --git a/scripts/test-deploy-logic.sh b/scripts/test-deploy-logic.sh index 75e265b..da85379 100755 --- a/scripts/test-deploy-logic.sh +++ b/scripts/test-deploy-logic.sh @@ -309,6 +309,24 @@ for cf in "$ROOT"/clusters/*.yaml; do bad_channel=1 fi done < <(yq -r '.spec.alerting.channels[]? | [.name, .type, .secretKey] | @tsv' "$cf") + + # A channel scoped to an environment that does not exist on this cluster + # matches nothing, so that channel silently receives only the cluster-scoped + # alerts and nobody notices the tenant's own alerts are going elsewhere. + # Typos here are invisible at render time: the matcher is just a regex. + while IFS=$'\t' read -r cname cenv; do + [[ -n "$cenv" ]] || continue + found=0 + for f in "$ROOT"/environments/*/env.yaml; do + [[ "$(yq -r '.spec.cluster' "$f")" == "$cluster" ]] || continue + [[ "$(yq -r '.spec.namespace' "$f")" == "$cenv" ]] && { found=1; break; } + done + if [[ $found -eq 0 ]]; then + bad "$cluster channel '$cname' is scoped to namespace '$cenv'" \ + "no environment with that namespace lives on $cluster, so the route matches nothing" + bad_channel=1 + fi + done < <(yq -r '.spec.alerting.channels[]? as $c | ($c.environments // [])[] | [$c.name, .] | @tsv' "$cf") [[ $bad_channel -eq 0 ]] && ok "$cluster: alerting on, $n channel(s), minSeverity $sev" done