Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
121 changes: 114 additions & 7 deletions .github/workflows/bootstrap-cluster.yml
Original file line number Diff line number Diff line change
Expand Up @@ -273,15 +273,54 @@
# two environments were being scraped and the rest were not.
# A cluster where every environment opted out gets monitoring
# switched off, rather than PodMonitors that watch nothing.
if [[ -s namespaces.txt ]]; then
# Two independent reasons to switch monitoring on: environments opted
# into it, or the cluster asked for something cluster-scoped
# (cert-manager scraping, alerting). Gating the second on the first
# would mean a cluster whose environments all opt out silently loses
# its certificate alerts, which are not about environments at all.
CERT_MANAGER=$(yq -r '.spec.monitorCertManager // false' "clusters/${CLUSTER}.yaml")
ALERTING=$(yq -r '.spec.alerting.enabled // false' "clusters/${CLUSTER}.yaml")
if [[ -s namespaces.txt || "$CERT_MANAGER" == "true" || "$ALERTING" == "true" ]]; then
{
echo "monitoring:"
echo " enabled: true"
echo " targetNamespaces:"
sed 's/^/ - /' namespaces.txt
echo " sessionNamespaces:"
sed 's/^/ - /' namespaces.txt
} >> listeners.yaml
if [[ -s namespaces.txt ]]; then
{
echo " targetNamespaces:"
sed 's/^/ - /' namespaces.txt
} >> listeners.yaml
else
# Supported, and the chart handles it: the namespace regex becomes
# ^$ so per-environment alerts match nothing and the dashboard
# pickers are empty, while the cluster-scoped rules still work.
echo "::warning::no environment on ${CLUSTER} opts into monitoring; only cluster-scoped rules will fire"
fi
# cert-manager exports certificate expiry but ships no
# ServiceMonitor, so by default nothing watches it. Opt in per
# cluster: the webview wildcard is renewed by hand once a year and
# has never had anything warning about it.
if [[ "$CERT_MANAGER" == "true" ]]; then
printf ' certManager:\n enabled: true\n' >> listeners.yaml
fi
# Alerting. The channel list is in the manifest; the webhook URLs
# are not - they are credentials and come from the environment's
# secrets, written to a separate file below.
if [[ "$ALERTING" == "true" ]]; then
{
echo " alerting:"
echo " enabled: true"
echo " minSeverity: $(yq -r '.spec.alerting.minSeverity // "warning"' "clusters/${CLUSTER}.yaml")"
if [[ "$(yq -r '.spec.alerting.grafanaUrl // ""' "clusters/${CLUSTER}.yaml")" != "" ]]; then
echo " grafanaUrl: $(yq -r '.spec.alerting.grafanaUrl' "clusters/${CLUSTER}.yaml")"
fi
# Passed through as YAML rather than rebuilt field by field: the
# manifest's channel keys are exactly the chart's, so there is
# nothing to translate and nothing to forget when a key is added.
echo " channels:"
yq -r '.spec.alerting.channels' "clusters/${CLUSTER}.yaml" | sed 's/^/ /'
} >> listeners.yaml
fi
Comment thread
Mtze marked this conversation as resolved.
else
echo "::warning::no environment on ${CLUSTER} opts into monitoring"
printf 'monitoring:\n enabled: false\n' >> listeners.yaml
Expand Down Expand Up @@ -360,12 +399,80 @@
key: "$(printf '%s' '${{ secrets.THEIA_WILDCARD_CERTIFICATE_KEY }}' | base64 | tr -d '\n')"
EOF

- name: Collect the alert webhook URLs
env:
# Looked up by name rather than listed one per type, because one
# cluster can host installations that belong to different people and
# want different channels: Bonn and Mannheim share the `eduide`
# cluster and each has its own Discord. A channel's secretKey
# `discord-mannheim` reads `ALERT_WEBHOOK_DISCORD_MANNHEIM`.
#
# GitHub expressions cannot index `secrets` by a computed name, so the
# whole map is passed in and one key is picked out with jq. Nothing in
# this step echoes it, and Actions masks secret values in logs
# regardless.
ALL_SECRETS: ${{ toJSON(secrets) }}
run: |
set -euo pipefail
# A Slack or Discord webhook URL is a credential: anyone holding it can
# post into the channel. It goes in a values file read from a secret,
# never on a `--set`, which would put it in the process list and in
# Actions debug logs.
#
# Always written, even when alerting is off, so the later helm calls
# can name the file unconditionally.
install -m 600 /dev/null alert-secrets.yaml
if [[ "$(yq -r '.spec.alerting.enabled // false' "clusters/${CLUSTER}.yaml")" != "true" ]]; then
echo "{}" > alert-secrets.yaml
echo "alerting is off for ${CLUSTER}"
exit 0
fi
# Every channel names the key it reads. A channel whose secret is not
# set would render an AlertmanagerConfig that notifies nobody and
# reports no error, so fail here instead.
MISSING=()
{
echo "monitoring:"
echo " alerting:"
echo " webhookSecret:"
echo " create: true"
echo " data:"
} > alert-secrets.yaml
while read -r key; do
[[ -n "$key" ]] || continue
# Checked here as well as in the schema, because this is the step
# that writes the key into a Secret. Anything outside what
# Kubernetes allows in a data key would be rejected by the API
# server partway through the install, after the Gateway had already
# been reconciled.
if [[ ! "$key" =~ ^(slack|discord)-[A-Za-z0-9._-]*$ ]] || (( ${#key} > 253 )); then
echo "::error::channel secretKey '${key}' must match ^(slack|discord)-[A-Za-z0-9._-]*$ and be at most 253 characters."
echo "::error::The prefix selects the webhook secret; the rest has to be a valid Kubernetes Secret data key."
exit 1
fi
# discord-mannheim -> ALERT_WEBHOOK_DISCORD_MANNHEIM
secret_name="ALERT_WEBHOOK_$(printf '%s' "$key" | tr '[:lower:]-' '[:upper:]_')"
value=$(printf '%s' "$ALL_SECRETS" | jq -r --arg n "$secret_name" '.[$n] // ""')
if [[ -z "$value" ]]; then
MISSING+=("${key} (expected secret ${secret_name})")
continue
fi
echo " ${key}: \"$(printf '%s' "$value" | base64 | tr -d '\n')\"" >> alert-secrets.yaml
done < <(yq -r '.spec.alerting.channels[]?.secretKey' "clusters/${CLUSTER}.yaml")
if (( ${#MISSING[@]} > 0 )); then
echo "::error::alerting is enabled on ${CLUSTER} but no webhook URL was supplied for:"
for m in "${MISSING[@]}"; do echo "::error:: ${m}"; done
echo "::error::Set them on the '${{ needs.resolve.outputs.environment }}' environment. See docs/monitoring-setup.md."
exit 1
fi
echo "collected $(yq -r '.spec.alerting.channels | length' "clusters/${CLUSTER}.yaml") webhook(s)"

- name: Preview
run: |
set -euo pipefail
helm diff upgrade eduide-cluster oci://ghcr.io/eduide/charts/eduide-cluster \
--version "${{ inputs.chart_version }}" \
-n eduide-system -f listeners.yaml -f gw-secrets.yaml \
-n eduide-system -f listeners.yaml -f gw-secrets.yaml -f alert-secrets.yaml \
--allow-unreleased --no-color > gw.diff 2>&1 || true
{
echo "<details><summary>Cluster chart pending change</summary>"
Expand All @@ -379,7 +486,7 @@
helm upgrade --install eduide-cluster oci://ghcr.io/eduide/charts/eduide-cluster \
--version "${{ inputs.chart_version }}" \
-n eduide-system --create-namespace \
-f listeners.yaml -f gw-secrets.yaml --wait --timeout 10m
-f listeners.yaml -f gw-secrets.yaml -f alert-secrets.yaml --wait --timeout 10m

- name: Stamp this cluster with its name
if: ${{ !inputs.dry_run }}
Expand Down
32 changes: 29 additions & 3 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -76,12 +76,38 @@ both, with no clue why. An installation with no identity provider yet sets
`keycloak.allowUnauthenticated: true` and says so; Bonn does.

**Monitoring is on by default and opted out per environment** with
`monitoring.enabled: false`. The PodMonitors themselves are in the cluster
chart - they must be created in Rancher's namespace to be discovered, and one
`monitoring.enabled: false`. The PodMonitor itself is in the cluster
chart - it must be created in Rancher's namespace to be discovered, and one
per tenant would collide on names - so the flag decides whether the
environment's namespace is in the list they watch. Do not confuse it with
environment's namespace is in the list it watches. Do not confuse it with
`monitor.enable`, the operator's session activity tracker.

**Session pods export no metrics, and nothing should try to scrape them.** They
are Theia IDEs serving HTML; a PodMonitor pointed at them collects nothing and
reports every target as down. There was one, for a year, and it never produced
a sample. Everything the session dashboards show comes from cAdvisor, kubelet
and kube-state-metrics, which need no PodMonitor at all. The REST service is
the only EduIDE component that is scraped, and it exports exactly one custom
metric - session startup latency, registered lazily on the first session, so it
does not exist on a freshly restarted service.

**Alert `namespace` labels are a routing artifact; silence on
`eduide_namespace`.** Every EduIDE alert claims `namespace: eduide-system`
whatever environment it concerns, because the Prometheus Operator prepends a
`namespace = <the AlertmanagerConfig's own namespace>` matcher to every route it
generates and the alert would otherwise reach no receiver. The real environment
is in `eduide_namespace`, and grouping and inhibition must use that one too.
Webhook URLs never appear in a manifest: `spec.alerting.channels` names a
`secretKey`, whose `slack-`/`discord-` prefix tells bootstrap which GitHub
Environment secret to read. See `docs/monitoring-setup.md`.

**A selector that matches nothing looks exactly like a healthy platform.** Four
dashboard panels selected on `service=~"theia-.*"`, a label that stopped
existing in 2.0.0, and rendered empty for months; the namespace picker offered
`theia` and `test1` long after both namespaces were gone. Neither could fail a
render test. Check new expressions against a live Prometheus before shipping
them, not just `helm template`.

**Keycloak is per environment, including `authUrl`.** TUM installations share a
server and differ only by realm; Bonn and Mannheim bring their own. Nothing
about the identity provider belongs in `_base.yaml`, and secrets never go in a
Expand Down
32 changes: 32 additions & 0 deletions clusters/eduide.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -89,3 +89,35 @@ spec:

runner: ubuntu-latest
bootstrapEnvironment: cluster-eduide

# Scrape cert-manager so certificates can be alerted on before they lapse.
# cert-manager exports expiry but ships no ServiceMonitor, so nothing watches
# it by default. The webview wildcard is renewed by hand once a year.
monitorCertManager: true

# Where alerts go. The channel list is here; the webhook URLs are not - they
# are credentials and live in this cluster's GitHub Environment. A channel's
# `secretKey` names the secret: `discord-mannheim` is read from
# ALERT_WEBHOOK_DISCORD_MANNHEIM, so one cluster can hold a different webhook
# per installation.
#
# `minSeverity` is what reaches the channels, not what fires. Everything below
# it still fires and is visible in Alertmanager and on the dashboards; it just
# does not page anyone. See docs/monitoring-setup.md.
alerting:
enabled: true
minSeverity: warning
# Bonn and Mannheim share this cluster and belong to different people, so
# each has its own Discord and neither sees the other's incidents. Alerts
# that belong to neither namespace - a certificate expiring, the conversion
# webhook failing - are cluster-scoped and go to both, because they affect
# both installations.
channels:
- name: mannheim
type: discord
secretKey: discord-mannheim
environments: [eduide-mannheim]
- name: bonn
type: discord
secretKey: discord-bonn
environments: [eduide-bonn]
19 changes: 19 additions & 0 deletions clusters/tum-production.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -74,3 +74,22 @@ spec:

runner: ubuntu-latest
bootstrapEnvironment: cluster-tum-production

# Scrape cert-manager so certificates can be alerted on before they lapse.
# cert-manager exports expiry but ships no ServiceMonitor, so nothing watches
# it by default. The webview wildcard is renewed by hand once a year.
monitorCertManager: true

# Where alerts go. The channel list is here; the webhook URLs are not - they
# are credentials and live in this cluster's GitHub Environment. A channel's
# `secretKey` names the secret: `discord-mannheim` is read from
# ALERT_WEBHOOK_DISCORD_MANNHEIM, so one cluster can hold a different webhook
# per installation.
#
# `minSeverity` is what reaches the channels, not what fires. Everything below
# it still fires and is visible in Alertmanager and on the dashboards; it just
# does not page anyone. See docs/monitoring-setup.md.
alerting:
enabled: false
minSeverity: warning
channels: []
19 changes: 19 additions & 0 deletions clusters/tum-student.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -40,3 +40,22 @@ spec:
runner: ubuntu-latest

bootstrapEnvironment: cluster-tum-student

# Scrape cert-manager so certificates can be alerted on before they lapse.
# cert-manager exports expiry but ships no ServiceMonitor, so nothing watches
# it by default. The webview wildcard is renewed by hand once a year.
monitorCertManager: true

# Where alerts go. The channel list is here; the webhook URLs are not - they
# are credentials and live in this cluster's GitHub Environment. A channel's
# `secretKey` names the secret: `discord-mannheim` is read from
# ALERT_WEBHOOK_DISCORD_MANNHEIM, so one cluster can hold a different webhook
# per installation.
#
# `minSeverity` is what reaches the channels, not what fires. Everything below
# it still fires and is visible in Alertmanager and on the dashboards; it just
# does not page anyone. See docs/monitoring-setup.md.
alerting:
enabled: false
minSeverity: warning
channels: []
Loading
Loading