diff --git a/.github/workflows/molecule.yml b/.github/workflows/molecule.yml index feaa798..d2f3d88 100644 --- a/.github/workflows/molecule.yml +++ b/.github/workflows/molecule.yml @@ -36,6 +36,7 @@ jobs: - name: Install Galaxy collections run: make deps - name: molecule test - # --all runs both scenarios: default (operator-provisioned keystore) and - # generate-keystore (opt-in host-side wallet generation). + # --all runs every scenario under ansible/molecule/: default + # (operator-provisioned keystore), generate-keystore (opt-in host-side + # wallet generation), and slow-readiness (advisory /metrics probe timeout). run: molecule test --all diff --git a/ansible/molecule/default/files/decdn-node-stub b/ansible/molecule/default/files/decdn-node-stub index 401cfa8..6196a2f 100755 --- a/ansible/molecule/default/files/decdn-node-stub +++ b/ansible/molecule/default/files/decdn-node-stub @@ -10,6 +10,12 @@ can be exercised end-to-end without a published release or a live chain: the role's `/metrics` readiness probe succeeds, and block forever under systemd (Type=simple). CLI args (`--config`, `--keystore-password-file`) are accepted and ignored. + If the marker file NO_METRICS_MARKER exists, it instead + blocks forever WITHOUT binding metrics — standing in for a + healthy node whose metrics listener binds late (behind the + startup PaymentChannel bootstrap). The `slow-readiness` + scenario uses this to prove the role's advisory probe warns + (not fails) on timeout while the unit stays `running`. * `key-gen ...` -> stand in for the CLI's wallet generator (exercised by the `generate-keystore` scenario). Enforces the contract the ROLE ASSUMES (not a verified real-CLI fact): the --password-file is @@ -25,11 +31,16 @@ prep, config/unit rendering, hardening, and loopback binding, not node logic. """ import os import sys +import time from http.server import BaseHTTPRequestHandler, HTTPServer METRICS_ADDR = ("127.0.0.1", 9090) # matches the role's default metrics_bind/port # Must match decdn_node_version in converge.yml (the version-match backstop). VERSION = "decdn-node 0.0.0-molecule-stub" +# When this file exists, `run` blocks WITHOUT binding metrics — simulating a +# healthy node whose metrics listener binds late. The slow-readiness scenario +# stages it (see that scenario's prepare.yml) to exercise the advisory-timeout path. +NO_METRICS_MARKER = "/etc/decdn/stub-no-metrics" class _Handler(BaseHTTPRequestHandler): @@ -92,6 +103,11 @@ def main(): if args and args[0] == "key-gen": return key_gen(args[1:]) if args and args[0] == "run": + # Late-bind simulation: stay alive (Type=simple => unit stays `running`) + # but never bind metrics, so the role's advisory probe times out. + if os.path.isfile(NO_METRICS_MARKER): + while True: + time.sleep(3600) try: HTTPServer(METRICS_ADDR, _Handler).serve_forever() except OSError as exc: diff --git a/ansible/molecule/slow-readiness/converge.yml b/ansible/molecule/slow-readiness/converge.yml new file mode 100644 index 0000000..bb7466c --- /dev/null +++ b/ansible/molecule/slow-readiness/converge.yml @@ -0,0 +1,32 @@ +--- +# Converge the decdn_node role against the stub in late-bind mode with a tiny +# readiness window. The advisory /metrics probe MUST time out and WARN without +# failing — so this play reaching the end (and the idempotence re-run) is itself +# the core assertion of the fix. verify.yml then confirms the running-but-no- +# metrics end state. Chain/contract knobs mirror the default scenario's +# well-formed placeholders (the role's fail-loud asserts still run against them). +- name: Converge + hosts: all + become: true + vars: + # Reuse the default scenario's stub binary rather than duplicating it. + stub_bin: "{{ lookup('ansible.builtin.env', 'MOLECULE_PROJECT_DIRECTORY') }}/molecule/default/files/decdn-node-stub" + decdn_node_install_method: manual + decdn_node_manual_bin_src: "{{ stub_bin }}" + decdn_cli_manual_bin_src: "{{ stub_bin }}" + decdn_node_version: "0.0.0-molecule-stub" + decdn_rpc_url: "https://rpc.example.invalid/" + decdn_chain_id: 421614 + decdn_region: "US" + decdn_payment_channel_address: "0x1111111111111111111111111111111111111111" + decdn_capacity_bond_address: "0x2222222222222222222222222222222222222222" + decdn_slash_judge_address: "0x3333333333333333333333333333333333333333" + decdn_cache_origin_kind: "http" + decdn_cache_origin_url: "https://origin.example.invalid/" + # Tiny probe window (~2s) so the advisory timeout is exercised fast. The stub + # never binds metrics in this scenario, so any window times out — this just + # keeps the scenario quick. + decdn_readiness_retries: 2 + decdn_readiness_delay: 1 + roles: + - role: decdn_node diff --git a/ansible/molecule/slow-readiness/molecule.yml b/ansible/molecule/slow-readiness/molecule.yml new file mode 100644 index 0000000..cb20924 --- /dev/null +++ b/ansible/molecule/slow-readiness/molecule.yml @@ -0,0 +1,44 @@ +--- +# Regression scenario for the ADVISORY /metrics readiness probe. Runs the stub +# daemon in its "late bind" mode (it stays alive but never binds metrics — see +# molecule/default/files/decdn-node-stub, gated on /etc/decdn/stub-no-metrics), +# with a deliberately tiny readiness window. It proves the fix: the role's probe +# WARNS (does not fail) on timeout, the deploy still succeeds, and the hard +# service-state gate confirms the unit is `running`. Reuses the default +# scenario's stub binary rather than duplicating it. +# +# `baseline` is NOT exercised here (same rationale as the default scenario: +# host-level hardening is real-host-only — see molecule/default/molecule.yml). +dependency: + name: galaxy + options: + requirements-file: ../../requirements.yml +driver: + name: docker +platforms: + - name: decdn-node-slow-readiness + # Same digest-pinned image as the default scenario (repo convention). Re-resolve + # both together to bump. + image: geerlingguy/docker-debian12-ansible@sha256:4553092be2c00b1ffe580927b9ff03f3c3a0df32b7dd693a3eb02efb6c2b77b7 + pre_build_image: true + command: /usr/lib/systemd/systemd + privileged: true + cgroupns_mode: host + volumes: + - /sys/fs/cgroup:/sys/fs/cgroup:rw +provisioner: + name: ansible + env: + ANSIBLE_ROLES_PATH: "${MOLECULE_PROJECT_DIRECTORY}/roles" + ANSIBLE_COLLECTIONS_PATH: "${MOLECULE_PROJECT_DIRECTORY}/collections" +verifier: + name: ansible +scenario: + test_sequence: + - dependency + - create + - prepare + - converge + - idempotence + - verify + - destroy diff --git a/ansible/molecule/slow-readiness/prepare.yml b/ansible/molecule/slow-readiness/prepare.yml new file mode 100644 index 0000000..a8abe3a --- /dev/null +++ b/ansible/molecule/slow-readiness/prepare.yml @@ -0,0 +1,48 @@ +--- +# Same operator-keystore staging as the default scenario (the role's pre-start +# gate needs keystore + node.secret + password to exist), PLUS the marker that +# puts the stub daemon into "late bind" mode so it never binds /metrics — which +# is what makes the role's advisory readiness probe time out. +- name: Prepare + hosts: all + become: true + vars: + decdn_home: /var/lib/decdn + decdn_etc: /etc/decdn + tasks: + - name: Ensure data + config directories exist + ansible.builtin.file: + path: "{{ item }}" + state: directory + mode: "0755" + loop: + - "{{ decdn_home }}" + - "{{ decdn_etc }}" + + # Staged at 0644 (looser than target); the role locks them to 0600. + - name: Stage a placeholder eth keystore + ansible.builtin.copy: + dest: "{{ decdn_home }}/keystore.json" + content: "{{ '{}' }}\n" + mode: "0644" + + - name: Stage a placeholder keystore password file + ansible.builtin.copy: + dest: "{{ decdn_etc }}/keystore.password" + content: "molecule-placeholder\n" + mode: "0644" + + - name: Stage a placeholder node identity + ansible.builtin.copy: + dest: "{{ decdn_home }}/node.secret" + content: "molecule-placeholder-node-secret\n" + mode: "0644" + + # The stub reads this marker (NO_METRICS_MARKER in decdn-node-stub) and, when + # present, blocks forever WITHOUT binding metrics — simulating a healthy node + # whose metrics listener binds late. That drives the advisory-timeout path. + - name: Put the stub daemon into late-bind (no-metrics) mode + ansible.builtin.copy: + dest: "{{ decdn_etc }}/stub-no-metrics" + content: "1\n" + mode: "0644" diff --git a/ansible/molecule/slow-readiness/verify.yml b/ansible/molecule/slow-readiness/verify.yml new file mode 100644 index 0000000..dababcc --- /dev/null +++ b/ansible/molecule/slow-readiness/verify.yml @@ -0,0 +1,43 @@ +--- +# The core assertion of this scenario is IMPLICIT: molecule only reaches `verify` +# if `converge` (and the idempotence re-run) succeeded — i.e. the advisory +# /metrics probe timed out WITHOUT failing the deploy. This play nails down the +# rest of that end state: the unit is running (the hard service-state gate still +# passed) even though metrics never bound (proving the timeout path was actually +# taken, not silently short-circuited by a metrics socket that came up anyway). +- name: Verify + hosts: all + become: true + tasks: + - name: Confirm the stub really ran in late-bind (no-metrics) mode + ansible.builtin.stat: + path: /etc/decdn/stub-no-metrics + register: no_metrics_marker + + - name: Collect listening TCP sockets + ansible.builtin.command: + cmd: ss -ltn + register: listeners + changed_when: false + + - name: Gather service facts + ansible.builtin.service_facts: + + - name: Assert the deploy survived a metrics-probe timeout with the unit running + ansible.builtin.assert: + that: + # The marker is present => the stub was in late-bind mode, so the probe + # genuinely timed out rather than passing against a bound socket. + - no_metrics_marker.stat.exists + # Metrics never bound — the timeout path is what we exercised. + - "'127.0.0.1:9090' not in listeners.stdout" + - "'0.0.0.0:9090' not in listeners.stdout" + - "'[::]:9090' not in listeners.stdout" + - "'*:9090' not in listeners.stdout" + # The load-bearing liveness gate still holds: the unit is running. + - >- + 'decdn-node.service' in ansible_facts.services + and ansible_facts.services['decdn-node.service'].state == 'running' + fail_msg: >- + Expected a running decdn-node with NO metrics socket (advisory-timeout + path). Sockets: {{ listeners.stdout }} diff --git a/ansible/roles/decdn_node/README.md b/ansible/roles/decdn_node/README.md index 84892c3..4e6f82a 100644 --- a/ansible/roles/decdn_node/README.md +++ b/ansible/roles/decdn_node/README.md @@ -13,7 +13,8 @@ Per the deCDN node-onboarding ADR (019), a node only serves paid traffic after - **Ansible (this role): Phase 1 host prep + Phase 3 startup** — install binaries, create the `decdn` user + dirs, render `node.toml` + a hardened unit, open the - public QUIC port, start the daemon, wait for `/metrics`. + public QUIC port, start the daemon, and run a best-effort `/metrics` readiness + probe (see [Readiness](#readiness)). - **Operator (manual, NOT automated here):** Phase 1 **key material** (generate the node + eth keys) and Phase 2 **on-chain** (fund + stake the wallet, register the node). See "register the node" below — there is no turnkey CLI for it yet. @@ -112,6 +113,10 @@ Optional (omitted from `node.toml` unless set): (funding + on-chain staking/registration stay manual — see the eth-wallet step above). Leave `false` to keep the operator-provisioned posture (the fail-loud gate then requires you to provision the keystore, `node.secret`, and password yourself). +- `decdn_readiness_retries` (default `30`) + `decdn_readiness_delay` (default `2`, seconds) + — bound the `/metrics` readiness probe window (`retries × delay`, so ~60 s at these + defaults). A **timeout** only warns; it never fails the deploy — but a non-200 *answer* + does. See [Readiness](#readiness). Source contract addresses / chain-id from the deCDN contract deployment for your target chain, or the relevant ADR — never guess. See `roles/decdn_node/defaults/main.yml` for the @@ -125,6 +130,42 @@ full knob list and defaults. | metrics (9090) | TCP | 127.0.0.1 | no | | admin RPC (9191) | TCP | 127.0.0.1 | no | +## Readiness + +After starting the daemon the role runs a bounded probe against +`http://127.0.0.1:/metrics`, then a hard assert that the systemd +unit is in the `running` state. + +The probe is **advisory** on timeout: exhausting the window logs a warning but +does **not** fail the deploy. The daemon can bind its loopback metrics/admin +listeners well after process start on an otherwise-healthy node — they come up +behind the startup PaymentChannel buyer bootstrap (an upstream startup-ordering +issue tracked in `decdn/decdn`, not here), which can take many minutes. A short +probe that hard-failed would therefore false-fail a healthy deploy, and +stretching it to cover the worst case would hang every deploy for that whole +window — neither is acceptable, so a timeout warns and moves on. A metrics +endpoint that *answers* with a non-200 error is treated differently: that is a +fault, not slow startup, so it **fails** the deploy loudly. + +The follow-on assert that the `decdn-node.service` unit is `running` (via +`service_facts`) is a **backstop, not a full health check**. It catches a daemon +that exited/failed or a fast crash-loop that tripped systemd's start limit — but +the unit is `Type=simple`, so a wedged-but-alive or slow-crash-looping process +still reports `running`. A *persistent* probe timeout is therefore **not provably +benign**: confirm the node out-of-band before trusting it. Tune the probe window +with `decdn_readiness_retries` × `decdn_readiness_delay` (see +`defaults/main.yml`). When the probe times out, confirm readiness once the node +has settled: + +```bash +curl -s http://127.0.0.1:9090/metrics # loopback metrics (once bound) +journalctl -u decdn-node -e # look for "node runtime ready" +``` + +Note that neither check proves the node is serving **paid** traffic — that +additionally requires on-chain stake + registration (ADR 019 Phase 2, manual); +verify with `decdn node health`. + ## Files on the host - `/usr/local/bin/decdn-node`, `/usr/local/bin/decdn` — daemon + CLI diff --git a/ansible/roles/decdn_node/defaults/main.yml b/ansible/roles/decdn_node/defaults/main.yml index f5236b7..50ba679 100644 --- a/ansible/roles/decdn_node/defaults/main.yml +++ b/ansible/roles/decdn_node/defaults/main.yml @@ -83,6 +83,23 @@ decdn_metrics_port: 9090 decdn_metrics_bind: "127.0.0.1" # keep loopback — Prometheus scrape is a follow-up decdn_admin_port: 9191 +# --- Readiness probe (advisory) ----------------------------------------------- +# After start, the role probes http://127.0.0.1:/metrics as a +# best-effort readiness signal. A probe TIMEOUT only WARNS (it does NOT fail the +# deploy): the daemon can bind its metrics/admin listeners well after process +# start on an otherwise-healthy node (they come up behind the startup +# PaymentChannel buyer bootstrap — an upstream startup-ordering issue tracked in +# decdn/decdn, not here), so a bounded probe would otherwise false-fail a healthy +# deploy. A metrics endpoint that ANSWERS with a non-200 error, however, fails +# loud — that's a fault, not slow startup. The systemd service-state assert +# backstops this but is not a full health check: a Type=simple unit reports +# `running` even for a wedged-but-alive daemon, so it catches an exited/failed or +# fast-crash-looping unit, not a hung one. These two knobs bound the probe window +# (retries × delay; ≈60s at the values below) — enough for a node that binds +# promptly, without hanging the play on a slow-but-healthy startup. +decdn_readiness_retries: 30 # probe attempts before giving up (advisory) +decdn_readiness_delay: 2 # seconds between attempts + # --- Fixed on-host identity/paths --------------------------------------------- decdn_user: decdn decdn_group: decdn diff --git a/ansible/roles/decdn_node/tasks/main.yml b/ansible/roles/decdn_node/tasks/main.yml index 89d743b..d937783 100644 --- a/ansible/roles/decdn_node/tasks/main.yml +++ b/ansible/roles/decdn_node/tasks/main.yml @@ -384,18 +384,65 @@ state: started # --- Readiness ---------------------------------------------------------------- -- name: Wait for the node metrics endpoint +# ADVISORY probe: the daemon binds its loopback /metrics listener only after its +# startup bootstrap completes, which on a healthy node can lag process start by a +# long way (metrics/admin bind behind the PaymentChannel buyer bootstrap — an +# upstream startup-ordering issue tracked in decdn/decdn, not here). So a probe +# TIMEOUT must NOT fail the deploy — but a metrics endpoint that ANSWERS with an +# error is a real fault, not slow startup, so that still should. failed_when below +# encodes exactly that (connection-refused advisory, non-200 fails loud). The +# service-state assert further down is a necessary backstop but NOT sufficient: +# the unit is Type=simple, so a wedged-but-alive or slow-crash-looping daemon can +# still report `running` (see that task) — a persistent timeout here is therefore +# not provably benign. Tune the window with +# decdn_readiness_retries × decdn_readiness_delay (≈60s at the defaults). +- name: Wait for the node metrics endpoint (advisory) ansible.builtin.uri: url: "http://127.0.0.1:{{ decdn_metrics_port }}/metrics" status_code: 200 register: decdn_metrics_probe - retries: 30 - delay: 2 - until: decdn_metrics_probe.status == 200 + retries: "{{ decdn_readiness_retries }}" + delay: "{{ decdn_readiness_delay }}" + # default(-1): on connection-refused the uri module returns status -1 (and may + # omit the key). Retry only while the endpoint has NOT answered (status -1) — + # stop the moment it answers (any other status), so a real fault (e.g. a 500) + # surfaces immediately via failed_when rather than after the whole window. + until: decdn_metrics_probe.status | default(-1) != -1 changed_when: false - -# A stale/crash-looping process can answer /metrics, so also confirm THIS unit is -# running. FULL readiness (serving paid traffic) additionally requires on-chain + # Scope the suppression, don't blanket it: -1 (not bound yet) stays advisory and + # 200 is ready, but a bound endpoint answering any other status (e.g. a 500 from + # a broken /metrics handler) is a genuine fault — fail loud on it instead of + # masking it as slow startup (AGENTS.md hard-rule 4). + failed_when: decdn_metrics_probe.status | default(-1) not in [-1, 200] + # `--check` never starts the service, so the probe would be meaningless (and + # uri skips in check mode anyway, leaving no status for the warn below). + when: not ansible_check_mode + +- name: Warn if the metrics endpoint was not ready within the probe window + ansible.builtin.debug: + # Literal block so the curl/journalctl hints print on their own lines, + # copy-pasteable, rather than folded onto one. + msg: |- + decdn-node /metrics did not answer 200 within + ~{{ (decdn_readiness_retries | int) * (decdn_readiness_delay | int) }}s + ({{ decdn_readiness_retries }}×{{ decdn_readiness_delay }}), so readiness + could NOT be confirmed. This is OFTEN benign — a healthy node binds + metrics/admin late (behind the startup PaymentChannel bootstrap) and logs + "node runtime ready" once bound — but a PERSISTENT timeout can also mean a + wedged or mis-started daemon (a Type=simple unit still reports `running` + even when wedged). Verify before trusting the node: + curl -s http://127.0.0.1:{{ decdn_metrics_port }}/metrics + journalctl -u decdn-node -e (look for "node runtime ready") + when: + - not ansible_check_mode + - decdn_metrics_probe.status | default(-1) != 200 + +# A stale process can still answer /metrics, so also confirm THIS unit exists and +# is running. This is a backstop, not a full health check: it catches a daemon +# that exited/failed or a fast crash-loop that tripped systemd's start limit (unit +# state `failed`), but Type=simple reports `running` the instant the process +# forks, so a wedged-but-alive or slow-crash-looping daemon can still pass here. +# FULL readiness (serving paid traffic) additionally requires on-chain # registration — `decdn node health`. - name: Gather service facts ansible.builtin.service_facts: