From cff582093103a58f7f6512452354e11793277b5b Mon Sep 17 00:00:00 2001 From: 0xb10c Date: Fri, 25 Sep 2026 02:41:45 +0200 Subject: [PATCH 1/2] feat(ci): render the stale block header tree with fork-observer The deploy workflow now renders a header tree of every stale block with a known header next to the main chain. It seeds a fork-observer database with the main chain headers from block-dn.org and the stale headers from stale-blocks.csv, starts fork-observer on it, and exports the tree into site/tree for the dashboard to embed. The binary is the fork-observer package of the 0xb10c/nix repository, fetched from its cachix cache. The scripts and styles are those of fork-observer's main branch. The page drawing the tree is our own: it provides the little that fork-observer's main.js does on top of blocktree.js on a live instance, and leaves out everything a snapshot doesn't need. A made-up "stale-blocks dataset" node marks the tips of the stale branches, since no real node reports them, with the dataset's own statuses: full-block and header-only, for what the dataset holds of the block. Stale headers whose parent is unknown can't be placed in the tree and are skipped. --- .github/workflows/deploy.yaml | 33 ++++++ .gitignore | 3 + ci/fork-observer-db.py | 205 ++++++++++++++++++++++++++++++++++ ci/fork-observer-site.py | 165 +++++++++++++++++++++++++++ ci/fork-observer/README.md | 66 +++++++++++ ci/fork-observer/config.toml | 37 ++++++ ci/fork-observer/tree.html | 148 ++++++++++++++++++++++++ 7 files changed, 657 insertions(+) create mode 100644 ci/fork-observer-db.py create mode 100644 ci/fork-observer-site.py create mode 100644 ci/fork-observer/README.md create mode 100644 ci/fork-observer/config.toml create mode 100644 ci/fork-observer/tree.html diff --git a/.github/workflows/deploy.yaml b/.github/workflows/deploy.yaml index 970c167..2d568b5 100644 --- a/.github/workflows/deploy.yaml +++ b/.github/workflows/deploy.yaml @@ -22,6 +22,39 @@ jobs: steps: - uses: actions/checkout@v6 - run: python3 ci/generate-website.py + + # The header tree on the dashboard is rendered by a fork-observer + # instance that runs only for this job: its database is built from + # block-dn.org and the dataset, and its tree is exported as static + # files. See ci/fork-observer/README.md. + - name: Install nix + uses: cachix/install-nix-action@v31 + with: + # The 0xb10c/nix CI builds its packages against this channel and + # pushes them to its cachix cache. Using the same nixpkgs makes this + # a download instead of a build. + nix_path: nixpkgs=channel:nixos-26.05 + - name: Set up the b10c-nixpkgs cachix cache + uses: cachix/cachix-action@v17 + with: + name: b10c-nixpkgs + - name: Build fork-observer + run: nix-build https://github.com/0xb10c/nix/archive/master.tar.gz -A fork-observer -o fork-observer + + - name: Build the fork-observer database + run: python3 ci/fork-observer-db.py fork-observer-db/fork-observer.sqlite + - name: Start fork-observer + run: | + CONFIG_FILE=ci/fork-observer/config.toml RUST_LOG=info \ + nohup ./fork-observer/bin/fork-observer > fork-observer.log 2>&1 & + - name: Export the header tree + run: python3 ci/fork-observer-site.py + - name: Stop fork-observer + if: always() + run: | + cat fork-observer.log || true + pkill fork-observer || true + - uses: actions/upload-pages-artifact@v4 with: path: site diff --git a/.gitignore b/.gitignore index 45ddf0a..1f38b9b 100644 --- a/.gitignore +++ b/.gitignore @@ -1 +1,4 @@ site/ +/fork-observer-db/ +/fork-observer +__pycache__/ diff --git a/ci/fork-observer-db.py b/ci/fork-observer-db.py new file mode 100644 index 0000000..25ff2d6 --- /dev/null +++ b/ci/fork-observer-db.py @@ -0,0 +1,205 @@ +#!/usr/bin/env python3 + +# Builds the SQLite database that a fork-observer instance renders the stale +# block header tree from. The main chain headers come from block-dn.org, the +# stale headers from stale-blocks.csv. The result is the `headers` table +# fork-observer itself would have written, so it can start from it directly: +# +# https://github.com/0xB10C/fork-observer/blob/main/src/db.rs +# +# Both have to be in the database before fork-observer starts. It links a +# header to its parent only when the parent is already known, so stale headers +# loaded before the main chain would stay disconnected and be drawn in the +# wrong place. +# +# Usage: fork-observer-db.py [--cache-dir DIR] +# +# --cache-dir keeps the downloaded block-dn header files around for repeated +# local runs. The CI doesn't need it. + +import argparse +import csv +import hashlib +import json +import sqlite3 +import sys +from pathlib import Path +from urllib.request import Request, urlopen + +REPO_ROOT = Path(__file__).resolve().parent.parent +CSV_PATH = REPO_ROOT / "stale-blocks.csv" + +BLOCK_DN_URL = "https://block-dn.org" +# block-dn serves its headers in files of 100'000 headers, 80 bytes each. The +# public instances all use this size (see /status: entries_per_header_file). +HEADERS_PER_FILE = 100_000 +HEADER_SIZE = 80 +TIMEOUT_S = 120 +USER_AGENT = "stale-blocks-ci (https://github.com/bitcoin-data/stale-blocks)" + +GENESIS_HASH = "000000000019d6689c085ae165831e934ff763ae46a2a6c172b3f1b60a8ce26f" + +# The id fork-observer's config gives the network. Must match +# ci/fork-observer/config.toml. +NETWORK_ID = 1 + +# Identical to fork-observer's CREATE_STMT_TABLE_HEADERS. The hash and header +# columns hold hex strings, as fork-observer writes them. +CREATE_TABLE = """ +CREATE TABLE IF NOT EXISTS headers ( + height INT, + network INT, + hash BLOB, + header BLOB, + miner TEXT, + PRIMARY KEY (network, hash, header) +) +""" + + +def block_hash(header): + return hashlib.sha256(hashlib.sha256(header).digest()).digest()[::-1].hex() + + +def prev_hash(header): + return header[4:36][::-1].hex() + + +def fetch(path): + # block-dn is behind Cloudflare, which rejects the default Python user agent. + request = Request(f"{BLOCK_DN_URL}{path}", headers={"User-Agent": USER_AGENT}) + with urlopen(request, timeout=TIMEOUT_S) as response: + return response.read() + + +def fetch_header_file(start, cache_dir, sealed): + # Only sealed files (100'000 headers) are cached: the last file still grows + # with every block. + cached = cache_dir / f"headers-{start}.bin" if cache_dir and sealed else None + if cached and cached.exists(): + return cached.read_bytes() + + print(f"Fetching {BLOCK_DN_URL}/headers/{start}..") + data = fetch(f"/headers/{start}") + if len(data) % HEADER_SIZE != 0: + sys.exit(f"headers file {start} has {len(data)} bytes, not a multiple of {HEADER_SIZE}") + if sealed and len(data) != HEADERS_PER_FILE * HEADER_SIZE: + sys.exit(f"headers file {start} holds {len(data) // HEADER_SIZE} headers, expected {HEADERS_PER_FILE}") + + if cached: + cache_dir.mkdir(parents=True, exist_ok=True) + cached.write_bytes(data) + return data + + +# Downloads the main chain headers from block-dn and checks that they link up +# to the genesis block. block-dn is an untrusted source, so a download that +# doesn't form a chain from genesis to the tip it reports is refused. +def main_chain_headers(cache_dir): + status = json.loads(fetch("/status")) + tip_height = status["best_block_height"] + tip_hash = status["best_block_hash"] + if status.get("entries_per_header_file", HEADERS_PER_FILE) != HEADERS_PER_FILE: + sys.exit(f"block-dn uses {status['entries_per_header_file']} headers per file, expected {HEADERS_PER_FILE}") + print(f"block-dn tip: {tip_height} {tip_hash}") + + headers = [] # (height, hash, header hex) + expected_prev = "00" * 32 + last_file_start = tip_height - tip_height % HEADERS_PER_FILE + for start in range(0, last_file_start + 1, HEADERS_PER_FILE): + data = fetch_header_file(start, cache_dir, sealed=start < last_file_start) + for i in range(0, len(data), HEADER_SIZE): + header = data[i:i + HEADER_SIZE] + if prev_hash(header) != expected_prev: + sys.exit(f"header at height {start + i // HEADER_SIZE} doesn't build on the previous one") + expected_prev = block_hash(header) + headers.append((start + i // HEADER_SIZE, expected_prev, header.hex())) + + if headers[0][1] != GENESIS_HASH: + sys.exit(f"first header is {headers[0][1]}, not the genesis block") + # The tip can move between the /status request and the file downloads, so + # the chain may end a few blocks past the reported tip. It must not end + # before it, and the reported tip must be in it. + if headers[-1][0] < tip_height or headers[tip_height][1] != tip_hash: + sys.exit(f"downloaded chain ends at {headers[-1][0]}, doesn't contain the reported tip {tip_height} {tip_hash}") + + print(f"Got {len(headers)} main chain headers up to height {headers[-1][0]}") + return headers + + +# The stale headers from the CSV that can be placed in the tree: their hash has +# to match the header, the parent has to be known (main chain or another stale +# header) and the height has to be the parent's plus one. Rows without a header +# can't be placed at all. +def stale_headers(known): + with open(CSV_PATH, newline="") as f: + rows = [r for r in csv.DictReader(f) if r["header"]] + + headers = [] + pending = [] + for row in rows: + header = bytes.fromhex(row["header"]) + if block_hash(header) != row["hash"]: + print(f"Skipping {row['height']} {row['hash']}: header doesn't hash to the block hash") + continue + if row["hash"] in known: + print(f"Skipping {row['height']} {row['hash']}: is on the main chain") + continue + pending.append((int(row["height"]), row["hash"], header)) + + # A stale header's parent can be another stale header further down the + # CSV, so keep going until nothing new can be placed. + placed = True + while placed and pending: + placed = False + remaining = [] + for height, hash, header in pending: + parent_height = known.get(prev_hash(header)) + if parent_height is None: + remaining.append((height, hash, header)) + continue + if parent_height + 1 != height: + print(f"Skipping {height} {hash}: parent is at height {parent_height}") + continue + known[hash] = height + headers.append((height, hash, header.hex())) + placed = True + pending = remaining + + for height, hash, header in pending: + print(f"Skipping {height} {hash}: parent {prev_hash(header)} is unknown") + + print(f"Got {len(headers)} stale headers from {CSV_PATH.name} ({len(rows) - len(headers)} skipped)") + return headers + + +def write_db(path, headers): + path.parent.mkdir(parents=True, exist_ok=True) + if path.exists(): + path.unlink() + db = sqlite3.connect(path) + db.execute(CREATE_TABLE) + db.executemany( + "INSERT OR IGNORE INTO headers (height, network, hash, header, miner) VALUES (?, ?, ?, ?, ?)", + ((height, NETWORK_ID, hash, header, "") for height, hash, header in headers), + ) + db.commit() + db.close() + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("output", type=Path) + parser.add_argument("--cache-dir", type=Path) + args = parser.parse_args() + + main_chain = main_chain_headers(args.cache_dir) + known = {hash: height for height, hash, _ in main_chain} + stale = stale_headers(known) + + write_db(args.output, main_chain + stale) + print(f"Wrote {len(main_chain) + len(stale)} headers to {args.output}") + + +if __name__ == "__main__": + main() diff --git a/ci/fork-observer-site.py b/ci/fork-observer-site.py new file mode 100644 index 0000000..685b621 --- /dev/null +++ b/ci/fork-observer-site.py @@ -0,0 +1,165 @@ +#!/usr/bin/env python3 + +# Exports the header tree of a running fork-observer instance into site/tree, +# which the dashboard embeds. Started from the database ci/fork-observer-db.py +# builds, the instance's tree holds every stale block of the dataset next to +# the main chain. site/tree gets: +# +# index.html ci/fork-observer/tree.html, the page drawing the tree +# data.json the instance's api//data.json response +# static/ fork-observer's css, js and img directories +# +# The scripts, styles and images are those of fork-observer's main branch, +# fetched from GitHub, so they don't go stale. +# +# fork-observer only marks blocks that a node reports as a chain tip. No node +# reports the stale blocks, so a made-up node is added to the response with +# every stale branch's tip. The frontend then draws them with a tip status. +# Instead of the statuses of Bitcoin Core's getchaintips, which fork-observer +# uses, these are the dataset's own: full-block where the dataset holds the +# full block and header-only where it holds only the header. Neither says +# anything about validity. ci/fork-observer/tree.html supplies their colours. + +import csv +import io +import json +import os +import shutil +import sys +import tarfile +import time +from pathlib import Path +from urllib.error import URLError +from urllib.request import Request, urlopen + +REPO_ROOT = Path(__file__).resolve().parent.parent +CSV_PATH = REPO_ROOT / "stale-blocks.csv" +BLOCKS_DIR = REPO_ROOT / "blocks" +TREE_HTML_PATH = REPO_ROOT / "ci" / "fork-observer" / "tree.html" +OUT_DIR = REPO_ROOT / "site" / "tree" + +FORK_OBSERVER_REPO = "0xB10C/fork-observer" +FORK_OBSERVER_BRANCH = "main" +USER_AGENT = "stale-blocks-ci (https://github.com/bitcoin-data/stale-blocks)" + +FORK_OBSERVER_URL = os.environ.get("FORK_OBSERVER_URL", "http://127.0.0.1:2323") +# Must match ci/fork-observer/config.toml. +NETWORK_ID = 1 +# The instance polls its block-dn node once per query_interval (15s) after a +# short initial delay. Loading the tree from the database takes a while too. +READY_TIMEOUT_S = 600 +TIMEOUT_S = 120 + + +def fetch(url): + request = Request(url, headers={"User-Agent": USER_AGENT}) + with urlopen(request, timeout=TIMEOUT_S) as response: + return response.read() + + +# The header tree is only complete once the block-dn node reported its tip and +# fork-observer fetched the blocks found since the database was built. The +# node's tips show up in the response a moment before the tree is updated, so +# the tip block has to be in the tree as well. +def wait_for_data(): + deadline = time.monotonic() + READY_TIMEOUT_S + while time.monotonic() < deadline: + try: + data = json.loads(fetch(f"{FORK_OBSERVER_URL}/api/{NETWORK_ID}/data.json")) + in_tree = {h["hash"] for h in data["header_infos"]} + tips = [tip for node in data["nodes"] for tip in node["tips"]] + if tips and all(tip["hash"] in in_tree for tip in tips): + return data + print("fork-observer is up, waiting for its node's tip to be in the tree..") + except (URLError, TimeoutError, ConnectionError): + print("waiting for fork-observer..") + time.sleep(5) + sys.exit(f"fork-observer at {FORK_OBSERVER_URL} didn't become ready within {READY_TIMEOUT_S}s") + + +# The tips of the stale branches: stale blocks no other stale block builds on. +# Only blocks that are in the response can be marked, so the ones whose header +# couldn't be placed in the tree (see ci/fork-observer-db.py) are left out. +def stale_tips(header_infos): + with open(CSV_PATH, newline="") as f: + rows = [r for r in csv.DictReader(f) if r["header"]] + + in_tree = {h["hash"]: h["height"] for h in header_infos} + built_on = {bytes.fromhex(r["header"])[4:36][::-1].hex() for r in rows} + + tips = [] + for r in rows: + if r["hash"] not in in_tree or r["hash"] in built_on: + continue + has_block = (BLOCKS_DIR / f"{r['height']}-{r['hash']}.bin").exists() + tips.append({ + "hash": r["hash"], + "status": "full-block" if has_block else "header-only", + "height": in_tree[r["hash"]], + }) + return tips + + +def dataset_node(header_infos): + tips = stale_tips(header_infos) + print(f"Marking {len(tips)} stale branch tips") + return { + "id": 1000, + "name": "stale-blocks dataset", + "description": "", + "implementation": "", + "tips": tips, + "last_changed_timestamp": 0, + "version": "", + "reachable": True, + } + + +# The css, js and img directories of fork-observer's main branch as +# {path: bytes}, with paths relative to www. +def fetch_frontend(): + print(f"Fetching the fork-observer frontend from the {FORK_OBSERVER_BRANCH} branch..") + archive = fetch(f"https://github.com/{FORK_OBSERVER_REPO}/archive/refs/heads/{FORK_OBSERVER_BRANCH}.tar.gz") + files = {} + with tarfile.open(fileobj=io.BytesIO(archive), mode="r:gz") as tar: + for member in tar.getmembers(): + # The archive's top-level directory is named after the repository + # and branch. + parts = Path(member.name).parts[1:] + if len(parts) < 3 or parts[0] != "www" or parts[1] not in ("css", "js", "img"): + continue + if member.isfile(): + files[str(Path(*parts[1:]))] = tar.extractfile(member).read() + + for path in ("css/bootstrap.min.css", "css/style.css", "js/d3.v7.min.js", "js/blocktree.js"): + if path not in files: + sys.exit(f"the fork-observer archive doesn't contain www/{path}") + return files + + +def write_site(frontend, data): + if OUT_DIR.exists(): + shutil.rmtree(OUT_DIR) + OUT_DIR.mkdir(parents=True) + + shutil.copy(TREE_HTML_PATH, OUT_DIR / "index.html") + (OUT_DIR / "data.json").write_text(json.dumps(data)) + for path, content in frontend.items(): + target = OUT_DIR / "static" / path + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(content) + + +def main(): + frontend = fetch_frontend() + + data = wait_for_data() + print(f"Got {len(data['header_infos'])} headers from {FORK_OBSERVER_URL}") + data["nodes"].append(dataset_node(data["header_infos"])) + + write_site(frontend, data) + print(f"Generated {OUT_DIR.relative_to(REPO_ROOT)}") + + +if __name__ == "__main__": + main() diff --git a/ci/fork-observer/README.md b/ci/fork-observer/README.md new file mode 100644 index 0000000..fc0684a --- /dev/null +++ b/ci/fork-observer/README.md @@ -0,0 +1,66 @@ +# fork-observer + +The dashboard's header tree is drawn with +[fork-observer](https://github.com/0xB10C/fork-observer). This directory holds +what that takes. + +The deploy workflow builds a fork-observer database from the main chain headers +of block-dn.org and the stale headers of `stale-blocks.csv` +(`ci/fork-observer-db.py`), starts a fork-observer instance on it with +`config.toml`, and exports the instance's tree into `site/tree` +(`ci/fork-observer-site.py`), which the dashboard embeds as an iframe. + +## Where fork-observer comes from + +Nothing is vendored, so the tree follows fork-observer's development: + +- The binary is the `fork-observer` package of the + [0xb10c/nix](https://github.com/0xb10c/nix) repository's master branch. That + repository's CI builds its packages against the `nixos-26.05` channel and + pushes them to the `b10c-nixpkgs` cachix cache, so the workflow uses the + same channel and gets a cached build instead of compiling. When the channel + moved since that CI last ran, the package is built from source, which takes + a few minutes longer. +- The scripts, styles and images are those of fork-observer's `main` branch, + fetched from GitHub by `ci/fork-observer-site.py`. + +The package is bumped to new fork-observer commits by a bot, so it can lag +behind the frontend by a few days. If a change to the API response format and +the frontend lands in between, the tree breaks until the package catches up. + +## The page + +`tree.html` is the document the dashboard embeds. It's separate from the +dashboard because fork-observer's stylesheets restyle the whole page. It loads +fork-observer's `blocktree.js`, which draws the tree, and does the little that +fork-observer's `main.js` does on top of that on a live instance: define the +globals `blocktree.js` reads, load the data, draw once. Everything a live +instance has beyond the tree, like the node table, live updates or the mining +jobs feed, is left out. + +The stale blocks are marked with two tip statuses of the dataset's own, +`full-block` and `header-only`, instead of the getchaintips statuses +fork-observer uses. `tree.html` supplies their colours, and it rewrites the +tip labels after every draw to drop the "1x " count fork-observer puts in +front of the status, which is meaningless with a single source. The "active" +tip is the main chain tip when the page was generated. A block's info card +offers stale blocks for download from the instance's API; `tree.html` points +that link at the full block in this repository instead, or drops it where the +dataset has only the header. + +If a fork-observer change makes `blocktree.js` expect something new from +`main.js`, or renames what `tree.html` hooks into (`draw`, `onBlockClick`, +`recalc_tip_boxes`, `descLayer`, the `.tip-info-row` labels and the info +card's download links), `tree.html` has to follow. The browser console shows +the resulting error. + +## Running it locally + +```bash +nix-build https://github.com/0xb10c/nix/archive/master.tar.gz -A fork-observer -o fork-observer +python3 ci/fork-observer-db.py fork-observer-db/fork-observer.sqlite +CONFIG_FILE=ci/fork-observer/config.toml ./fork-observer/bin/fork-observer & +python3 ci/fork-observer-site.py +python3 ci/generate-website.py +python3 -m http.server --directory site +``` diff --git a/ci/fork-observer/config.toml b/ci/fork-observer/config.toml new file mode 100644 index 0000000..d420ae4 --- /dev/null +++ b/ci/fork-observer/config.toml @@ -0,0 +1,37 @@ +# fork-observer configuration for rendering the stale-block header tree in CI. +# See ci/fork-observer/README.md for how it's used. +# +# Paths are relative to the repository root, where the deploy workflow runs +# fork-observer. The database is built by ci/fork-observer-db.py. The www +# directory is only served by fork-observer itself and never fetched here: +# ci/fork-observer-site.py deploys the frontend from fork-observer's main branch. +database_path = "fork-observer-db/fork-observer.sqlite" +www_path = "fork-observer/www" + +query_interval = 15 +address = "127.0.0.1:2323" +footer_html = "" + +[[networks]] +id = 1 +name = "Mainnet" +description = "Every stale block in the stale-blocks dataset with a known header, next to the main chain. Blocks between two stale blocks are hidden." +min_fork_height = 0 +# Every height with a stale block is 'interesting' and kept in the response. +# The dataset has a few thousand of those, so this has to be well above that. +max_interesting_heights = 10000 + + # Pool identification fetches full blocks to read their coinbase. block-dn + # can't serve stale blocks, so it'd only produce errors. + [networks.pool_identification] + enable = false + + # The block-dn node only provides the active chain tip. The headers are + # already in the database, so this fetches at most the few blocks found + # since the database was built. + [[networks.nodes]] + id = 0 + name = "block-dn.org" + description = "The active chain tip as seen by the public block-dn instance the main chain headers were fetched from." + rpc_host = "https://block-dn.org" + implementation = "block-dn" diff --git a/ci/fork-observer/tree.html b/ci/fork-observer/tree.html new file mode 100644 index 0000000..be3d7a0 --- /dev/null +++ b/ci/fork-observer/tree.html @@ -0,0 +1,148 @@ + + + + + + + + + + + + + +
+
+
+ +
+ + + + +
+ + activemain chain tip when this page was generated + full-blockstale block, full block in the dataset + header-onlystale block, only the header in the dataset + +
+
+ + + + + + + From e7981e4514c62857c159ccd20f20f17d425a5678 Mon Sep 17 00:00:00 2001 From: 0xb10c Date: Fri, 25 Sep 2026 02:41:45 +0200 Subject: [PATCH 2/2] feat(website): embed the header tree in the dashboard Between the stale block rate chart and the list of blocks. --- README.md | 7 +++++-- ci/template.html | 12 ++++++++++++ 2 files changed, 17 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 0061a7c..2696b07 100644 --- a/README.md +++ b/README.md @@ -14,8 +14,11 @@ so passing them does not establish full consensus validity. The dashboard at [bitcoin-data.github.io/stale-blocks](https://bitcoin-data.github.io/stale-blocks/) -shows the current contents of the dataset: the stale-block rate over time and -every stale block with its decoded header. +shows the current contents of the dataset: the stale-block rate over time, a +header tree of every stale block with a known header next to the main chain +(drawn with [fork-observer](https://github.com/0xB10C/fork-observer), see +[`ci/fork-observer`](./ci/fork-observer)), and every stale block with its +decoded header. The data itself lives in two places: diff --git a/ci/template.html b/ci/template.html index b16f8fc..8da2483 100644 --- a/ci/template.html +++ b/ci/template.html @@ -111,6 +111,18 @@

Stale Block Rate

})(); +
+

Header Tree

+
+ +
+

+ Every stale block with a known header, next to the main chain. Blocks between two stale blocks are hidden. + Drag to pan, scroll to zoom, click a block for its header. Drawn with + fork-observer. +

+
+