Skip to content
Merged
66 changes: 25 additions & 41 deletions .github/workflows/build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -9,42 +9,47 @@ on:
- cron: '20 11 * * *'
- cron: '23 17 * * *'
workflow_dispatch:
inputs:
replay_closed_at:
description: >-
Backfill cases.closed_at by replaying every published release DB.
Off by default; needed once after the v2 schema lands, and again only
if the DB is ever restored from an older release.
type: boolean
default: false

permissions:
contents: write

# Guards against a manual dispatch landing on top of a scheduled run — two
# overlapping runs would silently discard one's closed_at observations. Queue
# rather than cancel: a cancelled run has already read the feed and would lose
# the same way. See notes/data-quality.md ("closed_at is a floor").
concurrency:
group: build-db
cancel-in-progress: false

jobs:
# A dispatch from a branch would publish that branch's DB as the latest
# release and push to main. Its own job, outside the build-db group, so a
# refused dispatch fails visibly without displacing a queued scheduled run.
refuse:
if: github.ref != 'refs/heads/main'
runs-on: ubuntu-latest
steps:
- run: |
echo "::error::Build DB publishes the latest release and pushes to main; dispatch it from main, not $GITHUB_REF"
exit 1

build:
if: github.ref == 'refs/heads/main'
runs-on: ubuntu-latest
# Guards against a manual dispatch landing on top of a scheduled run - two
# overlapping runs would silently discard one's closed_at observations. Queue
# rather than cancel: a cancelled run has already read the feed and would lose
# the same way. See notes/data-quality.md ("closed_at is a floor").
concurrency:
group: build-db
cancel-in-progress: false
steps:
- uses: actions/checkout@v7

- uses: astral-sh/setup-uv@v9.0.0
- uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
with:
# checked with `uv lock --check` on 0.12.18; bump deliberately, with uv.lock
version: "0.12.18"
enable-cache: true
prune-cache: true

- name: Install dependencies
run: uv sync
run: uv sync --locked

- name: Download existing DB
run: gh release download --pattern "uisce.db" --dir out/
run: scripts/fetch-db.sh
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}

Expand All @@ -53,22 +58,6 @@ jobs:
LOCATIONIQ_API_KEY: ${{ secrets.LOCATIONIQ_API_KEY }}
run: uv run uisce-pipeline

# Runs after the pipeline so the DB is already migrated to v2 and has this
# build's own transitions stamped — the replay never overwrites those, and
# they are the more precise value. The two together leave no gap: the
# replay covers snapshot-to-snapshot, the pipeline's upsert covers the
# last snapshot to now. Idempotent, so a repeat run is harmless.
- name: Backfill closed_at from published snapshots
if: ${{ inputs.replay_closed_at }}
run: |
mkdir -p snaps
for TAG in $(gh release list --limit 100 --json tagName --jq '.[].tagName'); do
gh release download "$TAG" --pattern uisce.db -O "snaps/$TAG.db" || true
done
uv run uisce-replay-closed-at --snapshots snaps --write
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}

# The rules answer the templated ~93% here; the abstentions wait for a
# local run with the LLM. See notes/rules-vs-llm-end-times.md.
- name: Infer end times by rules
Expand All @@ -78,12 +67,7 @@ jobs:
run: uv run uisce-build-inferred

- name: Publish DB
run: |
TAG="$(date +%Y-%m-%d)"
gh release create "$TAG" out/uisce.db \
--title "$TAG" \
--notes "Data refresh" \
|| gh release upload "$TAG" out/uisce.db --clobber
run: scripts/publish-db.sh out/uisce.db
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}

Expand Down
6 changes: 4 additions & 2 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -10,13 +10,15 @@ jobs:
steps:
- uses: actions/checkout@v7

- uses: astral-sh/setup-uv@v9.0.0
- uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
with:
# checked with `uv lock --check` on 0.12.18; bump deliberately, with uv.lock
version: "0.12.18"
enable-cache: true
prune-cache: true

- name: Install dependencies
run: uv sync --group dev
run: uv sync --locked --group dev

- name: Lint
run: uv run ruff check
Expand Down
13 changes: 9 additions & 4 deletions .github/workflows/pages.yml
Original file line number Diff line number Diff line change
Expand Up @@ -23,24 +23,29 @@ concurrency:

jobs:
build:
if: github.event_name != 'workflow_run' || github.event.workflow_run.conclusion == 'success'
if: >-
github.event_name != 'workflow_run' ||
(github.event.workflow_run.conclusion == 'success' &&
github.event.workflow_run.head_branch == 'main')
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v7

- uses: astral-sh/setup-uv@v9.0.0
- uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
with:
# checked with `uv lock --check` on 0.12.18; bump deliberately, with uv.lock
version: "0.12.18"
enable-cache: true
prune-cache: true

- name: Install dependencies
run: uv sync
run: uv sync --locked

# The site is a projection of the published DB, so a UI change deploys
# from the latest release without re-reading the feed. The freshness
# banner follows the DB, not this build's clock.
- name: Download the published DB
run: gh release download --pattern "uisce.db" --dir out/
run: scripts/fetch-db.sh
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}

Expand Down
2 changes: 2 additions & 0 deletions CLAUDE.md
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,8 @@ with the evidence that closed them.
| Duration outliers are categorical, not statistical. The 14-day cap is a backstop, not the outlier strategy. | data-quality.md — "Duration outliers are categorical" |
| "We are investigating" reference pairing works but rescues almost nothing — not worth building. | data-quality.md — "'We are investigating' notices" (corrected 2026-07-20) |
| `closed_at` is a floor: short-lived cases are never observed open. Twice-daily builds are the settled cadence. | data-quality.md — "`closed_at` is a floor" (re-measured 2026-07-31) |
| The `replay_closed_at` dispatch input is gone: a dry run over all 75 releases found 0 values to stamp or change. The script stays for a DB restored from an older release, run by hand. | data-quality.md - "The replay has nothing left to recover" (2026-09-24) |
| Every data build publishes its **own release**, tagged `YYYY-MM-DD-HHMM` (UTC); `scripts/publish-db.sh` and `scripts/fetch-db.sh` are the only publish and download. Replacing the asset in a per-day release was rejected: `--clobber` deletes first, and a rename swap stranded releases in three ways. | data-quality.md - "One release per build" (2026-09-24) |
| A case the feed drops while `Open` is stamped `vanished_at` (schema v4) and is closed with no signal on the site, never `closed_at`; the stamp touches closed rows too. It is safe only behind the feed-count guard (`FEED_COUNT_TOLERANCE`), which refuses a short download before anything touches the DB, and behind the empty-download refusal. Paging is by `OBJECTID`, refused unless each page is strictly ascending. | data-quality.md - "Cases that vanish from the feed" (2026-09-05, amended 2026-09-24) |
| A feature with no pin (no `geometry`, or `"NaN"`) keeps the pin the DB last stored, or is set aside with a `::warning::` until the feed pins it. Nullable coordinates were rejected: not an additive migration. | data-quality.md - "A feature with no pin" (2026-09-24) |
| **A case is open only while nothing its own text has ended.** `is_open(row, now)` reads `status`, `vanished_at` and a passed *observed* end, decided once in `resolve_case` and carried on `Case.is_open` for every surface that says open. The close date follows the same reading: the notice's own completion, else `closed_at` (`closed_on`, 2026-09-24). The feed closes a case a median 72h after the notice reports completion; 216 of 562 `Open` cases were past one, 0 of 7,667 completions were ever followed up. Scheduled ends do not close a case for display. | statuspage-methodology.md - "The notice's own completion closes it" (2026-09-05) |
Expand Down
9 changes: 6 additions & 3 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,12 +8,15 @@ The [website that this repo generates](https://baz8080.github.io/uisce/) is rebu

## Just want the data?

Grab the latest `uisce.db` from [releases](https://github.com/baz8080/uisce/releases) — no setup needed:
Grab the latest `uisce.db` from [releases](https://github.com/baz8080/uisce/releases) - no setup needed:

```
gh release download --clobber --pattern "uisce.db" --dir out/
scripts/fetch-db.sh
```

It wraps `gh release download --pattern uisce.db`; every data build publishes its own
release, so the latest one always holds a complete DB.

Tables:

* `cases` — one row per published notice pin (title, description, dates, status, impact flags, WGS84 coordinates). `work_category` is a slug normalised from the title (`burst_main`, `essential_works`, …); `work_type` (Planned/Unplanned) is taken from the feed but overridden for categories where the label is unambiguous (a burst main is never planned).
Expand Down Expand Up @@ -97,7 +100,7 @@ CI runs the rules half on every data build (`uisce-infer --rules-only`) and comm

1. `git pull` — CI appends to `data/inferred_end_times.jsonl`; `.gitattributes` merges a concurrent local append rather than conflicting
2. Start the LLM server on :1234
3. `gh release download --clobber --pattern "uisce.db" --dir out/`
3. `scripts/fetch-db.sh`
4. `uv run uisce-infer` — appends results to `data/inferred_end_times.jsonl` (committed to the repo; only new/changed descriptions are processed); commit and push
5. (Local check only — CI rebuilds the table itself) `uv run uisce-build-inferred`

Expand Down
18 changes: 18 additions & 0 deletions notes/data-quality.md
Original file line number Diff line number Diff line change
Expand Up @@ -253,6 +253,24 @@ The 12% figure above is stale, and the paragraph's implied remedy — shrink the

If a closure *series* is ever published (month-over-month counts, or a time-to-close metric keyed on `closed_at`), this stops being a prose caveat and needs the cadence recorded alongside the data so the series can be corrected rather than annotated.

### The replay has nothing left to recover (2026-09-24)

A dry run of `uisce-replay-closed-at` over all 75 release snapshots (2026-06-30 to 2026-09-23) against a copy of the 2026-09-23 release finds 6,961 transitions, and every one of those cases already carries a `closed_at` on the same date as the replayed tag: **0 rows to stamp, and 0 that would change** even if the replay were allowed to overwrite. None of the 5,872 closed cases with a NULL `closed_at` appears in the replay at all; they closed before the first snapshot or were never seen `Open`. Since the v2 schema landed, the live upsert has stamped every transition a snapshot can see, so the `replay_closed_at` dispatch input was dropped from Build DB (owner, 2026-09-24). The script stays for the one case it still serves, a DB restored from an older release, and is run by hand as its docstring shows.

### One release per build (2026-09-24)

Until this date each day had one release, and the second build of the day replaced its
`uisce.db` with `gh release upload --clobber`, which deletes the old asset before uploading
the new one: a failure in between left the day's release with no DB, and every later build
and Pages deploy, which all start from the latest release, failed until someone fixed it by
hand. A same-day swap (upload as `uisce.db.next`, then delete and rename) was built and
rejected on two rounds of review: each round found another state it could strand, all of
them coming from juggling two names inside one release. From this date every build publishes
its own release, tagged `YYYY-MM-DD-HHMM` (UTC); `gh release create` makes a draft, uploads,
then publishes, so the latest release always holds a complete `uisce.db`. About two releases
a day instead of one. `uisce-replay-closed-at` reads the date from the tag's first ten
characters, so the daily and per-build names replay alike.

### Twice-daily builds: why, and why not three (2026-07-31)

The second daily build slot exists for publication latency, not to sharpen `closed_at` (see above — past a daily cadence, Uisce Éireann's own administrative lag dominates, not the build gap). Notices publish between 07:00 and 16:00 UTC (staffed office hours), so a second build only helps if it lands inside that window: measured over 8,135 cases, a single evening build leaves a mean **7.7h** from publication to the site, a midday build halves that to **3.9h**, and an overnight build would only have bought **0.9h**. A third build takes 3.9h to 3.5h — not worth the run.
Expand Down
2 changes: 1 addition & 1 deletion notes/pipeline-dependencies.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
`uisce-build-inferred` (`src/uisce/build.py`) checks for this up front and fails with a clear message naming the missing case_id range, rather than a raw `sqlite3.IntegrityError`. The fix is always the same: get a DB that's at least as new as whatever the inference run used, e.g.:

```
gh release download --pattern uisce.db --dir out/ --clobber
scripts/fetch-db.sh
```

(defaults to the latest release; pass a specific tag if you know which one you need). There's no automatic reconciliation here on purpose — the inference run itself doesn't record which DB snapshot it used (see the description-hash discussion elsewhere in this repo's history for why the hash alone is enough for correctness, just not for provenance), so "grab the latest release" is the practical default rather than something that could be automated reliably.
Expand Down
9 changes: 9 additions & 0 deletions scripts/fetch-db.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
#!/usr/bin/env bash
# Download a release's uisce.db, the latest release unless TAG is given.
# usage: scripts/fetch-db.sh [TAG] [OUT]
set -euo pipefail

tag="${1:-}"
out="${2:-out/uisce.db}"
mkdir -p "$(dirname "$out")"
gh release download ${tag:+"$tag"} --pattern uisce.db -O "$out" --clobber
15 changes: 15 additions & 0 deletions scripts/publish-db.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
#!/usr/bin/env bash
# Publish DB as a release of its own, tagged with the build's UTC date and time.
# gh creates the release as a draft, uploads, then publishes it, so the latest
# release always holds a complete uisce.db. usage: scripts/publish-db.sh [DB]
set -euo pipefail

db="${1:-out/uisce.db}"
tag="$(date -u +%Y-%m-%d-%H%M)"
# the asset takes the file's name, and fetch-db.sh asks for uisce.db
if [ "$(basename "$db")" != uisce.db ]; then
staged="$(mktemp -d)/uisce.db"
cp "$db" "$staged"
db="$staged"
fi
gh release create "$tag" "$db" --title "$tag" --notes "Data refresh" --latest
2 changes: 1 addition & 1 deletion src/uisce/build.py
Original file line number Diff line number Diff line change
Expand Up @@ -205,7 +205,7 @@ def check_cases_cover(conn, case_ids):
f"{len(missing)} case_id(s) in {JSONL_PATH} are not present in {DB_PATH} "
f"(range {missing[0]}-{missing[-1]}). The local DB is likely older than "
"whatever DB the inference run used. Refresh it first, e.g.:\n"
" gh release download --pattern uisce.db --dir out/ --clobber"
" scripts/fetch-db.sh"
)


Expand Down
19 changes: 10 additions & 9 deletions src/uisce/replay_closed_at.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,8 @@

Download the snapshots first; they are ~10-20MB each:

for T in $(gh release list --limit 100 | cut -f1); do
gh release download "$T" --pattern uisce.db -O "snaps/$T.db"
for T in $(gh release list --limit 1000 --exclude-drafts --json tagName --jq '.[].tagName'); do
scripts/fetch-db.sh "$T" "snaps/$T.db"
done
"""

Expand All @@ -21,17 +21,18 @@

from uisce.config import DB_PATH

# Snapshot files are named for their release tag (YYYY-MM-DD.db), which is also
# the value written to closed_at, so replayed rows carry the date of the build
# Snapshot files are named for their release tag: YYYY-MM-DD for the daily
# releases, YYYY-MM-DD-HHMM for the per-build ones from 2026-09-24. The date part
# is the value written to closed_at, so replayed rows carry the date of the build
# that observed the closure.
SNAPSHOT_GLOB = "*.db"


def snapshot_files(directory):
"""Snapshot paths in tag order. Names are ISO dates, so lexical sort is
chronological; anything else is a caller error rather than something to
guess at."""
paths = sorted(Path(directory).glob(SNAPSHOT_GLOB))
"""Snapshot paths in tag order. Tags are ISO dates, with a time on the
per-build ones, so sorting the tags is chronological; the file names are not
("-" sorts before ".db"). Anything else is a caller error, not a guess."""
paths = sorted(Path(directory).glob(SNAPSHOT_GLOB), key=lambda p: p.stem)
if not paths:
raise SystemExit(f"No {SNAPSHOT_GLOB} snapshots in {directory}")
return [(p.stem, p) for p in paths]
Expand All @@ -58,7 +59,7 @@ def replay(snapshots):
seen_open.add(case_id)
closed_at.pop(case_id, None)
elif case_id in seen_open and case_id not in closed_at:
closed_at[case_id] = tag
closed_at[case_id] = tag[:10]
return closed_at


Expand Down
8 changes: 8 additions & 0 deletions tests/test_replay_closed_at.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,14 @@ def test_stamps_the_first_snapshot_observing_the_close(self, tmp_path):
# the later snapshots must not drag the stamp forward
assert replay(snaps) == {1: "2026-07-08"}

def test_a_per_build_tag_stamps_its_date(self, tmp_path):
snaps = [
_snapshot(tmp_path, "2026-09-24", [(1, "Open")]),
_snapshot(tmp_path, "2026-09-24-1845", [(1, "Closed")]),
]
assert [tag for tag, _ in snapshot_files(tmp_path)] == ["2026-09-24", "2026-09-24-1845"]
assert replay(snaps) == {1: "2026-09-24"}

def test_case_never_seen_open_is_not_stamped(self, tmp_path):
# Created and closed inside one gap: no transition was ever observed, so
# there is nothing to recover. ~12% of new cases look like this.
Expand Down
Loading