From 09907910872409e0615b96408ea2e690d52e42e4 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 08:26:47 +0000 Subject: [PATCH 1/7] Run the data build from main only, sync from the lock, pin setup-uv Build DB could be dispatched from any branch, and a branch run publishes its DB as the latest release and pushes to main. Its first step now fails the job with an error unless the ref is refs/heads/main: a failure shows up in the Actions list, a skipped job would not. Build site's workflow_run trigger also requires head_branch == main, so a branch run of Build DB can never deploy the site. uv sync takes --locked in all three workflows. statusui is a git dependency with no rev in pyproject.toml, so uv.lock is its only pin; a pyproject/uv.lock mismatch now fails the job instead of re-resolving statusui to whatever its default branch holds. uv lock --check passes on this commit. astral-sh/setup-uv is pinned to the commit of v9.0.0 (a lightweight tag, resolved with git ls-remote). It is the only third-party action; the actions/* ones are left on their tags. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0168hLW2X3mkV3Jhm26LJSwQ --- .github/workflows/build.yml | 12 ++++++++++-- .github/workflows/ci.yml | 4 ++-- .github/workflows/pages.yml | 9 ++++++--- 3 files changed, 18 insertions(+), 7 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 73c0161..a20a14b 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -33,15 +33,23 @@ jobs: build: runs-on: ubuntu-latest steps: + # A dispatch from a branch would publish that branch's DB as the latest + # release and push to main. Failed rather than skipped, so it is seen. + - name: Refuse to run off main + if: github.ref != 'refs/heads/main' + run: | + echo "::error::Build DB publishes the latest release and pushes to main; dispatch it from main, not $GITHUB_REF" + exit 1 + - uses: actions/checkout@v7 - - uses: astral-sh/setup-uv@v9.0.0 + - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: enable-cache: true prune-cache: true - name: Install dependencies - run: uv sync + run: uv sync --locked - name: Download existing DB run: gh release download --pattern "uisce.db" --dir out/ diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 75ae8ac..52fe72e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -10,13 +10,13 @@ jobs: steps: - uses: actions/checkout@v7 - - uses: astral-sh/setup-uv@v9.0.0 + - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: enable-cache: true prune-cache: true - name: Install dependencies - run: uv sync --group dev + run: uv sync --locked --group dev - name: Lint run: uv run ruff check diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index 3f2f7b0..85ec1e8 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -23,18 +23,21 @@ concurrency: jobs: build: - if: github.event_name != 'workflow_run' || github.event.workflow_run.conclusion == 'success' + if: >- + github.event_name != 'workflow_run' || + (github.event.workflow_run.conclusion == 'success' && + github.event.workflow_run.head_branch == 'main') runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 - - uses: astral-sh/setup-uv@v9.0.0 + - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: enable-cache: true prune-cache: true - name: Install dependencies - run: uv sync + run: uv sync --locked # The site is a projection of the published DB, so a UI change deploys # from the latest release without re-reading the feed. The freshness From 1d99c71ca802dc48b33beab6ef546e493448be4c Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 08:27:00 +0000 Subject: [PATCH 2/7] Never leave the latest release without a uisce.db The second build of a day re-publishes to the release the first one created, with gh release upload --clobber. --clobber deletes the old asset and then uploads, so an upload that fails (a 34 MB transfer is the step most likely to) left the latest release with no uisce.db, and every later Build DB and Build site run then failed at its download. The upload now goes to uisce.db.next, which is never read while uisce.db exists. Only once it has landed is the old uisce.db deleted and uisce.db.next renamed to uisce.db (PATCH on the release asset). A failed upload therefore touches nothing a reader uses. What is left is the gap between the DELETE and the PATCH, two small API calls: GitHub has no atomic replace, and two assets cannot share a name, so some gap is unavoidable while the name stays uisce.db. It is covered twice: - Both downloads fall back to uisce.db.next when the latest release has no uisce.db. In that state uisce.db.next is always the complete new DB, because the old one is only deleted after the upload succeeds. - The next Publish DB finishes an interrupted swap (renames a lone uisce.db.next to uisce.db) before it stages its own upload, so its --clobber on the staging name can never delete the only copy. A first build of the day still uses gh release create, which uploads into a draft and publishes it only once the asset is there, so it never exposed an empty release. Rejected: a new tag per build (the replay writes the tag into closed_at as a date, and the release count would double) and readers walking back to an older release (it would build on a DB that lacks the earlier build's observations, and archive a stale DB as the day's). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0168hLW2X3mkV3Jhm26LJSwQ --- .github/workflows/build.yml | 38 ++++++++++++++++++++++++++++++++----- .github/workflows/pages.yml | 6 +++++- 2 files changed, 38 insertions(+), 6 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index a20a14b..28bfecc 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -52,7 +52,10 @@ jobs: run: uv sync --locked - name: Download existing DB - run: gh release download --pattern "uisce.db" --dir out/ + run: | + mkdir -p out + gh release download --pattern "uisce.db" --dir out/ \ + || gh release download --pattern "uisce.db.next" -O out/uisce.db --clobber env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} @@ -85,13 +88,38 @@ jobs: - name: Build inferred cases table run: uv run uisce-build-inferred + # A second build the same day replaces the asset. --clobber deletes the old + # one before uploading, so the upload goes to uisce.db.next and only a + # landed upload is swapped in. If a run dies between the delete and the + # rename, the downloads fall back to uisce.db.next and the next run here + # finishes the swap before it stages anything. - name: Publish DB run: | TAG="$(date +%Y-%m-%d)" - gh release create "$TAG" out/uisce.db \ - --title "$TAG" \ - --notes "Data refresh" \ - || gh release upload "$TAG" out/uisce.db --clobber + if ! gh release view "$TAG" > /dev/null 2>&1; then + gh release create "$TAG" out/uisce.db --title "$TAG" --notes "Data refresh" + exit 0 + fi + asset_id() { + gh api "repos/{owner}/{repo}/releases/tags/$TAG" \ + --jq ".assets[] | select(.name == \"$1\") | .id" + } + rename() { + gh api -X PATCH "repos/{owner}/{repo}/releases/assets/$1" -f name=uisce.db > /dev/null + } + OLD="$(asset_id uisce.db)" + NEXT="$(asset_id uisce.db.next)" + if [ -z "$OLD" ] && [ -n "$NEXT" ]; then + rename "$NEXT" + fi + ln -f out/uisce.db out/uisce.db.next + gh release upload "$TAG" out/uisce.db.next --clobber + OLD="$(asset_id uisce.db)" + NEXT="$(asset_id uisce.db.next)" + if [ -n "$OLD" ]; then + gh api -X DELETE "repos/{owner}/{repo}/releases/assets/$OLD" + fi + rename "$NEXT" env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index 85ec1e8..a515ca5 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -42,8 +42,12 @@ jobs: # The site is a projection of the published DB, so a UI change deploys # from the latest release without re-reading the feed. The freshness # banner follows the DB, not this build's clock. + # uisce.db.next: see the Publish DB step in build.yml. - name: Download the published DB - run: gh release download --pattern "uisce.db" --dir out/ + run: | + mkdir -p out + gh release download --pattern "uisce.db" --dir out/ \ + || gh release download --pattern "uisce.db.next" -O out/uisce.db --clobber env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} From ea2e367006a94eab6dcbbb9d9620e7482d82491f Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 08:27:00 +0000 Subject: [PATCH 3/7] Record that the closed_at replay recovers nothing today Dry run of uisce-replay-closed-at over all 75 release snapshots (2026-06-30 to 2026-09-23) against a copy of the 2026-09-23 release: 6,961 transitions found, every one already stamped with the same date, 0 rows to stamp and 0 that would change. The replay_closed_at input is left in place for the owner to decide on; its --limit 100 runs out around 2026-10-19. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0168hLW2X3mkV3Jhm26LJSwQ --- notes/data-quality.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/notes/data-quality.md b/notes/data-quality.md index 206eefb..85bfbe2 100644 --- a/notes/data-quality.md +++ b/notes/data-quality.md @@ -249,6 +249,10 @@ The 12% figure above is stale, and the paragraph's implied remedy — shrink the If a closure *series* is ever published (month-over-month counts, or a time-to-close metric keyed on `closed_at`), this stops being a prose caveat and needs the cadence recorded alongside the data so the series can be corrected rather than annotated. +### The replay has nothing left to recover (2026-09-24) + +A dry run of `uisce-replay-closed-at` over all 75 release snapshots (2026-06-30 to 2026-09-23) against a copy of the 2026-09-23 release finds 6,961 transitions, and every one of those cases already carries a `closed_at` on the same date as the replayed tag: **0 rows to stamp, and 0 that would change** even if the replay were allowed to overwrite. None of the 5,872 closed cases with a NULL `closed_at` appears in the replay at all; they closed before the first snapshot or were never seen `Open`. Since the v2 schema landed, the live upsert has stamped every transition a snapshot can see, so the `replay_closed_at` dispatch input only earns its place if the DB is restored from an older release. Whether to drop it is the owner's call. If it stays, its `gh release list --limit 100` stops reaching the earliest snapshots once the release count passes 100, around 2026-10-19. + ### Twice-daily builds: why, and why not three (2026-07-31) The second daily build slot exists for publication latency, not to sharpen `closed_at` (see above — past a daily cadence, Uisce Éireann's own administrative lag dominates, not the build gap). Notices publish between 07:00 and 16:00 UTC (staffed office hours), so a second build only helps if it lands inside that window: measured over 8,135 cases, a single evening build leaves a mean **7.7h** from publication to the site, a midday build halves that to **3.9h**, and an overnight build would only have bought **0.9h**. A third build takes 3.9h to 3.5h — not worth the run. From 46dc53e100c7fdc8bf275296546f688215417940 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 08:29:21 +0000 Subject: [PATCH 4/7] List the replay input as a decision waiting on the owner Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0168hLW2X3mkV3Jhm26LJSwQ --- notes/roadmap.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/notes/roadmap.md b/notes/roadmap.md index 9adc879..daa203f 100644 --- a/notes/roadmap.md +++ b/notes/roadmap.md @@ -44,6 +44,10 @@ Started 2026-09-05, from the follow-ups the nine PRs of the missing-features sur ## Decisions waiting on the owner +- **Keep the `replay_closed_at` dispatch input?** A dry run over all 75 releases on + 2026-09-24 found 0 values to stamp or change (data-quality.md, "The replay has nothing left + to recover"). It only earns its place if the DB is ever restored from an older release. If + it stays, its `gh release list --limit 100` needs raising before about 2026-10-19. - **`IGNORE_BOIL_NOTICES`.** Recommendation is to leave it off (boil-notices.md, re-measured 2026-09-05): the two accruing notices and the one paired one are the live warnings the health marker exists for. The cost of leaving it off is 13 of 17 issue events excluded as stale From b0481241ed32eca8a025807725276f560174e459 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 08:46:11 +0000 Subject: [PATCH 5/7] Drop the replay_closed_at dispatch input The dry run over all 75 releases found nothing left to recover, so the input and its step go (owner decision). The script stays for a DB restored from an older release, run by hand. The roadmap entry is closed and the outcome recorded in data-quality.md. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0168hLW2X3mkV3Jhm26LJSwQ --- .github/workflows/build.yml | 24 ------------------------ notes/data-quality.md | 2 +- notes/roadmap.md | 4 ---- 3 files changed, 1 insertion(+), 29 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 28bfecc..7f9f17d 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -9,14 +9,6 @@ on: - cron: '20 11 * * *' - cron: '23 17 * * *' workflow_dispatch: - inputs: - replay_closed_at: - description: >- - Backfill cases.closed_at by replaying every published release DB. - Off by default; needed once after the v2 schema lands, and again only - if the DB is ever restored from an older release. - type: boolean - default: false permissions: contents: write @@ -64,22 +56,6 @@ jobs: LOCATIONIQ_API_KEY: ${{ secrets.LOCATIONIQ_API_KEY }} run: uv run uisce-pipeline - # Runs after the pipeline so the DB is already migrated to v2 and has this - # build's own transitions stamped — the replay never overwrites those, and - # they are the more precise value. The two together leave no gap: the - # replay covers snapshot-to-snapshot, the pipeline's upsert covers the - # last snapshot to now. Idempotent, so a repeat run is harmless. - - name: Backfill closed_at from published snapshots - if: ${{ inputs.replay_closed_at }} - run: | - mkdir -p snaps - for TAG in $(gh release list --limit 100 --json tagName --jq '.[].tagName'); do - gh release download "$TAG" --pattern uisce.db -O "snaps/$TAG.db" || true - done - uv run uisce-replay-closed-at --snapshots snaps --write - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - # The rules answer the templated ~93% here; the abstentions wait for a # local run with the LLM. See notes/rules-vs-llm-end-times.md. - name: Infer end times by rules diff --git a/notes/data-quality.md b/notes/data-quality.md index 85bfbe2..993e8e4 100644 --- a/notes/data-quality.md +++ b/notes/data-quality.md @@ -251,7 +251,7 @@ If a closure *series* is ever published (month-over-month counts, or a time-to-c ### The replay has nothing left to recover (2026-09-24) -A dry run of `uisce-replay-closed-at` over all 75 release snapshots (2026-06-30 to 2026-09-23) against a copy of the 2026-09-23 release finds 6,961 transitions, and every one of those cases already carries a `closed_at` on the same date as the replayed tag: **0 rows to stamp, and 0 that would change** even if the replay were allowed to overwrite. None of the 5,872 closed cases with a NULL `closed_at` appears in the replay at all; they closed before the first snapshot or were never seen `Open`. Since the v2 schema landed, the live upsert has stamped every transition a snapshot can see, so the `replay_closed_at` dispatch input only earns its place if the DB is restored from an older release. Whether to drop it is the owner's call. If it stays, its `gh release list --limit 100` stops reaching the earliest snapshots once the release count passes 100, around 2026-10-19. +A dry run of `uisce-replay-closed-at` over all 75 release snapshots (2026-06-30 to 2026-09-23) against a copy of the 2026-09-23 release finds 6,961 transitions, and every one of those cases already carries a `closed_at` on the same date as the replayed tag: **0 rows to stamp, and 0 that would change** even if the replay were allowed to overwrite. None of the 5,872 closed cases with a NULL `closed_at` appears in the replay at all; they closed before the first snapshot or were never seen `Open`. Since the v2 schema landed, the live upsert has stamped every transition a snapshot can see, so the `replay_closed_at` dispatch input was dropped from Build DB (owner, 2026-09-24). The script stays for the one case it still serves, a DB restored from an older release, and is run by hand as its docstring shows. ### Twice-daily builds: why, and why not three (2026-07-31) diff --git a/notes/roadmap.md b/notes/roadmap.md index daa203f..9adc879 100644 --- a/notes/roadmap.md +++ b/notes/roadmap.md @@ -44,10 +44,6 @@ Started 2026-09-05, from the follow-ups the nine PRs of the missing-features sur ## Decisions waiting on the owner -- **Keep the `replay_closed_at` dispatch input?** A dry run over all 75 releases on - 2026-09-24 found 0 values to stamp or change (data-quality.md, "The replay has nothing left - to recover"). It only earns its place if the DB is ever restored from an older release. If - it stays, its `gh release list --limit 100` needs raising before about 2026-10-19. - **`IGNORE_BOIL_NOTICES`.** Recommendation is to leave it off (boil-notices.md, re-measured 2026-09-05): the two accruing notices and the one paired one are the live warnings the health marker exists for. The cost of leaving it off is 13 of 17 issue events excluded as stale From 291aae2b29332ce4b37d8818bb45b4b3eb17f389 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 10:20:36 +0000 Subject: [PATCH 6/7] Move the DB fetch and publish into scripts that survive the review From a code review of this PR: - The swap only repaired today's release, so a run dying mid-swap on its day's last build stranded that release with only uisce.db.next. publish-db.sh repairs the latest published release first. - releases/tags/ 404s on a draft that `gh release view` finds, which wedged the day after a killed create. The script reads assets through `gh release view` (apiUrl) and replaces a leftover draft. - It deleted uisce.db without checking the staged upload was listed; it now stops first. A 5xx on the view was read as "no release" and then failed on create; only "release not found" means absent now. - fetch-db.sh is the one download, with the .next fallback, used by both workflows, the README, build.py's hint and the replay docstring (whose loop now lists up to 1000 releases). - The main-only refusal is its own job and the build-db concurrency group moved onto the build job, so a refused dispatch cannot displace a queued scheduled run. - uv is pinned (0.8.17) beside the setup-uv SHA, so --locked cannot fail on a new uv release. CLAUDE.md gains the replay row. Both scripts were run against a fake gh through: no release, second build, stranded yesterday, stranded today, leftover draft, a 502 on the view, and a lost upload; actionlint and shellcheck pass. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0168hLW2X3mkV3Jhm26LJSwQ --- .github/workflows/build.yml | 71 +++++++++++------------------------ .github/workflows/ci.yml | 2 + .github/workflows/pages.yml | 8 ++-- CLAUDE.md | 1 + README.md | 9 +++-- scripts/fetch-db.sh | 11 ++++++ scripts/publish-db.sh | 52 +++++++++++++++++++++++++ src/uisce/build.py | 2 +- src/uisce/replay_closed_at.py | 4 +- 9 files changed, 100 insertions(+), 60 deletions(-) create mode 100755 scripts/fetch-db.sh create mode 100755 scripts/publish-db.sh diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 7f9f17d..7e7cf1a 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -13,30 +13,35 @@ on: permissions: contents: write -# Guards against a manual dispatch landing on top of a scheduled run — two -# overlapping runs would silently discard one's closed_at observations. Queue -# rather than cancel: a cancelled run has already read the feed and would lose -# the same way. See notes/data-quality.md ("closed_at is a floor"). -concurrency: - group: build-db - cancel-in-progress: false - jobs: - build: + # A dispatch from a branch would publish that branch's DB as the latest + # release and push to main. Its own job, outside the build-db group, so a + # refused dispatch fails visibly without displacing a queued scheduled run. + refuse: + if: github.ref != 'refs/heads/main' runs-on: ubuntu-latest steps: - # A dispatch from a branch would publish that branch's DB as the latest - # release and push to main. Failed rather than skipped, so it is seen. - - name: Refuse to run off main - if: github.ref != 'refs/heads/main' - run: | + - run: | echo "::error::Build DB publishes the latest release and pushes to main; dispatch it from main, not $GITHUB_REF" exit 1 + build: + if: github.ref == 'refs/heads/main' + runs-on: ubuntu-latest + # Guards against a manual dispatch landing on top of a scheduled run - two + # overlapping runs would silently discard one's closed_at observations. Queue + # rather than cancel: a cancelled run has already read the feed and would lose + # the same way. See notes/data-quality.md ("closed_at is a floor"). + concurrency: + group: build-db + cancel-in-progress: false + steps: - uses: actions/checkout@v7 - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: + # the uv the lock was checked with; bump together with uv.lock + version: "0.8.17" enable-cache: true prune-cache: true @@ -44,10 +49,7 @@ jobs: run: uv sync --locked - name: Download existing DB - run: | - mkdir -p out - gh release download --pattern "uisce.db" --dir out/ \ - || gh release download --pattern "uisce.db.next" -O out/uisce.db --clobber + run: scripts/fetch-db.sh env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} @@ -64,38 +66,9 @@ jobs: - name: Build inferred cases table run: uv run uisce-build-inferred - # A second build the same day replaces the asset. --clobber deletes the old - # one before uploading, so the upload goes to uisce.db.next and only a - # landed upload is swapped in. If a run dies between the delete and the - # rename, the downloads fall back to uisce.db.next and the next run here - # finishes the swap before it stages anything. + # The swap and its recovery are in the script; see its header. - name: Publish DB - run: | - TAG="$(date +%Y-%m-%d)" - if ! gh release view "$TAG" > /dev/null 2>&1; then - gh release create "$TAG" out/uisce.db --title "$TAG" --notes "Data refresh" - exit 0 - fi - asset_id() { - gh api "repos/{owner}/{repo}/releases/tags/$TAG" \ - --jq ".assets[] | select(.name == \"$1\") | .id" - } - rename() { - gh api -X PATCH "repos/{owner}/{repo}/releases/assets/$1" -f name=uisce.db > /dev/null - } - OLD="$(asset_id uisce.db)" - NEXT="$(asset_id uisce.db.next)" - if [ -z "$OLD" ] && [ -n "$NEXT" ]; then - rename "$NEXT" - fi - ln -f out/uisce.db out/uisce.db.next - gh release upload "$TAG" out/uisce.db.next --clobber - OLD="$(asset_id uisce.db)" - NEXT="$(asset_id uisce.db.next)" - if [ -n "$OLD" ]; then - gh api -X DELETE "repos/{owner}/{repo}/releases/assets/$OLD" - fi - rename "$NEXT" + run: scripts/publish-db.sh out/uisce.db env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 52fe72e..486ecc3 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -12,6 +12,8 @@ jobs: - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: + # the uv the lock was checked with; bump together with uv.lock + version: "0.8.17" enable-cache: true prune-cache: true diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index a515ca5..b799ebc 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -33,6 +33,8 @@ jobs: - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: + # the uv the lock was checked with; bump together with uv.lock + version: "0.8.17" enable-cache: true prune-cache: true @@ -42,12 +44,8 @@ jobs: # The site is a projection of the published DB, so a UI change deploys # from the latest release without re-reading the feed. The freshness # banner follows the DB, not this build's clock. - # uisce.db.next: see the Publish DB step in build.yml. - name: Download the published DB - run: | - mkdir -p out - gh release download --pattern "uisce.db" --dir out/ \ - || gh release download --pattern "uisce.db.next" -O out/uisce.db --clobber + run: scripts/fetch-db.sh env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/CLAUDE.md b/CLAUDE.md index 1b145ad..12294d0 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -47,6 +47,7 @@ with the evidence that closed them. | Duration outliers are categorical, not statistical. The 14-day cap is a backstop, not the outlier strategy. | data-quality.md — "Duration outliers are categorical" | | "We are investigating" reference pairing works but rescues almost nothing — not worth building. | data-quality.md — "'We are investigating' notices" (corrected 2026-07-20) | | `closed_at` is a floor: short-lived cases are never observed open. Twice-daily builds are the settled cadence. | data-quality.md — "`closed_at` is a floor" (re-measured 2026-07-31) | +| The `replay_closed_at` dispatch input is gone: a dry run over all 75 releases found 0 values to stamp or change. The script stays for a DB restored from an older release, run by hand. | data-quality.md - "The replay has nothing left to recover" (2026-09-24) | | A case the feed drops while `Open` is stamped `vanished_at` (schema v4) and is closed with no signal on the site, never `closed_at`. The stamp is safe only behind the feed-count guard (`FEED_COUNT_TOLERANCE`), which refuses a short download before anything touches the DB. | data-quality.md - "Cases that vanish from the feed" (2026-09-05) | | **A case is open only while nothing its own text has ended.** `is_open(row, now)` reads `status`, `vanished_at` and a passed *observed* end, decided once in `resolve_case` and carried on `Case.is_open` for every surface that says open. The close date follows the same reading: the notice's own completion, else `closed_at` (`closed_on`, 2026-09-24). The feed closes a case a median 72h after the notice reports completion; 216 of 562 `Open` cases were past one, 0 of 7,667 completions were ever followed up. Scheduled ends do not close a case for display. | statuspage-methodology.md - "The notice's own completion closes it" (2026-09-05) | | gemma-4-12b-qat over qwen3.5-9b for end-time extraction; prompt version is at v3. | model-and-runtime-benchmarks.md, end-time-eval.md | diff --git a/README.md b/README.md index 2bd71d4..0f2ebc1 100644 --- a/README.md +++ b/README.md @@ -8,12 +8,15 @@ The [website that this repo generates](https://baz8080.github.io/uisce/) is rebu ## Just want the data? -Grab the latest `uisce.db` from [releases](https://github.com/baz8080/uisce/releases) — no setup needed: +Grab the latest `uisce.db` from [releases](https://github.com/baz8080/uisce/releases) - no setup needed: ``` -gh release download --clobber --pattern "uisce.db" --dir out/ +scripts/fetch-db.sh ``` +It wraps `gh release download --pattern uisce.db`, falling back to `uisce.db.next` in the +moment a build is swapping a new copy in. + Tables: * `cases` — one row per published notice pin (title, description, dates, status, impact flags, WGS84 coordinates). `work_category` is a slug normalised from the title (`burst_main`, `essential_works`, …); `work_type` (Planned/Unplanned) is taken from the feed but overridden for categories where the label is unambiguous (a burst main is never planned). @@ -97,7 +100,7 @@ CI runs the rules half on every data build (`uisce-infer --rules-only`) and comm 1. `git pull` — CI appends to `data/inferred_end_times.jsonl`; `.gitattributes` merges a concurrent local append rather than conflicting 2. Start the LLM server on :1234 -3. `gh release download --clobber --pattern "uisce.db" --dir out/` +3. `scripts/fetch-db.sh` 4. `uv run uisce-infer` — appends results to `data/inferred_end_times.jsonl` (committed to the repo; only new/changed descriptions are processed); commit and push 5. (Local check only — CI rebuilds the table itself) `uv run uisce-build-inferred` diff --git a/scripts/fetch-db.sh b/scripts/fetch-db.sh new file mode 100755 index 0000000..dc16b81 --- /dev/null +++ b/scripts/fetch-db.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +# Download a release's uisce.db, the latest release unless TAG is given. +# Falls back to uisce.db.next, the name publish-db.sh stages an upload under +# while it swaps it in. usage: scripts/fetch-db.sh [TAG] [OUT] +set -euo pipefail + +tag="${1:-}" +out="${2:-out/uisce.db}" +mkdir -p "$(dirname "$out")" +gh release download ${tag:+"$tag"} --pattern uisce.db -O "$out" --clobber \ + || gh release download ${tag:+"$tag"} --pattern uisce.db.next -O "$out" --clobber diff --git a/scripts/publish-db.sh b/scripts/publish-db.sh new file mode 100755 index 0000000..c8befd6 --- /dev/null +++ b/scripts/publish-db.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Publish DB as today's release asset uisce.db without ever leaving a release +# with neither a uisce.db nor a complete uisce.db.next. `gh release upload +# --clobber` deletes before it uploads, so a second build on the same day +# uploads to uisce.db.next and renames it in only once it has landed; +# fetch-db.sh falls back to it in between. usage: scripts/publish-db.sh [DB] +set -euo pipefail + +db="${1:-out/uisce.db}" +tag="$(date -u +%Y-%m-%d)" + +# " " per asset. `gh release view` finds a draft by its pending +# tag, which GET releases/tags/ does not. +assets() { gh release view "$@" --json assets --jq '.assets[] | "\(.name) \(.apiUrl)"'; } +url() { awk -v n="$1" '$1 == n { print $2 }' <<<"$2"; } +rename_in() { gh api -X PATCH "$1" -f name=uisce.db >/dev/null; } +create() { gh release create "$tag" "$db" --title "$tag" --notes "Data refresh"; } + +# A swap an earlier run left half done, on the latest published release: that +# is yesterday's when the run that died was the last of its day. +latest="$(assets || true)" +if [ -z "$(url uisce.db "$latest")" ] && [ -n "$(url uisce.db.next "$latest")" ]; then + rename_in "$(url uisce.db.next "$latest")" +fi + +if ! draft="$(gh release view "$tag" --json isDraft --jq .isDraft 2>&1)"; then + # anything but a missing release (a 5xx, a rate limit) fails the step + [[ "$draft" == *"release not found"* ]] || { echo "$draft" >&2; exit 1; } + create + exit 0 +fi +if [ "$draft" = "true" ]; then + # a create killed between its draft and its publish; nothing reads a draft + gh release delete "$tag" --yes + create + exit 0 +fi + +staged="$(dirname "$db")/uisce.db.next" +ln -f "$db" "$staged" +gh release upload "$tag" "$staged" --clobber +list="$(assets "$tag")" +next="$(url uisce.db.next "$list")" +if [ -z "$next" ]; then + echo "::error::uisce.db.next is not on release $tag after its upload; uisce.db left in place" >&2 + exit 1 +fi +old="$(url uisce.db "$list")" +if [ -n "$old" ]; then + gh api -X DELETE "$old" +fi +rename_in "$next" diff --git a/src/uisce/build.py b/src/uisce/build.py index 641e8e0..4167814 100644 --- a/src/uisce/build.py +++ b/src/uisce/build.py @@ -199,7 +199,7 @@ def check_cases_cover(conn, case_ids): f"{len(missing)} case_id(s) in {JSONL_PATH} are not present in {DB_PATH} " f"(range {missing[0]}-{missing[-1]}). The local DB is likely older than " "whatever DB the inference run used. Refresh it first, e.g.:\n" - " gh release download --pattern uisce.db --dir out/ --clobber" + " scripts/fetch-db.sh" ) diff --git a/src/uisce/replay_closed_at.py b/src/uisce/replay_closed_at.py index 9d7d1b4..f09f9bf 100644 --- a/src/uisce/replay_closed_at.py +++ b/src/uisce/replay_closed_at.py @@ -9,8 +9,8 @@ Download the snapshots first; they are ~10-20MB each: - for T in $(gh release list --limit 100 | cut -f1); do - gh release download "$T" --pattern uisce.db -O "snaps/$T.db" + for T in $(gh release list --limit 1000 --json tagName --jq '.[].tagName'); do + scripts/fetch-db.sh "$T" "snaps/$T.db" done """ From ef16ec49e7a71fb77d62a07cf6ece0949f7f1afa Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 11:37:56 +0000 Subject: [PATCH 7/7] Publish one release per build instead of swapping a same-day asset A second code review of the swap found three more ways it could strand a release or quietly lose a build's DB, all from juggling two asset names inside one release. Owner decision: every build publishes its own release, tagged YYYY-MM-DD-HHMM (UTC). gh creates it as a draft, uploads and publishes, so the latest release always holds a complete uisce.db. - publish-db.sh is a single `gh release create --latest`, staging the file as uisce.db whatever it is called (the `#` suffix is only a display label). fetch-db.sh is a plain download with no fallback, so its errors are its own. - replay_closed_at stamps the tag's date part and sorts by tag, since "2026-09-24-1845.db" sorts before "2026-09-24.db" by file name; the documented loop excludes drafts. - uv is pinned at 0.12.18, not 0.8.17, which only offers CPython 3.14.0rc2; 0.12.18 passes `uv lock --check` and installs 3.14.7. - pipeline-dependencies.md's recipe uses fetch-db.sh; data-quality.md records the decision and the rejected swap; CLAUDE.md gains a row. Checked: shellcheck, actionlint, a fake gh (publish, same-minute refusal, a non-uisce file name, fetch), `uvx uv@0.12.18 lock --check`, ruff, pytest. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_0168hLW2X3mkV3Jhm26LJSwQ --- .github/workflows/build.yml | 5 ++- .github/workflows/ci.yml | 4 +-- .github/workflows/pages.yml | 4 +-- CLAUDE.md | 1 + README.md | 4 +-- notes/data-quality.md | 14 ++++++++ notes/pipeline-dependencies.md | 2 +- scripts/fetch-db.sh | 6 ++-- scripts/publish-db.sh | 59 +++++++--------------------------- src/uisce/replay_closed_at.py | 17 +++++----- tests/test_replay_closed_at.py | 8 +++++ 11 files changed, 54 insertions(+), 70 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 7e7cf1a..c2b5e1d 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -40,8 +40,8 @@ jobs: - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: - # the uv the lock was checked with; bump together with uv.lock - version: "0.8.17" + # checked with `uv lock --check` on 0.12.18; bump deliberately, with uv.lock + version: "0.12.18" enable-cache: true prune-cache: true @@ -66,7 +66,6 @@ jobs: - name: Build inferred cases table run: uv run uisce-build-inferred - # The swap and its recovery are in the script; see its header. - name: Publish DB run: scripts/publish-db.sh out/uisce.db env: diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 486ecc3..eb4adea 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -12,8 +12,8 @@ jobs: - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: - # the uv the lock was checked with; bump together with uv.lock - version: "0.8.17" + # checked with `uv lock --check` on 0.12.18; bump deliberately, with uv.lock + version: "0.12.18" enable-cache: true prune-cache: true diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index b799ebc..739e868 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -33,8 +33,8 @@ jobs: - uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 with: - # the uv the lock was checked with; bump together with uv.lock - version: "0.8.17" + # checked with `uv lock --check` on 0.12.18; bump deliberately, with uv.lock + version: "0.12.18" enable-cache: true prune-cache: true diff --git a/CLAUDE.md b/CLAUDE.md index 12294d0..e9db3dd 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -48,6 +48,7 @@ with the evidence that closed them. | "We are investigating" reference pairing works but rescues almost nothing — not worth building. | data-quality.md — "'We are investigating' notices" (corrected 2026-07-20) | | `closed_at` is a floor: short-lived cases are never observed open. Twice-daily builds are the settled cadence. | data-quality.md — "`closed_at` is a floor" (re-measured 2026-07-31) | | The `replay_closed_at` dispatch input is gone: a dry run over all 75 releases found 0 values to stamp or change. The script stays for a DB restored from an older release, run by hand. | data-quality.md - "The replay has nothing left to recover" (2026-09-24) | +| Every data build publishes its **own release**, tagged `YYYY-MM-DD-HHMM` (UTC); `scripts/publish-db.sh` and `scripts/fetch-db.sh` are the only publish and download. Replacing the asset in a per-day release was rejected: `--clobber` deletes first, and a rename swap stranded releases in three ways. | data-quality.md - "One release per build" (2026-09-24) | | A case the feed drops while `Open` is stamped `vanished_at` (schema v4) and is closed with no signal on the site, never `closed_at`. The stamp is safe only behind the feed-count guard (`FEED_COUNT_TOLERANCE`), which refuses a short download before anything touches the DB. | data-quality.md - "Cases that vanish from the feed" (2026-09-05) | | **A case is open only while nothing its own text has ended.** `is_open(row, now)` reads `status`, `vanished_at` and a passed *observed* end, decided once in `resolve_case` and carried on `Case.is_open` for every surface that says open. The close date follows the same reading: the notice's own completion, else `closed_at` (`closed_on`, 2026-09-24). The feed closes a case a median 72h after the notice reports completion; 216 of 562 `Open` cases were past one, 0 of 7,667 completions were ever followed up. Scheduled ends do not close a case for display. | statuspage-methodology.md - "The notice's own completion closes it" (2026-09-05) | | gemma-4-12b-qat over qwen3.5-9b for end-time extraction; prompt version is at v3. | model-and-runtime-benchmarks.md, end-time-eval.md | diff --git a/README.md b/README.md index 0f2ebc1..acbf43f 100644 --- a/README.md +++ b/README.md @@ -14,8 +14,8 @@ Grab the latest `uisce.db` from [releases](https://github.com/baz8080/uisce/rele scripts/fetch-db.sh ``` -It wraps `gh release download --pattern uisce.db`, falling back to `uisce.db.next` in the -moment a build is swapping a new copy in. +It wraps `gh release download --pattern uisce.db`; every data build publishes its own +release, so the latest one always holds a complete DB. Tables: diff --git a/notes/data-quality.md b/notes/data-quality.md index f294055..62faa27 100644 --- a/notes/data-quality.md +++ b/notes/data-quality.md @@ -257,6 +257,20 @@ If a closure *series* is ever published (month-over-month counts, or a time-to-c A dry run of `uisce-replay-closed-at` over all 75 release snapshots (2026-06-30 to 2026-09-23) against a copy of the 2026-09-23 release finds 6,961 transitions, and every one of those cases already carries a `closed_at` on the same date as the replayed tag: **0 rows to stamp, and 0 that would change** even if the replay were allowed to overwrite. None of the 5,872 closed cases with a NULL `closed_at` appears in the replay at all; they closed before the first snapshot or were never seen `Open`. Since the v2 schema landed, the live upsert has stamped every transition a snapshot can see, so the `replay_closed_at` dispatch input was dropped from Build DB (owner, 2026-09-24). The script stays for the one case it still serves, a DB restored from an older release, and is run by hand as its docstring shows. +### One release per build (2026-09-24) + +Until this date each day had one release, and the second build of the day replaced its +`uisce.db` with `gh release upload --clobber`, which deletes the old asset before uploading +the new one: a failure in between left the day's release with no DB, and every later build +and Pages deploy, which all start from the latest release, failed until someone fixed it by +hand. A same-day swap (upload as `uisce.db.next`, then delete and rename) was built and +rejected on two rounds of review: each round found another state it could strand, all of +them coming from juggling two names inside one release. From this date every build publishes +its own release, tagged `YYYY-MM-DD-HHMM` (UTC); `gh release create` makes a draft, uploads, +then publishes, so the latest release always holds a complete `uisce.db`. About two releases +a day instead of one. `uisce-replay-closed-at` reads the date from the tag's first ten +characters, so the daily and per-build names replay alike. + ### Twice-daily builds: why, and why not three (2026-07-31) The second daily build slot exists for publication latency, not to sharpen `closed_at` (see above — past a daily cadence, Uisce Éireann's own administrative lag dominates, not the build gap). Notices publish between 07:00 and 16:00 UTC (staffed office hours), so a second build only helps if it lands inside that window: measured over 8,135 cases, a single evening build leaves a mean **7.7h** from publication to the site, a midday build halves that to **3.9h**, and an overnight build would only have bought **0.9h**. A third build takes 3.9h to 3.5h — not worth the run. diff --git a/notes/pipeline-dependencies.md b/notes/pipeline-dependencies.md index aaae349..0b1670e 100644 --- a/notes/pipeline-dependencies.md +++ b/notes/pipeline-dependencies.md @@ -9,7 +9,7 @@ `uisce-build-inferred` (`src/uisce/build.py`) checks for this up front and fails with a clear message naming the missing case_id range, rather than a raw `sqlite3.IntegrityError`. The fix is always the same: get a DB that's at least as new as whatever the inference run used, e.g.: ``` -gh release download --pattern uisce.db --dir out/ --clobber +scripts/fetch-db.sh ``` (defaults to the latest release; pass a specific tag if you know which one you need). There's no automatic reconciliation here on purpose — the inference run itself doesn't record which DB snapshot it used (see the description-hash discussion elsewhere in this repo's history for why the hash alone is enough for correctness, just not for provenance), so "grab the latest release" is the practical default rather than something that could be automated reliably. diff --git a/scripts/fetch-db.sh b/scripts/fetch-db.sh index dc16b81..27c1784 100755 --- a/scripts/fetch-db.sh +++ b/scripts/fetch-db.sh @@ -1,11 +1,9 @@ #!/usr/bin/env bash # Download a release's uisce.db, the latest release unless TAG is given. -# Falls back to uisce.db.next, the name publish-db.sh stages an upload under -# while it swaps it in. usage: scripts/fetch-db.sh [TAG] [OUT] +# usage: scripts/fetch-db.sh [TAG] [OUT] set -euo pipefail tag="${1:-}" out="${2:-out/uisce.db}" mkdir -p "$(dirname "$out")" -gh release download ${tag:+"$tag"} --pattern uisce.db -O "$out" --clobber \ - || gh release download ${tag:+"$tag"} --pattern uisce.db.next -O "$out" --clobber +gh release download ${tag:+"$tag"} --pattern uisce.db -O "$out" --clobber diff --git a/scripts/publish-db.sh b/scripts/publish-db.sh index c8befd6..a3bf2fe 100755 --- a/scripts/publish-db.sh +++ b/scripts/publish-db.sh @@ -1,52 +1,15 @@ #!/usr/bin/env bash -# Publish DB as today's release asset uisce.db without ever leaving a release -# with neither a uisce.db nor a complete uisce.db.next. `gh release upload -# --clobber` deletes before it uploads, so a second build on the same day -# uploads to uisce.db.next and renames it in only once it has landed; -# fetch-db.sh falls back to it in between. usage: scripts/publish-db.sh [DB] +# Publish DB as a release of its own, tagged with the build's UTC date and time. +# gh creates the release as a draft, uploads, then publishes it, so the latest +# release always holds a complete uisce.db. usage: scripts/publish-db.sh [DB] set -euo pipefail db="${1:-out/uisce.db}" -tag="$(date -u +%Y-%m-%d)" - -# " " per asset. `gh release view` finds a draft by its pending -# tag, which GET releases/tags/ does not. -assets() { gh release view "$@" --json assets --jq '.assets[] | "\(.name) \(.apiUrl)"'; } -url() { awk -v n="$1" '$1 == n { print $2 }' <<<"$2"; } -rename_in() { gh api -X PATCH "$1" -f name=uisce.db >/dev/null; } -create() { gh release create "$tag" "$db" --title "$tag" --notes "Data refresh"; } - -# A swap an earlier run left half done, on the latest published release: that -# is yesterday's when the run that died was the last of its day. -latest="$(assets || true)" -if [ -z "$(url uisce.db "$latest")" ] && [ -n "$(url uisce.db.next "$latest")" ]; then - rename_in "$(url uisce.db.next "$latest")" -fi - -if ! draft="$(gh release view "$tag" --json isDraft --jq .isDraft 2>&1)"; then - # anything but a missing release (a 5xx, a rate limit) fails the step - [[ "$draft" == *"release not found"* ]] || { echo "$draft" >&2; exit 1; } - create - exit 0 -fi -if [ "$draft" = "true" ]; then - # a create killed between its draft and its publish; nothing reads a draft - gh release delete "$tag" --yes - create - exit 0 -fi - -staged="$(dirname "$db")/uisce.db.next" -ln -f "$db" "$staged" -gh release upload "$tag" "$staged" --clobber -list="$(assets "$tag")" -next="$(url uisce.db.next "$list")" -if [ -z "$next" ]; then - echo "::error::uisce.db.next is not on release $tag after its upload; uisce.db left in place" >&2 - exit 1 -fi -old="$(url uisce.db "$list")" -if [ -n "$old" ]; then - gh api -X DELETE "$old" -fi -rename_in "$next" +tag="$(date -u +%Y-%m-%d-%H%M)" +# the asset takes the file's name, and fetch-db.sh asks for uisce.db +if [ "$(basename "$db")" != uisce.db ]; then + staged="$(mktemp -d)/uisce.db" + cp "$db" "$staged" + db="$staged" +fi +gh release create "$tag" "$db" --title "$tag" --notes "Data refresh" --latest diff --git a/src/uisce/replay_closed_at.py b/src/uisce/replay_closed_at.py index f09f9bf..4a4b6fc 100644 --- a/src/uisce/replay_closed_at.py +++ b/src/uisce/replay_closed_at.py @@ -9,7 +9,7 @@ Download the snapshots first; they are ~10-20MB each: - for T in $(gh release list --limit 1000 --json tagName --jq '.[].tagName'); do + for T in $(gh release list --limit 1000 --exclude-drafts --json tagName --jq '.[].tagName'); do scripts/fetch-db.sh "$T" "snaps/$T.db" done """ @@ -21,17 +21,18 @@ from uisce.config import DB_PATH -# Snapshot files are named for their release tag (YYYY-MM-DD.db), which is also -# the value written to closed_at, so replayed rows carry the date of the build +# Snapshot files are named for their release tag: YYYY-MM-DD for the daily +# releases, YYYY-MM-DD-HHMM for the per-build ones from 2026-09-24. The date part +# is the value written to closed_at, so replayed rows carry the date of the build # that observed the closure. SNAPSHOT_GLOB = "*.db" def snapshot_files(directory): - """Snapshot paths in tag order. Names are ISO dates, so lexical sort is - chronological; anything else is a caller error rather than something to - guess at.""" - paths = sorted(Path(directory).glob(SNAPSHOT_GLOB)) + """Snapshot paths in tag order. Tags are ISO dates, with a time on the + per-build ones, so sorting the tags is chronological; the file names are not + ("-" sorts before ".db"). Anything else is a caller error, not a guess.""" + paths = sorted(Path(directory).glob(SNAPSHOT_GLOB), key=lambda p: p.stem) if not paths: raise SystemExit(f"No {SNAPSHOT_GLOB} snapshots in {directory}") return [(p.stem, p) for p in paths] @@ -58,7 +59,7 @@ def replay(snapshots): seen_open.add(case_id) closed_at.pop(case_id, None) elif case_id in seen_open and case_id not in closed_at: - closed_at[case_id] = tag + closed_at[case_id] = tag[:10] return closed_at diff --git a/tests/test_replay_closed_at.py b/tests/test_replay_closed_at.py index 34a9857..b285b4e 100644 --- a/tests/test_replay_closed_at.py +++ b/tests/test_replay_closed_at.py @@ -37,6 +37,14 @@ def test_stamps_the_first_snapshot_observing_the_close(self, tmp_path): # the later snapshots must not drag the stamp forward assert replay(snaps) == {1: "2026-07-08"} + def test_a_per_build_tag_stamps_its_date(self, tmp_path): + snaps = [ + _snapshot(tmp_path, "2026-09-24", [(1, "Open")]), + _snapshot(tmp_path, "2026-09-24-1845", [(1, "Closed")]), + ] + assert [tag for tag, _ in snapshot_files(tmp_path)] == ["2026-09-24", "2026-09-24-1845"] + assert replay(snaps) == {1: "2026-09-24"} + def test_case_never_seen_open_is_not_stamped(self, tmp_path): # Created and closed inside one gap: no transition was ever observed, so # there is nothing to recover. ~12% of new cases look like this.