From 6d96141c4f9c4729723c50c0f5aa2590954d8bdf Mon Sep 17 00:00:00 2001 From: Manuel Magnabosco Date: Tue, 25 Aug 2026 21:05:28 +0200 Subject: [PATCH] Refine GitHub Pages presentation layout --- .github/workflows/ci.yml | 3 + Makefile | 5 + docs/_includes/home-evidence.html | 30 + docs/_includes/home-output.html | 33 ++ docs/_includes/home-pipeline.html | 56 ++ docs/_includes/home-status.html | 30 + docs/_layouts/default.html | 2 +- docs/assets/css/site.css | 335 ++++++++++- docs/index.md | 149 +++-- docs/plans/2026-08-25-pages-overview-audit.md | 519 ++++++++++++++++++ tests/python/test_generated_pages.py | 111 ++++ tools/check_generated_pages.py | 290 ++++++++++ 12 files changed, 1503 insertions(+), 60 deletions(-) create mode 100644 docs/_includes/home-evidence.html create mode 100644 docs/_includes/home-output.html create mode 100644 docs/_includes/home-pipeline.html create mode 100644 docs/_includes/home-status.html create mode 100644 docs/plans/2026-08-25-pages-overview-audit.md create mode 100644 tests/python/test_generated_pages.py create mode 100644 tools/check_generated_pages.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d111f10..78699e8 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -36,6 +36,9 @@ jobs: source: ./docs destination: ./build/pages + - name: Check generated GitHub Pages site + run: make generated-pages-check + build-and-test: runs-on: ubuntu-latest diff --git a/Makefile b/Makefile index 7ce9537..bd44e23 100644 --- a/Makefile +++ b/Makefile @@ -3,6 +3,7 @@ AR ?= ar UV ?= uv PYTHON ?= python3 VALGRIND ?= valgrind +PAGES_BUILD_DIR ?= build/pages UV_CACHE_DIR ?= $(CURDIR)/.uv-cache export UV_CACHE_DIR @@ -83,6 +84,7 @@ SHARED_LIBRARY := $(LIB_DIR)/libwang.so OPENMP_LIBRARY := $(LIB_DIR)/libwang_openmp.a .PHONY: all setup serial shared openmp check c-check python-check pages-check \ + generated-pages-check \ strict-check sanitizer-check analyzer-check valgrind-check \ cachegrind-check benchmark benchmark-smoke benchmark-compare \ benchmark-compare-smoke coverage coverage-c coverage-python \ @@ -161,6 +163,9 @@ check: pages-check c-check openmp python-check benchmark-smoke benchmark-compare pages-check: $(PYTHON) tools/check_pages.py +generated-pages-check: + $(PYTHON) tools/check_generated_pages.py $(PAGES_BUILD_DIR) + c-check: serial $(C_TEST_BINS) @set -e; \ if [ -z "$(strip $(C_TEST_BINS))" ]; then \ diff --git a/docs/_includes/home-evidence.html b/docs/_includes/home-evidence.html new file mode 100644 index 0000000..b1da0fb --- /dev/null +++ b/docs/_includes/home-evidence.html @@ -0,0 +1,30 @@ +{% assign evidence = include.evidence %} +{% if evidence.eyebrow and evidence.title and evidence.items.size > 0 %} +
+
+

{{ evidence.eyebrow | escape }}

+

{{ evidence.title | escape }}

+
+ + {% if evidence.introduction %} +

{{ evidence.introduction | escape }}

+ {% endif %} + + +
+{% endif %} diff --git a/docs/_includes/home-output.html b/docs/_includes/home-output.html new file mode 100644 index 0000000..1a3ea6d --- /dev/null +++ b/docs/_includes/home-output.html @@ -0,0 +1,33 @@ +{% assign output = include.output %} +{% if output.eyebrow and output.title and output.image and output.alt and output.caption and output.width and output.height %} +
+
+

{{ output.eyebrow | escape }}

+

{{ output.title | escape }}

+
+ +
+
+ {{ output.alt | escape }} +
+
+

{{ output.caption | escape }}

+ {% if output.source and output.source_label %} + + {{ output.source_label | escape }} + + {% endif %} +
+
+
+{% endif %} diff --git a/docs/_includes/home-pipeline.html b/docs/_includes/home-pipeline.html new file mode 100644 index 0000000..1506668 --- /dev/null +++ b/docs/_includes/home-pipeline.html @@ -0,0 +1,56 @@ +{% assign pipeline = include.pipeline %} +{% if pipeline.eyebrow and pipeline.title and pipeline.stages.size > 0 %} +
+
+

{{ pipeline.eyebrow | escape }}

+

{{ pipeline.title | escape }}

+
+ + {% if pipeline.introduction %} +

{{ pipeline.introduction | escape }}

+ {% endif %} + +
+
    + {% for stage in pipeline.stages %} +
  1. + + {{ stage.label | escape }} + {% if stage.detail %} + {{ stage.detail | escape }} + {% endif %} +
  2. + {% endfor %} +
+ + {% if pipeline.checks.size > 0 %} + + {% endif %} + + {% if pipeline.caption %} +
{{ pipeline.caption | escape }}
+ {% endif %} +
+
+{% endif %} diff --git a/docs/_includes/home-status.html b/docs/_includes/home-status.html new file mode 100644 index 0000000..4ecacae --- /dev/null +++ b/docs/_includes/home-status.html @@ -0,0 +1,30 @@ +{% assign status = include.status %} +{% if status.eyebrow and status.title and status.items.size > 0 %} +
+
+

{{ status.eyebrow | escape }}

+

{{ status.title | escape }}

+
+ + {% if status.introduction %} +

{{ status.introduction | escape }}

+ {% endif %} + +
+ {% for item in status.items %} +
+
{{ item.label | escape }}
+
+ {{ item.state | escape }} + {% if item.detail %} + {{ item.detail | escape }} + {% endif %} +
+
+ {% endfor %} +
+
+{% endif %} diff --git a/docs/_layouts/default.html b/docs/_layouts/default.html index dc8ed49..8634d6d 100644 --- a/docs/_layouts/default.html +++ b/docs/_layouts/default.html @@ -7,7 +7,7 @@ {% assign page_description = page.abstract | default: page.description | default: site.description %} - {% if page.title %}{{ page.title | escape }} · {% endif %}{{ site.title | escape }} + {% if page.page_kind == 'home' %}{{ site.title | escape }}{% elsif page.title %}{{ page.title | escape }} · {{ site.title | escape }}{% else %}{{ site.title | escape }}{% endif %} diff --git a/docs/assets/css/site.css b/docs/assets/css/site.css index 0ec94e1..fc281c4 100644 --- a/docs/assets/css/site.css +++ b/docs/assets/css/site.css @@ -5,13 +5,15 @@ --ink: #d8dbdc; --ink-strong: #f1f2f2; --muted: #8c9396; - --faint: #555c5f; + --faint: #747c7f; --rule: #25292b; --rule-strong: #3a4043; --accent: #e3b63f; --accent-cool: #72b7d2; --content-width: 44rem; --reading-width: 42rem; + --presentation-width: 70rem; + --shell-width: 78rem; --gutter: clamp(1.25rem, 4vw, 3rem); --mono: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, "Liberation Mono", monospace; --body: ui-serif, Georgia, Cambria, "Times New Roman", Times, serif; @@ -108,7 +110,7 @@ button:focus-visible { .site-header__inner, .site-footer__inner { - width: min(100% - (2 * var(--gutter)), 78rem); + width: min(100% - (2 * var(--gutter)), var(--shell-width)); margin-inline: auto; } @@ -118,6 +120,7 @@ button:focus-visible { align-items: center; justify-content: space-between; gap: 1.5rem; + padding-block: 0.45rem; } .site-brand { @@ -139,9 +142,11 @@ button:focus-visible { .site-navigation { display: flex; + flex-wrap: wrap; align-items: center; + justify-content: flex-end; gap: clamp(1rem, 3vw, 2.25rem); - font: 0.7rem/1 var(--mono); + font: 0.7rem/1.35 var(--mono); letter-spacing: 0.12em; text-transform: uppercase; } @@ -160,13 +165,22 @@ button:focus-visible { } .article, -.home-hero, .home-notes, -.home-index { +.layout-reading { width: min(100% - (2 * var(--gutter)), var(--content-width)); margin-inline: auto; } +.layout-presentation { + width: min(100% - (2 * var(--gutter)), var(--presentation-width)); + margin-inline: auto; +} + +.layout-shell { + width: min(100% - (2 * var(--gutter)), var(--shell-width)); + margin-inline: auto; +} + .article { padding-block: clamp(4rem, 10vw, 8rem) 6rem; } @@ -258,9 +272,9 @@ h1 { } .article__body > h1:first-child { - max-width: 22ch; + max-width: 24ch; margin-bottom: clamp(3.5rem, 8vw, 6rem); - font-size: clamp(2.4rem, 5.4vw, 4.75rem); + font-size: clamp(2.4rem, 4.8vw, 4rem); } h2 { @@ -545,13 +559,36 @@ img { } .home-notes, -.home-index { +.home-section { padding-block: clamp(5rem, 10vw, 8rem); border-top: 1px solid var(--rule); } -.home-index { - margin-bottom: 4rem; +.home-section:last-child { + margin-bottom: clamp(2rem, 6vw, 5rem); +} + +.home-catalog-section { + padding-block: clamp(4.5rem, 8vw, 6.5rem); +} + +.home-section__prose { + max-width: var(--reading-width); + margin: 0 0 3rem; +} + +.home-section__prose > :first-child { + margin-top: 0; +} + +.home-section__prose > :last-child { + margin-bottom: 0; +} + +.home-catalog-section > .home-section__prose, +.home-visual > .home-section__prose { + max-width: min(var(--reading-width), calc(100% - 11rem)); + margin-left: 11rem; } .section-heading { @@ -569,6 +606,216 @@ img { border: 0; } +.home-visual { + padding-block: clamp(5.5rem, 10vw, 8.5rem); +} + +.home-pipeline { + margin: 0; +} + +.home-pipeline__track, +.home-pipeline__checks ul, +.evidence-band { + display: grid; + margin: 0; + padding: 0; + list-style: none; +} + +.home-pipeline__track { + grid-template-columns: repeat(auto-fit, minmax(min(9rem, 100%), 1fr)); + border-top: 1px solid var(--rule-strong); + border-left: 1px solid var(--rule-strong); +} + +.home-pipeline__stage { + min-width: 0; + padding: 1.25rem; + border-right: 1px solid var(--rule-strong); + border-bottom: 1px solid var(--rule-strong); +} + +.home-pipeline__stage strong, +.home-pipeline__detail, +.home-pipeline__checks li strong, +.evidence-band__detail, +.status-ledger__detail { + display: block; +} + +.home-pipeline__stage strong, +.home-pipeline__checks li strong, +.evidence-band strong, +.status-ledger dt, +.status-ledger dd strong { + color: var(--ink-strong); + font-family: var(--mono); +} + +.home-pipeline__stage strong { + margin-top: 1rem; + font-size: 0.82rem; + line-height: 1.35; +} + +.home-pipeline__detail, +.evidence-band__detail, +.status-ledger__detail { + margin-top: 0.45rem; + color: var(--muted); + font-size: 0.88rem; + line-height: 1.5; +} + +.home-pipeline__index, +.evidence-band__role { + color: var(--faint); + font: 0.64rem/1.2 var(--mono); + letter-spacing: 0.09em; + text-transform: uppercase; +} + +.home-pipeline__checks { + margin-top: 1.5rem; + padding-top: 1rem; + border-top: 1px dashed var(--rule-strong); +} + +.home-pipeline__checks > p { + margin: 0 0 0.85rem; + color: var(--accent-cool); + font: 0.65rem/1.4 var(--mono); + letter-spacing: 0.11em; + text-transform: uppercase; +} + +.home-pipeline__checks ul { + grid-template-columns: repeat(auto-fit, minmax(min(13rem, 100%), 1fr)); + border-top: 1px solid var(--rule); + border-left: 1px solid var(--rule); +} + +.home-pipeline__checks li { + padding: 1rem 1.15rem; + border-right: 1px solid var(--rule); + border-bottom: 1px solid var(--rule); +} + +.home-pipeline__checks li strong { + font-size: 0.76rem; + line-height: 1.4; +} + +.home-pipeline figcaption { + max-width: 58ch; + margin: 1.2rem 0 0 auto; + color: var(--muted); + font-size: 0.92rem; +} + +.home-output { + display: grid; + grid-template-columns: minmax(0, 2.2fr) minmax(15rem, 0.8fr); + margin: 0; + border: 1px solid var(--rule-strong); + background: rgba(11, 13, 14, 0.82); +} + +.home-output__media { + display: grid; + min-height: clamp(20rem, 52vw, 40rem); + place-items: center; + padding: clamp(1.25rem, 4vw, 3rem); +} + +.home-output__media img { + width: 100%; + max-width: 100%; + max-height: 36rem; + border: 0; + image-rendering: pixelated; + object-fit: contain; +} + +.home-output figcaption { + display: flex; + flex-direction: column; + justify-content: flex-end; + margin: 0; + padding: clamp(1.25rem, 3vw, 2.25rem); + border-left: 1px solid var(--rule-strong); + color: var(--muted); +} + +.home-output figcaption p { + margin: 0; +} + +.home-output figcaption .text-link { + align-self: flex-start; +} + +.evidence-band { + grid-template-columns: repeat(auto-fit, minmax(min(12rem, 100%), 1fr)); + border-top: 1px solid var(--rule-strong); + border-left: 1px solid var(--rule-strong); +} + +.evidence-band li { + min-width: 0; + padding: 1.2rem; + border-right: 1px solid var(--rule-strong); + border-bottom: 1px solid var(--rule-strong); +} + +.evidence-band strong, +.evidence-band__detail { + display: block; +} + +.evidence-band strong { + margin-top: 0.85rem; + font-size: 0.8rem; + line-height: 1.4; +} + +.status-ledger { + margin: 0; + border-top: 1px solid var(--rule-strong); +} + +.status-ledger > div { + display: grid; + grid-template-columns: minmax(10rem, 0.8fr) minmax(0, 1.2fr); + gap: 1.5rem; + padding-block: 1rem; + border-bottom: 1px solid var(--rule); +} + +.status-ledger dt, +.status-ledger dd { + margin: 0; +} + +.status-ledger dt { + font-size: 0.76rem; + line-height: 1.45; +} + +.status-ledger dd strong, +.status-ledger__detail { + display: block; +} + +.status-ledger dd strong { + color: var(--accent-cool); + font-size: 0.68rem; + letter-spacing: 0.08em; + line-height: 1.4; + text-transform: uppercase; +} + .post-list { margin: 0; padding: 0; @@ -691,7 +938,7 @@ img { /* Rouge tokens: restrained contrast without a theme dependency. */ .highlight .c, .highlight .c1, -.highlight .cm { color: #6f777a; } +.highlight .cm { color: #737b7e; } .highlight .k, .highlight .kd, .highlight .kn { color: #c69a63; } @@ -704,6 +951,34 @@ img { .highlight .mf { color: #b9a06c; } .highlight .o { color: #aab0b2; } +@media (min-width: 900px) { + .home-catalog-section .document-list a { + grid-template-columns: minmax(0, 1fr) 12rem; + gap: 2rem; + } + + .home-catalog-section .document-list__body { + display: grid; + grid-template-columns: minmax(13rem, 0.8fr) minmax(0, 1.2fr); + gap: 2rem; + } +} + +@media (max-width: 820px) { + .home-output { + grid-template-columns: 1fr; + } + + .home-output__media { + min-height: clamp(16rem, 75vw, 32rem); + } + + .home-output figcaption { + border-top: 1px solid var(--rule-strong); + border-left: 0; + } +} + @media (max-width: 720px) { :root { --gutter: 1.1rem; @@ -724,9 +999,9 @@ img { } .site-navigation { - gap: 0.9rem; - font-size: 0.62rem; - letter-spacing: 0.07em; + gap: 0.8rem; + font-size: 0.68rem; + letter-spacing: 0.055em; } .article { @@ -768,6 +1043,21 @@ img { margin-bottom: 0.8rem; } + .home-catalog-section > .home-section__prose, + .home-visual > .home-section__prose { + max-width: var(--reading-width); + margin-left: 0; + } + + .home-pipeline__track { + grid-template-columns: 1fr; + } + + .status-ledger > div { + grid-template-columns: 1fr; + gap: 0.45rem; + } + .post-list a { grid-template-columns: 1.7rem minmax(0, 1fr); } @@ -808,6 +1098,10 @@ img { } @media (max-width: 430px) { + .site-header__inner { + gap: 0.75rem; + } + .site-brand span { position: absolute; width: 1px; @@ -819,9 +1113,14 @@ img { } .site-navigation { - width: 100%; + flex: 1; + gap: 0.7rem; justify-content: flex-end; } + + .evidence-band { + grid-template-columns: 1fr; + } } @media (prefers-reduced-motion: reduce) { @@ -869,6 +1168,12 @@ img { min-height: 0; } + .layout-reading, + .layout-presentation, + .layout-shell { + width: 100%; + } + a { color: #173f73; text-decoration-color: #173f73; diff --git a/docs/index.md b/docs/index.md index 4dbc002..5747d20 100644 --- a/docs/index.md +++ b/docs/index.md @@ -5,108 +5,169 @@ page_kind: home description: A research software laboratory for finite Wang tilings and inspectable solver design. --- -
+

Finite tilings / algorithm laboratory

- # Tiling Foundry +

Tiling Foundry

- Tiling Foundry turns the Yang–Zhang reduction into an inspectable software - pipeline. Construction, solving, verification, witness correspondence, and - measurement remain separate so that each result can be audited rather than - merely observed. +

+ Tiling Foundry turns the Yang–Zhang reduction into an inspectable software + pipeline. Construction, solving, verification, witness correspondence, and + measurement remain separate so that each result can be audited rather than + merely observed. +

- [Explore the documentation](#documentation){: .text-link } + Explore the documentation
-
+

Reading path

Understand, inspect, verify

- Start with the architecture reference and reduction note to understand the - pipeline. Use implementation contracts and the data contract when inspecting - code or integrations. Use methodology pages before interpreting dated - benchmark, profile, coverage, or fuzzing reports. - - Each catalog entry shows both its document type and status. Current - specifications, contracts, references, notes, and designs describe maintained - behavior within their stated scope. Methodology pages define how evidence is - collected. Dated reports preserve results for a named source state and date. - Historical pages preserve earlier decisions and are not current API - documentation. - - Completed implementation plans remain versioned under `docs/plans/` for - operational history, but are excluded from this public catalog. +
+

+ Start with the architecture reference and reduction note to understand the + pipeline. Use implementation contracts and the data contract when inspecting + code or integrations. Use methodology pages before interpreting dated + benchmark, profile, coverage, or fuzzing reports. +

+ +

+ Each catalog entry shows both its document type and status. Current + specifications, contracts, references, notes, and designs describe maintained + behavior within their stated scope. Methodology pages define how evidence is + collected. Dated reports preserve results for a named source state and date. + Historical pages preserve earlier decisions and are not current API + documentation. +

+ +

+ Completed implementation plans remain versioned under docs/plans/ + for operational history, but are excluded from this public catalog. +

+
-{% assign architecture = site.pages | where: "section", "Architecture and correctness" | sort: "nav_order" %} +{% comment %} +Author-approved narrative can be inserted as layout-reading sections. The wide +components below emit no markup until their complete front-matter data exists. +{% endcomment %} +{% if page.pipeline %} + {% include home-pipeline.html pipeline=page.pipeline %} +{% endif %} +{% if page.featured_output %} + {% include home-output.html output=page.featured_output %} +{% endif %} +{% if page.evidence %} + {% include home-evidence.html evidence=page.evidence %} +{% endif %} +{% if page.implementation_status %} + {% include home-status.html status=page.implementation_status %} +{% endif %} + +{% assign architecture = site.pages + | where: "section", "Architecture and correctness" + | sort: "nav_order" %} {% assign reduction = site.pages | where: "section", "Yang–Zhang reduction" | sort: "nav_order" %} {% assign optimization = site.pages | where: "section", "Solver optimization" | sort: "nav_order" %} -{% assign comparisons = site.pages | where: "section", "Cross-engine benchmarks" | sort: "nav_order" %} +{% assign comparisons = site.pages + | where: "section", "Cross-engine benchmarks" + | sort: "nav_order" %} {% assign historical = site.pages | where: "section", "Historical material" | sort: "nav_order" %} -
+

Start here

Architecture and correctness

- These current references define the software boundaries that keep the - reduction, solver, independent verification, solution transport, and - Boolean–Wang witness correspondence auditable. +

+ These current references define the software boundaries that keep the + reduction, solver, independent verification, solution transport, and + Boolean–Wang witness correspondence auditable. +

{% include document-list.html documents=architecture %}
-
+

Construction

Yang–Zhang reduction

- The technical note distinguishes paper conventions from project conventions. - The builder page is the implementation contract for region geometry, - ownership, and black-box obligations. The bibliography records primary - sources. +

+ The technical note distinguishes paper conventions from project conventions. + The builder page is the implementation contract for region geometry, + ownership, and black-box obligations. The bibliography records primary + sources. +

{% include document-list.html documents=reduction %}
-
+

Measured mechanisms

Solver optimization

- Read the methodology first. The remaining pages are dated profile or - benchmark reports that preserve their corpus, environment, work counters, - timing method, and limitations. +

+ Read the methodology first. The remaining pages are dated profile or + benchmark reports that preserve their corpus, environment, work counters, + timing method, and limitations. +

{% include document-list.html documents=optimization %}
-
+

Native C / Z3

Cross-engine benchmarks

- The current protocol distinguishes solving the same prepared Wang region - from end-to-end decisions that begin with the same formula file. Dated - reports retain the measured source identity and interpretation limits. +

+ The current protocol distinguishes solving the same prepared Wang region + from end-to-end decisions that begin with the same formula file. Dated + reports retain the measured source identity and interpretation limits. +

{% include document-list.html documents=comparisons %}
-
+

Design history

Historical material

- Earlier proposals are retained to explain the project’s design trajectory. - They are not current API contracts. +

+ Earlier proposals are retained to explain the project’s design trajectory. + They are not current API contracts. +

{% include document-list.html documents=historical %}
diff --git a/docs/plans/2026-08-25-pages-overview-audit.md b/docs/plans/2026-08-25-pages-overview-audit.md new file mode 100644 index 0000000..e22b8ae --- /dev/null +++ b/docs/plans/2026-08-25-pages-overview-audit.md @@ -0,0 +1,519 @@ +# GitHub Pages Overview Audit and Deferred Work Plan + +**Status:** template foundation in progress; authorial publication deferred + +**Baseline:** `ad561eba1292470738cf3f095b6f057fda2d552e` on 25 August +2026. The public legacy Pages deployment was built from `main:/docs` at this +same commit. + +> This is an internal, non-normative audit and resumption plan. It records a +> proposed direction, not author-approved copy, public claims, roadmap +> commitments, or a requirement to implement every idea below. Unchecked work +> is not a commitment. + +## Executive judgment + +The central diagnosis is correct: the current site is a strong technical +documentation portal, but it is not yet the best conceptual introduction to +the project. The change should be an editorial reordering, not a redesign. + +The original brief is useful as an analysis, with these qualifications: + +- the proposed sequence is a reading goal, not a requirement for seven or more + visibly separate homepage components; +- no narrative page, CTA, or navigation item should be published before the + corresponding author-written content exists; +- the existing documentation catalog should remain automated and intact below + the conceptual overview; +- the current renderer golden is useful development evidence, but it is not a + suitable prominent example of the complete formula-to-image pipeline; +- social metadata, generated-site checks, and measured accessibility issues are + justified work; a deployment migration, SEO plugin, component framework, and + performance rewrite are not; +- sitemap, `robots.txt`, a custom 404 page, and automated public smoke checks + are optional follow-ups, not prerequisites for the information-architecture + change. + +## Audit baseline + +### Public structure + +The homepage currently contains: + +1. a short hero and documentation CTA; +2. a reading-path explanation that sends the reader to architecture and the + reduction note; +3. five full technical catalogs containing 21 public documents. + +The CTA in `docs/index.md` points to `#documentation`. The primary navigation in +`docs/_includes/header.html` exposes Documentation, Benchmarks, and Repository. +There is no conceptual overview, implemented-pipeline figure, real end-to-end +output, or compact implementation-state layer before the catalog. + +The catalog itself is strong and should be preserved: + +- `tools/check_pages.py` requires section, type, status, updated date, permalink, + and navigation order; +- `docs/_includes/document-list.html` presents title, description, type, and + status; +- reference pages retain breadcrumb, description, structured metadata, return + navigation, and a generated H2 table of contents; +- completed plans and the post template are excluded from the public build; +- the catalog is derived from Jekyll page metadata rather than copied by hand. + +### Deployment + +The verified GitHub Pages configuration is: + +- build type: legacy; +- source: `main:/docs`; +- HTTPS enforced; +- status: built; +- public commit at audit time: `ad561eba1292470738cf3f095b6f057fda2d552e`; +- both CI and the automatic `pages-build-deployment` run succeeded. + +The `documentation` CI job runs `make pages-check` and a real Jekyll Pages +build. It does not deploy; GitHub's legacy branch deployment does. There is no +observed stale deployment at this baseline and no reason to migrate deployment +models. + +Representative public routes, all catalog routes, the CSS, Wang JavaScript, +SVG mark, and historical PDF responded successfully during the audit. +`sitemap.xml`, `robots.txt`, and `404.html` were absent. Their absence does not +block the overview work; in particular, no `robots.txt` means there is no local +crawler prohibition. + +## Findings + +### A. Content architecture + +1. **The conceptual layer is missing.** The hero is concise, but the next + section recommends technical references. Documentation discovery therefore + still precedes intuitive understanding. +2. **The primary CTA confirms the old hierarchy.** It sends a new reader to the + documentation catalog rather than to an overview that does not yet exist. +3. **Benchmarks have top-level navigation prominence while conceptual + explanation has none.** This should change only after a real overview anchor + exists, so that the site never publishes a dead or empty navigation target. +4. **Reference and evidence are distinguished in metadata but not fully in the + homepage hierarchy.** Coverage and fuzz reports sit in the broad + Architecture and correctness catalog. Their type/status labels prevent a + false claim, but a later evidence layer should make the distinction easier + to scan. +5. **All homepage catalog sections have similar visual weight.** The catalog is + consequently the dominant public experience even though its internal + taxonomy is good. +6. **A separate `/documentation/` route is not required.** The existing + `/#documentation` target is coherent and is used by breadcrumbs and return + links. It can remain unless a later, evidence-backed need justifies a new + route. +7. **A separate narrative page is also not required initially.** The source + checker currently treats public Markdown pages as cataloged technical + documents. Homepage sections are the smallest safe introduction; a new page + type should be added only if the author later wants a durable standalone + narrative. + +### B. Visual and UX + +The visual identity does not need replacement. Preserve: + +- the dark laboratory palette; +- serif prose and monospace headings/navigation/metadata; +- narrow readable columns; +- the Wang field; +- the understated header and footer; +- the existing catalog, metadata, breadcrumb, TOC, code, and table treatment; +- system fonts and the current responsive philosophy. + +Actual defects and follow-up checks are narrower: + +1. **The deployed homepage has no semantic H1.** The indented Markdown heading + in the HTML section is rendered as `

# Tiling Foundry

`. The public page + therefore begins its heading outline at H2. This is a correctness and + accessibility defect, not a stylistic preference. +2. **The homepage title is duplicated.** The output is + `Tiling Foundry · Tiling Foundry` because the generic title + composition repeats the site name on the home page. +3. **The `--faint` text color is too dim for normal-sized text.** Its measured + contrast against `#070809` is approximately `2.94:1`. It is used for footer + text and small metadata. Other primary colors measured comfortably above + the normal-text threshold. The faint token or its textual uses should be + adjusted without changing the palette's character. +4. **Rouge comment text is marginal for normal text.** `#6f777a` measures about + `4.26:1` on the site background, just below the WCAG AA normal-text ratio. + Link underlines also use a low-contrast rule color, although link text and + underline shape still distinguish the link. Recheck both in the browser and + adjust only the affected tokens if necessary. +5. **New figures and status tables need explicit mobile verification.** Existing + images are responsive and existing tables scroll horizontally; those rules + should be reused before adding new layout behavior. +6. **Do not apply `home-index` blindly to every proposed layer.** Its large + padding plus bottom margin is appropriate for catalogs but would make a + multi-stage narrative unnecessarily long. Add one restrained narrative + spacing variant rather than cards or a replacement layout system. + +### C. Accessibility and performance + +Existing accessibility decisions worth retaining: + +- the skip link is present and becomes visible on focus; +- keyboard focus has a high-contrast visible outline; +- the Wang canvas is `aria-hidden`, non-interactive, and pointer-transparent; +- reduced motion disables smooth scrolling and displays completed tile growth; +- document metadata uses semantic description lists; +- breadcrumbs and navigation regions have labels; +- images are constrained responsively. + +The H1 and faint-text contrast findings above are real issues. The future +pipeline and output figure must use semantic headings, a meaningful figure and +caption relationship, explicit intrinsic dimensions, concise alt text, and no +ARIA where native HTML already expresses the relationship. Because the raster +encodes edge classes primarily through color, its caption, description, or +linked source data must provide a non-color-only way to inspect the result. + +No browser audit tool is installed locally at the baseline, so no Lighthouse, +axe, pa11y, or layout-shift score is claimed. Browser-based accessibility, +keyboard, responsive, and performance measurements remain required during +implementation. + +The current asset and source measurements do not justify rewriting the Wang +field: + +- homepage HTML transferred about 15.9 KiB during the audit; +- `site.css` is 14,767 bytes; +- all site JavaScript source is 36,653 bytes, of which the Wang modules are + 35,934 bytes; +- SVG image source is 1,299 bytes; +- the field caps a plan at 2,400 tiles, caps device pixel ratio at 2, debounces + rebuilds, schedules scroll work through `requestAnimationFrame`, and draws + only visible clusters; +- system fonts avoid a remote font request. + +These are source-level observations, not a performance score. Leave the Wang +field unchanged unless a repeatable browser measurement identifies a real +problem. + +### D. Metadata, links, and deployment checks + +Already correct: + +- `lang="en"`; +- page-specific description fallback; +- absolute canonical URLs; +- SVG favicon; +- `theme-color` and dark color scheme; +- successful production deployment from the expected commit; +- successful source-level catalog and literal-link checks. + +Missing or incomplete: + +1. Open Graph and Twitter/X metadata are absent. +2. There is no selected project-specific social preview image. +3. `tools/check_pages.py` checks sources, not the generated HTML. It cannot + detect the broken home H1, duplicate title, generated link/anchor failures, + missing output routes, or missing social metadata. +4. CI builds the site but does not run a semantic/link smoke over + `build/pages` after Jekyll completes. +5. There is no live post-deployment smoke. This is optional while the legacy + deployment remains demonstrably synchronized. + +## Recommended public information flow + +The exact number of visual sections remains an implementation choice. The +reader should nevertheless encounter this order: + +```text +Hero + -> question and project scope [author] + -> how the implemented pipeline works [author + approved diagram concept] + -> real verified square output [generated artifact + author caption] + -> correctness and evidence boundaries [author] + -> current implementation state [verified facts, not roadmap] + -> technical documentation catalog [existing system] + -> evidence and measurements [links to canonical reports] + -> historical material [existing catalog] +``` + +Question/scope and correctness/evidence may be combined if the final authorial +copy reads better that way. The goal is comprehension before catalog depth, not +a fixed component count. + +Do not publish visible filler copy. If layout work must precede authorial copy, +use short HTML comments or unmistakable development-only placeholders and do +not merge them into the public branch. + +## Real-output decision + +### Existing asset + +The existing files +`tests/fixtures/wang_solution_v1_square_sat.json` and +`renderer/test_data/wang_solution_v1_square_sat.png` are deterministic, +schema-checked, semantically checked, and pixel-golden tested. They are useful +for renderer and layout development. + +They are not the preferred prominent project result: + +- the `Region` and `TilingSolveResult` are constructed manually in + `tests/python/test_wang_solution_export.py`; +- the fixture metadata explicitly says it is not used to establish tiling + correctness; +- it does not begin with a `.cm13` formula or exercise the Yang--Zhang builder + and native solver. + +It must not be captioned as an end-to-end formula-to-image result. + +### Technically preferred candidate + +`tests/instances/pipeline_sat.cm13` already exercises the real parser, +Yang--Zhang builder, native solver, independent native verifier, copied Python +model, independent Python checker, and exporter in tests. The renderer can then +consume the exported `wang-solution-v1` document without loading the core. + +The candidate provenance is: + +```text +tests/instances/pipeline_sat.cm13 + -> solve_native_tiling(..., optimized=true) + -> independent native and Python tiling checks + -> dump_wang_solution(...) + -> renderer/wang_square.py + -> checked-in square PNG +``` + +Before publication, the author must decide whether this small regression input +has suitable scientific and explanatory value. A different input may be +selected, but it must be versioned, pass the same pipeline, and have an +author-approved caption. The paper example is another possible source only if +it is first represented as a reproducible versioned input and verified through +the current pipeline. + +### Reproducibility architecture + +Prefer a checked-in JSON solution and PNG plus a narrow generation command or +script. Record at least the source path and SHA-256, producer path, solver path, +schema name, renderer lock state, output dimensions, and output SHA-256. + +The Pages build should consume the checked-in files. It should not build the C +library, solve a formula, or install Pillow. Regeneration belongs in an explicit +developer command and, if its cost remains small, a focused CI comparison in +the existing root/renderer environments. + +Do not create a second schema, a Pages-only renderer, a fake intermediate +diagram, or any implication of hex output. + +## Verified implementation-state facts + +The homepage may later summarize these facts. Recheck them at the implementation +commit rather than copying this table blindly. + +| Capability | Baseline state | +| --- | --- | +| Yang--Zhang formula-to-region construction | Implemented and tested | +| Reference native solver | Implemented | +| Optimized native solver | Implemented with five isolated mechanisms | +| Independent native verifier | Implemented and required before SAT publication | +| Boolean Z3 oracle | Implemented over the copied immutable formula | +| Wang Z3 oracle | Implemented over copied region and canonical tileset | +| Boolean--Wang witness correspondence | Implemented with exhaustive small-instance evidence | +| Verified square export | Implemented as `wang-solution-v1` | +| Square diagnostic rendering | Implemented in the isolated renderer project | +| Square-to-hex translation and verification | Not implemented; placeholder modules are empty | +| Native C JSON | Not implemented; the translation unit is a placeholder | +| `TaskPlan` and native OpenMP solver | Not implemented; only scaffold/placeholders exist | + +This is a status summary, not a public roadmap. Empirical evidence must link to +its canonical reports and must not be presented as a proof of the theoretical +result. + +## Author input required before publication + +The implementation agent must not write these sections on the author's behalf. +The ranges below are planning estimates, not content requirements. + +| Location / insertion point | Author-provided material | Purpose | Approximate size | Constraints | +| --- | --- | --- | --- | --- | +| `README.md`, opening before Quick start | Final README introduction | Fast repository entry point | 150--300 words | Must remain distinct from the longer Pages narrative | +| `docs/index.md`, after the hero | The question / project overview | Explain why the problem and repository matter | 150--300 words | No slogans, inflated claims, or unexplained proof claim | +| `docs/index.md`, before any result figure | How Tiling Foundry works | Narrative bridge from formula to verified square output | 250--500 words | Keep production stages distinct from independent oracles | +| `docs/index.md`, within or beside How it works | Intuitive Yang--Zhang introduction | Explain the reduction before technical references | 200--450 words | Distinguish theorem, project convention, and implementation | +| `docs/index.md`, correctness/evidence layer | What does Tiling Foundry actually establish? | Bound public claims and evidence | 200--400 words | Empirical tests are not a proof of the theorem | +| Pipeline include or inline figure in `docs/index.md` | Conceptual pipeline design | Define nodes, ordering, branches, and labels | 6--10 named stages plus oracle/future annotations | Z3 paths are independent oracles; hex/OpenMP must be absent or explicitly future | +| `docs/index.md`, real-output figure | Caption and interpretation | Explain what the selected image shows and does not show | 50--120 words plus a short alt-text brief | No fabricated significance and no hex implication | +| README/Pages roadmap location, only if retained | Final public roadmap wording | Authorize any future-facing commitments | 100--200 words or omission | Must not be inferred from internal plans | + +Final CTA and navigation wording also require author confirmation after these +sections exist. Until then, keep the current working links. + +## Public roadmap statements requiring author review + +Do not copy the current README `Next milestones` block onto Pages without +explicit author approval. It presently commits to this order: + +1. square-to-hex formalization, implementation, verification, and renderer + mode; +2. MRV evaluation and a separate hard-UNSAT corpus before parallelism claims; +3. allocation/cleanup hardening before concurrent execution; +4. a minimal serial `TaskPlan` before OpenMP. + +The README also says OpenMP is introduced only after the serial path is correct +and measurable. Technical reports contain narrower conditional statements +about these topics; leave those reports intact. The author must decide whether +the README ordering remains a public commitment and whether Pages should expose +any roadmap at all. + +## Deferred implementation sequence + +Use one documentation staging branch for this work. Do not create a branch per +homepage section. Commit, push, and PR actions still require the authorization +defined by the repository's operating contract. + +### Phase 0: author decisions + +- [ ] Receive or confirm the authorial sections listed above. +- [ ] Confirm whether How it works is a homepage section or a standalone page. + Default recommendation: homepage section. +- [ ] Select the real input/artifact and approve its interpretation. +- [ ] Approve the pipeline topology and the distinction between production, + oracle, evidence, and future stages. +- [ ] Approve final CTA/navigation labels. +- [ ] Approve or omit public roadmap wording. + +Do not begin the public information-architecture switch while these decisions +would leave empty primary navigation or invented copy. + +### Phase 1: semantic and generated-output guardrails + +- [x] Fix the homepage H1 using generated HTML as the acceptance criterion. +- [x] Avoid duplicate site name in the home `` while retaining the + existing page-title format elsewhere. +- [x] Correct faint text contrast conservatively and remeasure affected uses. +- [ ] Recheck Rouge comments and link-decoration contrast in the browser; + change only tokens that fail the selected accessibility criterion. +- [x] Add a standard-library generated-site checker that runs after Jekyll and + verifies representative routes, one H1, title, description, canonical, + internal `href`/`src` targets, anchors, and expected homepage markers. +- [x] Keep `tools/check_pages.py` focused on source/catalog invariants, extending + it only where the chosen homepage architecture changes those invariants. + +The implementation also establishes explicit 44rem reading, 70rem +presentation, and 78rem shell widths. The current catalog uses the wider +presentation container while its prose remains constrained. Conditional +pipeline, output, evidence, and status includes emit no markup until complete +author-approved data is supplied. Browser verification remains open; no visual +claim is derived from the static CSS review. + +### Phase 2: reproducible example + +- [ ] Generate candidate JSON and PNG files in a temporary directory from the + selected versioned input. +- [ ] Verify the native and Python witness checks, schema validation, renderer + output, dimensions, hashes, and absence of hex semantics. +- [ ] Present candidate images and provenance to the author for selection. +- [ ] Add only the selected checked-in solution/image and the narrow + regeneration mechanism. +- [ ] Add a focused reproducibility check if it remains cheap and does not make + the Pages build solve or install renderer dependencies. + +### Phase 3: homepage information architecture + +- [ ] Insert author-provided question/scope, How it works, reduction, and claim + material without rewriting it into marketing copy. +- [ ] Implement the approved pipeline visual with semantic HTML/CSS by default; + use SVG only if relationships cannot remain clear and accessible in HTML. +- [ ] Add the selected real-output figure with author caption and appropriate + alt text, intrinsic dimensions, and a provenance/source-data link. +- [ ] Add a compact implementation-state table from reverified repository facts. +- [ ] Add a concise evidence layer that links to canonical reports and separates + theorem, implementation claim, verification evidence, and measurement. +- [ ] Move the existing catalog below those layers without weakening metadata, + section distinctions, breadcrumbs, TOC, or historical context. +- [ ] Change CTA and navigation only after their targets contain final content. + +Avoid a generic component library. Likely additions are limited to an +understated narrative section convention, one pipeline include, one figure +treatment, and one compact status/evidence treatment. + +### Phase 4: social and document metadata + +- [ ] Add escaped Open Graph title, description, canonical URL, type, and site + name in `docs/_layouts/default.html`. +- [ ] Add equivalent Twitter card metadata. +- [ ] Make the social image conditional on a real selected asset through page + or site configuration; emit no placeholder URL. +- [ ] Verify page descriptions, canonical paths, favicon, `lang`, title + hierarchy, and the selected social image on home and representative documents. +- [ ] Consider sitemap only as an optional small follow-up; do not add Ruby + dependencies or SEO plugins solely for completeness. + +### Phase 5: verification + +- [x] Run source/catalog and generated-site checks. +- [x] Build Pages with the same GitHub Pages action used by CI. +- [x] Run JavaScript syntax checks; no root or renderer implementation changed. +- [ ] Run browser accessibility checks, keyboard navigation, focus, heading, + contrast, alt/caption, reduced-motion, and TOC checks. +- [ ] Run browser performance measurements before and after any justified + performance change. If no Wang-field change is made, report the baseline only. +- [ ] Check large desktop, laptop, tablet, and narrow mobile widths for overflow, + navigation, pipeline, figure, status table, TOC, code, and metadata. +- [ ] Smoke the deployed homepage, a reference page, an evidence page, and their + expected markers after authorized publication. +- [ ] Run the complete repository-required CI gate before merge or direct-main + publication, according to the authorization for that task. + +## Expected file map, subject to author decisions + +- `docs/index.md`: information order, author copy insertion, output figure, + status/evidence summaries, unchanged catalog queries. +- `docs/_includes/header.html`: minimal navigation update after targets exist. +- `docs/_layouts/default.html`: home title handling and social metadata. +- `docs/_includes/`: at most one pipeline include and any genuinely reused + figure/status include. +- `docs/assets/css/site.css`: only styles required by the new narrative, + pipeline, figure, status/evidence layer, and verified contrast fix. +- `docs/assets/images/`: selected real generated output and possibly its + author-approved social composition. +- `docs/assets/data/` or another explicit asset location: selected versioned + solution document if it is useful for provenance/download. +- `tools/`: narrow example generator and generated-site checker if approved. +- `.github/workflows/ci.yml`: invoke generated-site checks after the existing + Jekyll build; do not migrate deployment. +- `README.md`: only the final introduction supplied or approved by the author + and clear cross-links; do not duplicate the Pages homepage. + +## Verification already performed for this audit + +- `make pages-check`: passed for 21 technical documents, the index, five + sections, and all literal internal links known to the checker. +- `node --check` on the current site JavaScript: passed. +- Public GitHub Pages configuration and recent CI/deployment runs: healthy and + synchronized at the baseline commit. +- Public URL smoke: catalog routes and representative assets succeeded. +- Live HTML inspection: exposed the missing semantic home H1 and duplicated + home title. +- Static contrast calculation: exposed the `2.94:1` faint-text token. +- Local tool inventory: Ruby/Jekyll, a browser, Lighthouse, pa11y, and dedicated + link-checker binaries were not available; no result from those tools is + claimed. + +At audit time, no implementation, artifact generation, full project test run, +browser audit, or deployment mutation was performed. The checklists above now +record the subsequent local template work; no public deployment was mutated. + +## Concrete remaining risks + +1. Publishing before author copy exists would replace a coherent documentation + portal with visible scaffolding or invented prose. +2. The small `pipeline_sat.cm13` input may be technically valid but visually or + scientifically weak; only the author can select its public significance. +3. A social crop/composition can hide cells or imply a different result; it + requires author review and traceable provenance. +4. The source-only checker has already missed one production heading defect; + generated HTML checks are required before the homepage grows. +5. New pipeline/status layout may overflow or reorder poorly on mobile; no local + browser measurement exists yet. +6. The legacy Pages deploy is currently healthy, but CI build success alone + does not prove that a later live deployment is current. A lightweight live + smoke becomes worthwhile only if stale publication is observed or the author + wants that operational guarantee. diff --git a/tests/python/test_generated_pages.py b/tests/python/test_generated_pages.py new file mode 100644 index 0000000..7082a73 --- /dev/null +++ b/tests/python/test_generated_pages.py @@ -0,0 +1,111 @@ +from pathlib import Path +import re +import sys +import tempfile +import unittest + + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "tools")) + +from check_generated_pages import REPRESENTATIVE_ROUTES, check_site # noqa: E402 + + +SITE_URL = "https://xtraid.github.io" +BASEURL = "/tiling-foundry" + + +def _html(route: str, title: str, body: str, kind: str = "page") -> str: + return f"""<!doctype html><html lang="en"><head> +<title>{title} + + + + +Skip
{body}
+""" + + +def _write(root: Path, route: str, html: str) -> Path: + path = root / "index.html" if route == "/" else root / route[1:] / "index.html" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(html, encoding="utf-8") + return path + + +def _valid_site(root: Path) -> None: + (root / "assets").mkdir() + (root / "assets/site.css").write_text("body {}\n", encoding="utf-8") + (root / "assets/main.js").write_text("export {};\n", encoding="utf-8") + body = f"""

Tiling Foundry

+
+
+
+Principles""" + _write(root, "/", _html("/", "Tiling Foundry", body, "home")) + for route in REPRESENTATIVE_ROUTES[1:]: + label = route.strip("/").replace("_", " ").title() + body = f'

{label}

Home' + _write(root, route, _html(route, f"{label} · Tiling Foundry", body)) + + +def _contrast(foreground: str, background: str) -> float: + def luminance(color: str) -> float: + channels = [int(color[i : i + 2], 16) / 255 for i in (1, 3, 5)] + linear = [ + value / 12.92 + if value <= 0.04045 + else ((value + 0.055) / 1.055) ** 2.4 + for value in channels + ] + return 0.2126 * linear[0] + 0.7152 * linear[1] + 0.0722 * linear[2] + + light, dark = sorted((luminance(foreground), luminance(background)), reverse=True) + return (light + 0.05) / (dark + 0.05) + + +class GeneratedPagesTests(unittest.TestCase): + def test_accepts_semantic_pages_and_resolved_targets(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + _valid_site(root) + result = check_site(root, site_url=SITE_URL, baseurl=BASEURL) + self.assertEqual(result.errors, ()) + + def test_rejects_the_known_home_regression_and_broken_links(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + _valid_site(root) + home = root / "index.html" + html = home.read_text(encoding="utf-8") + html = html.replace("

Tiling Foundry

", "

# Tiling Foundry

") + html = html.replace( + "Tiling Foundry", + "Tiling Foundry · Tiling Foundry", + ) + html = html.replace( + "", + f'Missing', + ) + home.write_text(html, encoding="utf-8") + result = check_site(root, site_url=SITE_URL, baseurl=BASEURL) + errors = "\n".join(result.errors) + self.assertIn("expected one non-empty H1", errors) + self.assertIn("unexpected home title", errors) + self.assertIn("unresolved href target", errors) + + def test_faint_and_comment_text_meet_normal_text_contrast(self) -> None: + css = (ROOT / "docs/assets/css/site.css").read_text(encoding="utf-8") + root_block = re.search(r":root\s*\{(.*?)\n\}", css, re.DOTALL) + self.assertIsNotNone(root_block) + colors = dict( + re.findall(r"--([\w-]+):\s*(#[0-9a-fA-F]{6});", root_block.group(1)) + ) + comment = re.search(r"\.highlight \.cm \{ color: (#[0-9a-fA-F]{6}); \}", css) + self.assertIsNotNone(comment) + self.assertGreaterEqual(_contrast(colors["faint"], colors["black"]), 4.5) + self.assertGreaterEqual(_contrast(comment.group(1), colors["black-raised"]), 4.5) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/check_generated_pages.py b/tools/check_generated_pages.py new file mode 100644 index 0000000..d22eb46 --- /dev/null +++ b/tools/check_generated_pages.py @@ -0,0 +1,290 @@ +#!/usr/bin/env python3 +"""Check semantics and internal targets in generated GitHub Pages HTML.""" + +from __future__ import annotations + +import argparse +import posixpath +import re +import sys +from dataclasses import dataclass, field +from html.parser import HTMLParser +from pathlib import Path +from urllib.parse import unquote, urljoin, urlsplit + + +ROOT = Path(__file__).resolve().parents[1] +DEFAULT_CONFIG = ROOT / "docs/_config.yml" +REPRESENTATIVE_ROUTES = ( + "/", + "/development_principles/", + "/coverage_baseline_2026-08-22/", + "/historical_architecture/", +) +HOME_IDS = { + "main-content", + "reading-path", + "documentation", + "cross-engine-benchmarks", +} +REFERENCE_ATTRIBUTES = { + "a": "href", + "img": "src", + "link": "href", + "script": "src", + "source": "src", +} + + +@dataclass +class Page: + path: Path + route: str + titles: list[str] = field(default_factory=list) + h1s: list[str] = field(default_factory=list) + descriptions: list[str] = field(default_factory=list) + canonicals: list[str] = field(default_factory=list) + ids: set[str] = field(default_factory=set) + classes: set[str] = field(default_factory=set) + page_kinds: list[str] = field(default_factory=list) + references: list[tuple[str, str]] = field(default_factory=list) + + +@dataclass(frozen=True) +class Result: + errors: tuple[str, ...] + page_count: int + reference_count: int + + +class Parser(HTMLParser): + def __init__(self, page: Page) -> None: + super().__init__(convert_charrefs=True) + self.page = page + self.capture: tuple[str, list[str]] | None = None + + def handle_starttag( + self, tag: str, attrs: list[tuple[str, str | None]] + ) -> None: + tag = tag.lower() + values = {name.lower(): value or "" for name, value in attrs} + if values.get("id"): + self.page.ids.add(values["id"]) + self.page.classes.update(values.get("class", "").split()) + if tag in {"title", "h1"}: + self.capture = (tag, []) + if tag == "meta" and values.get("name", "").lower() == "description": + self.page.descriptions.append(values.get("content", "")) + if tag == "link" and "canonical" in values.get("rel", "").lower().split(): + self.page.canonicals.append(values.get("href", "")) + if tag == "body": + self.page.page_kinds.append(values.get("data-page-kind", "")) + attribute = REFERENCE_ATTRIBUTES.get(tag) + if attribute and attribute in values: + self.page.references.append((attribute, values[attribute])) + + def handle_startendtag( + self, tag: str, attrs: list[tuple[str, str | None]] + ) -> None: + self.handle_starttag(tag, attrs) + + def handle_data(self, data: str) -> None: + if self.capture: + self.capture[1].append(data) + + def handle_endtag(self, tag: str) -> None: + if self.capture and tag.lower() == self.capture[0]: + text = " ".join("".join(self.capture[1]).split()) + target = self.page.titles if tag.lower() == "title" else self.page.h1s + target.append(text) + self.capture = None + + +def _route(root: Path, path: Path) -> str: + relative = path.relative_to(root).as_posix() + if relative == "index.html": + return "/" + if relative.endswith("/index.html"): + return f"/{relative[:-len('index.html')]}" + return f"/{relative}" + + +def _load_pages(root: Path, errors: list[str]) -> dict[str, Page]: + pages: dict[str, Page] = {} + for path in sorted(root.rglob("*.html")): + page = Page(path, _route(root, path)) + try: + parser = Parser(page) + parser.feed(path.read_text(encoding="utf-8")) + parser.close() + except (OSError, UnicodeError) as error: + errors.append(f"{path}: cannot read generated HTML: {error}") + continue + pages[page.route] = page + return pages + + +def _error(errors: list[str], root: Path, page: Page, message: str) -> None: + errors.append(f"{page.path.relative_to(root).as_posix()}: {message}") + + +def _check_semantics( + root: Path, + pages: dict[str, Page], + site_url: str, + baseurl: str, + errors: list[str], +) -> None: + for page in pages.values(): + for label, values in ( + ("H1", page.h1s), + ("title", page.titles), + ("meta description", page.descriptions), + ): + if len(values) != 1 or not values[0].strip(): + _error(errors, root, page, f"expected one non-empty {label}, found {values!r}") + expected = f"{site_url.rstrip('/')}{baseurl}{page.route}" + if page.canonicals != [expected]: + _error( + errors, + root, + page, + f"expected canonical {expected!r}, found {page.canonicals!r}", + ) + + home = pages.get("/") + if not home: + errors.append("missing generated homepage route '/'") + return + if home.titles != ["Tiling Foundry"]: + _error(errors, root, home, f"unexpected home title {home.titles!r}") + if home.h1s != ["Tiling Foundry"]: + _error(errors, root, home, f"unexpected home H1 {home.h1s!r}") + if home.page_kinds != ["home"]: + _error(errors, root, home, f"unexpected home body marker {home.page_kinds!r}") + missing = sorted(HOME_IDS - home.ids) + if missing: + _error(errors, root, home, f"missing home IDs: {', '.join(missing)}") + required_classes = {"layout-reading", "layout-presentation", "home-catalog-section"} + missing_classes = sorted(required_classes - home.classes) + if missing_classes: + _error( + errors, + root, + home, + f"missing home layout classes: {', '.join(missing_classes)}", + ) + if "home-index" in home.classes: + _error(errors, root, home, "legacy home-index layout class is still rendered") + + +def _local_target( + value: str, source_route: str, site_url: str, baseurl: str +) -> tuple[str, str] | None: + origin = urlsplit(site_url) + resolved = urlsplit(urljoin(f"{site_url.rstrip('/')}{baseurl}{source_route}", value)) + if resolved.scheme.lower() in {"data", "javascript", "mailto", "tel"}: + return None + if (resolved.scheme.lower(), resolved.netloc.lower()) != ( + origin.scheme.lower(), + origin.netloc.lower(), + ): + return None + + path = unquote(resolved.path) + if baseurl and path == baseurl: + path = "/" + elif baseurl and path.startswith(f"{baseurl}/"): + path = path[len(baseurl) :] + elif baseurl: + raise ValueError(f"internal URL escapes baseurl {baseurl!r}") + + trailing_slash = path.endswith("/") + path = posixpath.normpath(path) + if trailing_slash and path != "/": + path += "/" + return path, unquote(resolved.fragment) + + +def _check_references( + root: Path, + pages: dict[str, Page], + site_url: str, + baseurl: str, + errors: list[str], +) -> int: + pages_by_path = {page.path.resolve(): page for page in pages.values()} + count = 0 + for page in pages.values(): + for attribute, value in page.references: + count += 1 + try: + target = _local_target(value, page.route, site_url, baseurl) + except ValueError as error: + _error(errors, root, page, f"{attribute}={value!r}: {error}") + continue + if target is None: + continue + path, fragment = target + disk = root / path.lstrip("/") + if path.endswith("/") or disk.is_dir(): + disk /= "index.html" + if not disk.is_file(): + _error(errors, root, page, f"unresolved {attribute} target {value!r}") + elif fragment: + target_page = pages_by_path.get(disk.resolve()) + if not target_page or fragment not in target_page.ids: + _error(errors, root, page, f"unresolved fragment in {value!r}") + return count + + +def check_site(root: Path, *, site_url: str, baseurl: str) -> Result: + root = root.resolve() + if not root.is_dir(): + return Result((f"generated site directory not found: {root}",), 0, 0) + errors: list[str] = [] + pages = _load_pages(root, errors) + if not pages: + return Result(tuple(errors + [f"no HTML pages found under {root}"]), 0, 0) + for route in REPRESENTATIVE_ROUTES: + if route not in pages: + errors.append(f"missing representative route {route!r}") + _check_semantics(root, pages, site_url, baseurl, errors) + references = _check_references(root, pages, site_url, baseurl, errors) + return Result(tuple(errors), len(pages), references) + + +def _config_value(path: Path, key: str) -> str: + pattern = re.compile(rf"^{re.escape(key)}:\s*(.*?)\s*$") + for line in path.read_text(encoding="utf-8").splitlines(): + if match := pattern.match(line): + return match.group(1).strip().strip("'\"") + raise ValueError(f"missing {key!r} in {path}") + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("site_root", type=Path) + parser.add_argument("--config", type=Path, default=DEFAULT_CONFIG) + args = parser.parse_args(sys.argv[1:] if argv is None else argv) + try: + site_url = _config_value(args.config, "url") + baseurl = _config_value(args.config, "baseurl").rstrip("/") + except (OSError, UnicodeError, ValueError) as error: + print(f"ERROR: {error}", file=sys.stderr) + return 1 + + result = check_site(args.site_root, site_url=site_url, baseurl=baseurl) + if result.errors: + for error in result.errors: + print(f"ERROR: {error}", file=sys.stderr) + return 1 + print( + f"Generated Pages checks passed: {result.page_count} HTML pages and " + f"{result.reference_count} href/src references validated." + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())