diff --git a/.claude/CLAUDE.md b/.claude/CLAUDE.md index b2d8930..c89941c 100644 --- a/.claude/CLAUDE.md +++ b/.claude/CLAUDE.md @@ -71,6 +71,7 @@ enlace/ ├── supervise.py # Dev-mode asyncio process supervisor (health checks, restart, logs) ├── diagnose.py # diagnose_app(): scan an app dir for enlace compatibility issues ├── manifest.py # DeployManifest schema + /_meta endpoint + X-Deploy-* headers +├── analytics.py # Opt-in, cookie-free page-view counts: PageViewMiddleware, store, report ├── serve.py # Orchestrates gateway Uvicorn + supervised process-mode children ├── __main__.py # CLI via argh.dispatch_commands ├── __init__.py # Public API facade diff --git a/.claude/skills/enlace/SKILL.md b/.claude/skills/enlace/SKILL.md index 7a34292..d04b366 100644 --- a/.claude/skills/enlace/SKILL.md +++ b/.claude/skills/enlace/SKILL.md @@ -68,6 +68,7 @@ enlace show-config --json # Machine-readable enlace show-config --verbose # Show where each value came from enlace check # Validate config, check route conflicts enlace list-apps # Table: name, route, type, access +enlace analytics # Page views per day/path (opted-in apps) ``` ## Creating an App @@ -197,8 +198,13 @@ access = "public" display_name = "My Custom App" entry_point = "application.py" app_attr = "my_app" + +[analytics] # optional: cookie-free, server-side page-view counts +mode = "privacy" # absent / "none" = nothing recorded ``` +Read analytics with `enlace analytics [--app-name X] [--days 30] [--json]`. Storage, retention and timezone live in `platform.toml`'s `[analytics]` table; see `misc/docs/privacy_analytics.md` in the enlace repo. + For process-mode apps (non-Python or separate process): ```toml mode = "process" diff --git a/README.md b/README.md index 4f24c99..a42c2c6 100644 --- a/README.md +++ b/README.md @@ -157,6 +157,9 @@ enlace diagnose # Analyze an app for enlace compatibility enlace doctor --base-url http://127.0.0.1:8000 # Post-deploy smoke: probe /auth/csrf and every # mounted app; exit nonzero on any failure. +enlace analytics [--app-name kids] [--days 30] [--json] + # Page views per day and per path, for apps that + # opted in to privacy-first analytics. ``` ### Python API @@ -360,6 +363,28 @@ Home-screen icons must be raster: an app whose icon is only an SVG, emoji or monogram gets a favicon but no PNG. `enlace.app_icons.home_screen_gaps(config)` lists those apps. +### Privacy-first analytics + +An app can have page views counted with no cookie, no JavaScript and no third party. It opts in from its own `app.toml`: + +```toml +[analytics] +mode = "privacy" +``` + +The enlace server counts the app's HTML page views as it serves them, and the pages are not changed at all. It stores only daily aggregates per app: views per path (query strings dropped), referrer domain, primary language and device class (mobile/tablet/desktop). Each is kept as a separate count, never crossed with the others. No IP address is read and no visitor identifier is kept. Bots are counted separately. `DNT: 1` and `Sec-GPC: 1` are honoured, and `/_analytics/opt-out` is a page a privacy notice can link to. An app without the table records nothing. + +Read the counts on the serving host with `enlace analytics`, or from Python with `enlace.analytics_report()`. Storage defaults to JSON files under `~/.local/share/enlace/analytics`. It is configured in `platform.toml`, and `build_backend(config, analytics_store=...)` takes any `MutableMapping` (e.g. a `dol` store): + +```toml +[analytics] +store_path = "~/.local/share/enlace/analytics" +retention_days = 395 # at most 750 (under 25 months) +timezone = "Europe/Paris" # whose midnight starts a new day +``` + +The design and how it maps onto the CNIL's consent exemption for audience measurement are in [`misc/docs/privacy_analytics.md`](misc/docs/privacy_analytics.md). That doc also lists what a site operator still has to do: a privacy-notice entry, and handling their own access logs. + ### Deploy manifest (`/_meta`) enlace answers "what is actually deployed?" via an always-on, cheap manifest diff --git a/enlace/__init__.py b/enlace/__init__.py index 7f25d06..2a79e50 100644 --- a/enlace/__init__.py +++ b/enlace/__init__.py @@ -8,6 +8,13 @@ from importlib.metadata import version as _version from pathlib import Path +from enlace.analytics import ( + AppAnalyticsConfig, + JsonFileStore, + PlatformAnalyticsConfig, + analytics_report, + daily_counts, +) from enlace.base import ( AppConfig, AppImportError, @@ -37,6 +44,7 @@ __version__ = "0.0.0+local" __all__ = [ + "AppAnalyticsConfig", "AppConfig", "AppImportError", "BuildConfig", @@ -48,14 +56,18 @@ "DiagnosticReport", "ExternalRef", "Issue", + "JsonFileStore", "MANIFEST_SCHEMA_VERSION", + "PlatformAnalyticsConfig", "PlatformConfig", "Plugin", "ConventionDiscoverer", "EnlaceConfigError", "SourceRef", + "analytics_report", "build_backend", "create_app", + "daily_counts", "diagnose_app", "discover_apps", "load_manifest", diff --git a/enlace/__main__.py b/enlace/__main__.py index 6bb92b6..c5b63c6 100644 --- a/enlace/__main__.py +++ b/enlace/__main__.py @@ -6,6 +6,7 @@ enlace show-config # Show resolved configuration enlace check # Validate configuration enlace list-apps # List discovered apps + enlace analytics # Page-view counts for apps that opted in """ import json as json_module @@ -603,6 +604,60 @@ def app_meta( print() +def analytics( + app_name: str = "", + *, + days: int = 30, + json: bool = False, +): + """Show privacy-first page-view counts: per day and per path, for each app. + + Reads the analytics store named by ``platform.toml``'s ``[analytics]`` table + in the current directory (default ``~/.local/share/enlace/analytics``) — + run it on the host that serves the apps. Only apps whose ``app.toml`` has + ``[analytics] mode = "privacy"`` record anything. + + Args: + app_name: Report this app only (default: every app with data). + days: How many days back, today included. + json: Output the full report (daily series + totals) as JSON. + """ + from enlace.analytics import analytics_report, default_analytics_store + + config = PlatformConfig.from_toml() + store = default_analytics_store(config) + report = analytics_report(app_name, days=days, config=config, store=store) + if json: + print(json_module.dumps(report, indent=2, ensure_ascii=False)) + return + + def shown(value) -> str: + # Stored values are sanitized on write; this also covers older data. + return "".join(ch for ch in str(value) if ch.isprintable()) + + where = "platform.toml" if Path("platform.toml").exists() else "defaults" + print(f"store: {store.root} (from {where})", file=sys.stderr) + if not report: + print("No analytics recorded yet.") + return + for name, data in report.items(): + totals = data["totals"] + views = totals["pageviews"] + print(f"{shown(name)}: {views} page views in the last {days} days") + for day in reversed(data["days"]): + if not day["pageviews"]: + continue + print(f" {day['date']} {day['pageviews']:>6}") + for path, n in sorted(day["paths"].items(), key=lambda kv: -kv[1]): + print(f" {n:>6} {shown(path)}") + for dim in ("referrers", "languages", "devices"): + top = ", ".join(f"{shown(k)} {v}" for k, v in list(totals[dim].items())[:8]) + print(f" {dim}: {top or '(none)'}") + if totals["bot_hits"]: + print(f" bot and scanner hits (not counted above): {totals['bot_hits']}") + print() + + #: The SSOT for the CLI surface: a verb that is not in this list does not exist. COMMANDS = [ serve, @@ -613,6 +668,7 @@ def app_meta( build, diagnose, doctor, + analytics, ] diff --git a/enlace/analytics.py b/enlace/analytics.py new file mode 100644 index 0000000..3fd23ed --- /dev/null +++ b/enlace/analytics.py @@ -0,0 +1,1168 @@ +"""Privacy-first page-view analytics, turned on per app. + +An app opts in from its own ``app.toml``:: + + [analytics] + mode = "privacy" + +Without that table the app records nothing. With it, the enlace server that +already serves the app counts its page views **on the server, from its own +request handling**: no JavaScript, no beacon, no third-party request, and +nothing written to the visitor's device. The page a visitor receives is +byte-for-byte what it would be without analytics. + +What is kept: daily aggregates per app, one counter set per dimension: + +- ``pageviews``: total HTML page views; +- ``paths``: views per page path, relative to the app, with query string and + fragment dropped, and segments that look like an email, a UUID or a token + replaced by ``:id`` (best effort — see :func:`normalize_path`); +- ``referrers``: the referring *host name* only, or ``(direct)`` / + ``(internal)`` / ``(ip)``; +- ``languages``: the primary subtag of ``Accept-Language`` (``fr``, ``en``); +- ``devices``: ``mobile`` / ``tablet`` / ``desktop``; +- ``bot_hits``: page requests from crawlers and vulnerability scanners, + counted apart and nowhere else. + +Dimensions are stored as separate marginal counts and are **never crossed** +(no "path × language × device" table), so a rare combination cannot single +out one visitor on a low-traffic page. No IP address is read, and no +identifier of any kind is stored. Unique visitors are deliberately not +counted: see ``misc/docs/privacy_analytics.md`` for why, and for how this +design maps onto the CNIL's audience-measurement exemption. + +Visitors can object: ``DNT: 1`` and ``Sec-GPC: 1`` are honoured, and +``/_analytics/opt-out`` is a page a privacy notice can link to. It sets a +single first-party opt-out cookie when (and only when) the visitor asks. + +Storage is any ``MutableMapping[str, dict]`` (the ``store`` seam — a ``dol`` +store drops in unchanged); the default is :class:`JsonFileStore` under +``~/.local/share/enlace/analytics``. Each worker process writes only its own +records (``{app}/{day}/{writer}``), so workers never race on a record, and +readers sum the writers. Writes happen off the event loop, in a background +task started with the server; so does the daily maintenance, which purges +records past ``retention_days`` (whether or not anything was viewed that day) +and compacts each finished day's writer records into one. +""" + +import asyncio +import ipaddress +import json +import logging +import os +import re +import tempfile +import threading +import time +import uuid +from collections import defaultdict +from collections.abc import Iterator, MutableMapping +from contextlib import contextmanager, suppress +from datetime import date, datetime, timedelta +from pathlib import Path +from typing import TYPE_CHECKING, Callable, Literal, Optional, Sequence +from urllib.parse import quote, unquote, urlsplit +from zoneinfo import ZoneInfo, ZoneInfoNotFoundError + +from pydantic import BaseModel, ConfigDict, Field, field_validator + +if TYPE_CHECKING: # pragma: no cover + from fastapi import FastAPI + + from enlace.base import AppConfig, PlatformConfig + +_logger = logging.getLogger("enlace.analytics") + +# --------------------------------------------------------------------------- +# Configuration +# --------------------------------------------------------------------------- + +AnalyticsMode = Literal["none", "privacy"] + +#: The CNIL's ceiling for keeping audience-measurement data is 25 months; +#: 750 days stays under it for any 25 consecutive months. +MAX_RETENTION_DAYS = 750 + +#: Where the platform's analytics routes live (the opt-out page). +ANALYTICS_ROUTE_PREFIX = "/_analytics" + + +class AppAnalyticsConfig(BaseModel): + """An app's ``[analytics]`` table in ``app.toml``. Absent means ``none``. + + Strict (unknown keys and modes are errors), so a typo fails at discovery + instead of silently collecting nothing — or something else. + """ + + model_config = ConfigDict(extra="forbid") + + mode: AnalyticsMode = "none" + + @property + def enabled(self) -> bool: + """Whether this app's page views are counted.""" + return self.mode != "none" + + +class PlatformAnalyticsConfig(BaseModel): + """The platform's ``[analytics]`` table in ``platform.toml``. + + Every field has a working default; the table is only needed to change one. + Validated at config load, so ``enlace check`` catches a bad value before a + boot does. + """ + + model_config = ConfigDict(extra="forbid") + + store_path: Optional[Path] = Field( + default=None, + description="Directory of the default JSON store " + "(default: $XDG_DATA_HOME/enlace/analytics, i.e. ~/.local/share/...).", + ) + retention_days: int = Field( + default=395, + ge=1, + le=MAX_RETENTION_DAYS, + description="Days of daily aggregates kept, today included; older ones " + "are purged daily. At most 750 (under the CNIL's 25-month ceiling).", + ) + timezone: str = Field( + default="UTC", description="IANA zone whose midnight starts a new day." + ) + flush_interval_seconds: float = Field( + default=10.0, + ge=0, + description="How often each worker writes its buffered counts, in a " + "background thread (0 = on every view, inline).", + ) + max_values_per_dimension: int = Field( + default=500, + ge=1, + description="Distinct values kept per dimension per day and worker; " + "the rest are counted under '(other)'.", + ) + honor_opt_out_signals: bool = Field( + default=True, description="Skip requests carrying DNT: 1 or Sec-GPC: 1." + ) + exclude_prefixes: tuple[str, ...] = Field( + default=("/_", "/auth/"), + description="Platform paths never attributed to any app.", + ) + opt_out_cookie: str = "enlace_analytics_opt_out" + + @field_validator("timezone") + @classmethod + def _known_timezone(cls, value: str) -> str: + """Refuse an unknown zone here, not as a crash in every worker at boot.""" + try: + ZoneInfo(value) + except (ZoneInfoNotFoundError, ValueError) as exc: + raise ValueError(f"unknown IANA timezone: {value!r}") from exc + return value + + +def default_store_path() -> Path: + """``$XDG_DATA_HOME/enlace/analytics``, else ``~/.local/share/enlace/analytics``.""" + base = os.environ.get("XDG_DATA_HOME") or Path.home() / ".local" / "share" + return Path(base) / "enlace" / "analytics" + + +# --------------------------------------------------------------------------- +# Throttled warnings: a broken disk must not also fill the journal +# --------------------------------------------------------------------------- + +_WARN_EVERY_SECONDS = 3600 + + +class _Throttle: + """Log a warning (with traceback) at most once per key per interval.""" + + def __init__(self, interval: float = _WARN_EVERY_SECONDS): + self._interval = interval + self._last: dict[str, float] = {} + self._suppressed: dict[str, int] = defaultdict(int) + + def warning(self, key: str, msg: str, *args) -> None: + now = time.monotonic() + last = self._last.get(key) + if last is not None and now - last < self._interval: + self._suppressed[key] += 1 + return + extra = self._suppressed.pop(key, 0) + suffix = f" ({extra} similar suppressed)" if extra else "" + self._last[key] = now + _logger.warning(msg + suffix, *args, exc_info=True) + + +# --------------------------------------------------------------------------- +# Storage: the default store (any MutableMapping[str, dict] will do) +# --------------------------------------------------------------------------- + +_KEY_SEGMENT_RE = re.compile(r"^[A-Za-z0-9_.~%-]+$") +_TEMP_PREFIX = ".tmp-" +_STALE_TEMP_SECONDS = 3600 + + +class JsonFileStore(MutableMapping): + """``key -> dict``, one JSON file per key at ``{root}/{key}.json``. + + Keys are ``/``-separated relative paths of URL-safe segments. Writes are + atomic (temp file + ``os.replace``), so a reader never sees half a record. + Stdlib only; swap in any ``MutableMapping`` (e.g. a ``dol`` store over S3) + through the ``store`` argument of :func:`make_analytics` or + ``build_backend``. Two optional extras the maintenance uses when present: + :meth:`exclusive` (a cross-process lock) and :meth:`sweep_temp_files`. + """ + + _SUFFIX = ".json" + + def __init__(self, root: Path | str): + self.root = Path(root).expanduser() + + def _path(self, key: str) -> Path: + parts = key.split("/") + if not all(_KEY_SEGMENT_RE.match(p) and p not in (".", "..") for p in parts): + raise KeyError(f"invalid key: {key!r}") + return self.root.joinpath(*parts).with_name(parts[-1] + self._SUFFIX) + + def __getitem__(self, key: str) -> dict: + try: + return json.loads(self._path(key).read_text(encoding="utf-8")) + except FileNotFoundError: + raise KeyError(key) from None + + def __setitem__(self, key: str, value: dict) -> None: + path = self._path(key) + path.parent.mkdir(parents=True, exist_ok=True) + fd, tmp = tempfile.mkstemp( + dir=path.parent, prefix=_TEMP_PREFIX, suffix=self._SUFFIX + ) + try: + with os.fdopen(fd, "w", encoding="utf-8") as f: + json.dump(value, f, sort_keys=True) + os.replace(tmp, path) + except BaseException: + with suppress(OSError): + os.unlink(tmp) + raise + + def __delitem__(self, key: str) -> None: + path = self._path(key) + try: + path.unlink() + except FileNotFoundError: + raise KeyError(key) from None + with suppress(OSError): + path.parent.rmdir() # drop the day directory once it is empty + + def __iter__(self) -> Iterator[str]: + """Keys, lazily (so ``next(iter(store))`` does not walk the whole tree).""" + if not self.root.is_dir(): + return + for dirpath, dirnames, filenames in os.walk(self.root): + dirnames[:] = [d for d in dirnames if not d.startswith(".")] + rel = Path(dirpath).relative_to(self.root) + for name in filenames: + if name.endswith(self._SUFFIX) and not name.startswith("."): + yield (rel / name[: -len(self._SUFFIX)]).as_posix() + + def __len__(self) -> int: + return sum(1 for _ in self) + + @contextmanager + def exclusive(self) -> Iterator[bool]: + """Try to take the store-wide maintenance lock; yield whether we got it. + + Non-blocking: a worker that loses the race skips this round of + maintenance, it does not wait. ``False`` where ``fcntl`` is unavailable. + """ + try: + import fcntl + except ImportError: # pragma: no cover - Windows + yield False + return + self.root.mkdir(parents=True, exist_ok=True) + with open(self.root / ".maintenance.lock", "w") as f: + try: + fcntl.flock(f, fcntl.LOCK_EX | fcntl.LOCK_NB) + except OSError: + yield False + return + try: + yield True + finally: + fcntl.flock(f, fcntl.LOCK_UN) + + def sweep_temp_files(self, *, older_than: float = _STALE_TEMP_SECONDS) -> int: + """Delete temp files a killed write left behind; return how many.""" + if not self.root.is_dir(): + return 0 + cutoff = time.time() - older_than + removed = 0 + for path in self.root.rglob(f"{_TEMP_PREFIX}*"): + with suppress(OSError): + if path.stat().st_mtime < cutoff: + path.unlink() + removed += 1 + return removed + + +#: The writer id of a finished day's compacted record. +MERGED = "merged" + + +def _app_segment(app: str) -> str: + """An app name as one key segment (percent-encoded: ``café`` → ``caf%C3%A9``).""" + return quote(app, safe="") + + +def _record_key(app: str, day: str, writer: str) -> str: + return f"{_app_segment(app)}/{day}/{writer}" + + +def _split_key(key: str) -> Optional[tuple[str, str, str]]: + """``(app, day, writer)`` from a record key, or ``None`` for anything else.""" + parts = key.split("/") + if len(parts) != 3: + return None + return unquote(parts[0]), parts[1], parts[2] + + +# --------------------------------------------------------------------------- +# Classifying a request (pure functions) +# --------------------------------------------------------------------------- + +OTHER = "(other)" +DIRECT = "(direct)" +INTERNAL = "(internal)" +IP = "(ip)" +UNKNOWN = "(unknown)" +REDACTED = ":id" + +_BOT_RE = re.compile( + r"bot|crawl|spider|slurp|scrap|curl|wget|python-|httpx|go-http|java/|" + r"headless|lighthouse|pingdom|monitor|preview|facebookexternalhit|embedly", + re.IGNORECASE, +) +_TABLET_RE = re.compile( + r"ipad|tablet|kindle|silk|playbook|android(?!.*mobile)", re.IGNORECASE +) +_MOBILE_RE = re.compile( + r"mobi|iphone|ipod|android|blackberry|opera mini|iemobile", re.IGNORECASE +) +_LANG_RE = re.compile(r"^[a-z]{2,3}$") +_LABEL = r"[a-z0-9]([a-z0-9-]*[a-z0-9])?" +_HOST_RE = re.compile(rf"^{_LABEL}(\.{_LABEL})*$") +_UUID_RE = re.compile( + r"[0-9a-f]{8}-?[0-9a-f]{4}-?[0-9a-f]{4}-?[0-9a-f]{4}-?[0-9a-f]{12}" +) +_TOKEN_PIECE_RE = re.compile(r"(?=.*[A-Za-z])(?=.*[0-9])[A-Za-z0-9]{16,}") +_HEX_PIECE_RE = re.compile(r"[0-9a-fA-F]{16,}") +_PAGE_EXTENSIONS = (".html", ".htm") +_MAX_PATH_LENGTH = 200 + + +def is_page_request(method: str, headers: dict[str, str]) -> bool: + """Whether a request is a browser loading a page (not an asset, API or prefetch). + + Modern browsers say so directly (``Sec-Fetch-Dest: document``); older ones + are recognised by ``Accept: text/html``. Prefetches and prerenders are not + views. + """ + if method != "GET": + return False + purpose = headers.get("sec-purpose", "") + headers.get("purpose", "") + if "prefetch" in purpose.lower(): + return False + dest = headers.get("sec-fetch-dest") + if dest is not None: + return dest == "document" + return "text/html" in headers.get("accept", "").lower() + + +def is_page_response(status: int, headers: dict[str, str]) -> bool: + """Whether a response delivered a page: a 200 HTML body, or a 304 revalidation.""" + if status == 304: + return True + return status == 200 and headers.get("content-type", "").lower().startswith( + "text/html" + ) + + +def looks_like_probe(path: str) -> bool: + """A scanner's request, not a page: a dot-segment or a non-HTML file name. + + An SPA answers ``/app/wp-admin/setup.php`` or ``/app/.env`` with its + ``index.html``, so a browser-like scanner would otherwise pass as a reader. + """ + segments = [s for s in path.split("/") if s] + if any(s.startswith(".") for s in segments): + return True + last = segments[-1] if segments else "" + return "." in last and not last.lower().endswith(_PAGE_EXTENSIONS) + + +def has_opted_out(headers: dict[str, str], *, cookie_name: str) -> bool: + """``DNT: 1``, ``Sec-GPC: 1``, or the platform's opt-out cookie.""" + if headers.get("dnt") == "1" or headers.get("sec-gpc") == "1": + return True + for part in headers.get("cookie", "").split(";"): + name, _, value = part.strip().partition("=") + if name == cookie_name and value == "1": + return True + return False + + +def device_class(user_agent: str, *, client_hint_mobile: str = "") -> str: + """``bot``, ``tablet``, ``mobile`` or ``desktop``, from the User-Agent alone.""" + if not user_agent or _BOT_RE.search(user_agent): + return "bot" + if _TABLET_RE.search(user_agent): + return "tablet" + if client_hint_mobile == "?1" or _MOBILE_RE.search(user_agent): + return "mobile" + return "desktop" + + +def primary_language(accept_language: str) -> str: + """Primary subtag of the first ``Accept-Language`` entry (``fr-FR`` → ``fr``).""" + first = accept_language.split(",", 1)[0].split(";", 1)[0].strip().lower() + tag = first.split("-", 1)[0] + return tag if _LANG_RE.match(tag) else UNKNOWN + + +def referrer_domain(referer: str, *, host: str) -> str: + """The referring host name only. + + ``(direct)`` if none, ``(internal)`` if this site, ``(ip)`` for an address + (it may be a person's own machine), ``(unknown)`` for anything that is not + a plain host name. + """ + if not referer: + return DIRECT + try: + hostname = urlsplit(referer).hostname + except ValueError: + return UNKNOWN + if not hostname: + return UNKNOWN + own = host.rsplit(":", 1)[0].lower() if host else "" + if hostname == own: + return INTERNAL + with suppress(ValueError): + ipaddress.ip_address(hostname) + return IP + try: + hostname = hostname.encode("idna").decode("ascii") + except UnicodeError: + return UNKNOWN + return hostname if _HOST_RE.match(hostname) else UNKNOWN + + +def _redacted_segment(segment: str) -> str: + """The segment, or ``:id`` if it looks like an identifier someone could own.""" + if "@" in segment or _UUID_RE.search(segment.lower()): + return REDACTED + for piece in re.split(r"[-_.~]", segment): + if _TOKEN_PIECE_RE.fullmatch(piece) or _HEX_PIECE_RE.fullmatch(piece): + return REDACTED + return segment + + +def normalize_path(path: str) -> str: + """A page path fit to store and to print. + + ``index.html`` folds into its directory; non-printable characters (terminal + escapes) are dropped; segments that look like an email, a UUID or a long + token become ``:id``; the result is capped at 200 characters. Redaction is + best effort: an app that puts personal data in its URL paths should not + turn analytics on. + """ + if path.endswith("/index.html"): + path = path[: -len("index.html")] + path = "".join(ch for ch in path if ch.isprintable()) + path = "/".join(_redacted_segment(s) if s else s for s in path.split("/")) + return (path or "/")[:_MAX_PATH_LENGTH] + + +# --------------------------------------------------------------------------- +# Counting +# --------------------------------------------------------------------------- + +DIMENSIONS = ("paths", "referrers", "languages", "devices") + +#: How often the background task wakes when nothing sets a shorter interval. +_IDLE_TICK_SECONDS = 60.0 + + +def _empty_record() -> dict: + return {"pageviews": 0, "bot_hits": 0, **{d: {} for d in DIMENSIONS}} + + +def _copy_record(rec: dict) -> dict: + return {**rec, **{d: dict(rec[d]) for d in DIMENSIONS}} + + +def _merge(into: dict, rec: dict) -> None: + into["pageviews"] += rec.get("pageviews", 0) + into["bot_hits"] += rec.get("bot_hits", 0) + for dim in DIMENSIONS: + for value, n in (rec.get(dim) or {}).items(): + into[dim][value] = into[dim].get(value, 0) + n + + +class PageViewCounter: + """Buffers one worker's daily aggregates and maintains the ``store``. + + Each worker writes only under its own ``writer`` id (derived from its + process id at first use, so workers forked from one preloaded app still + differ), overwriting its own cumulative record for the day: no worker ever + read-modify-writes a record another is writing. + + Two schedules. :meth:`tick` — run by the server's background task, in a + thread — flushes every ``flush_interval_seconds`` and runs :meth:`maintain` + once per day. Without that task (``background=False``: a bare ASGI host, a + test client outside ``with``), :meth:`record_view` flushes inline instead. + """ + + def __init__( + self, + store: MutableMapping, + *, + retention_days: int = 395, + timezone: str = "UTC", + flush_interval_seconds: float = 10.0, + max_values_per_dimension: int = 500, + writer: Optional[str] = None, + clock: Callable[[], float] = time.time, + ): + self.store = store + self.retention_days = retention_days + self._tz = ZoneInfo(timezone) + self.flush_interval = flush_interval_seconds + self._max_values = max_values_per_dimension + self._fixed_writer = writer + self._writer: Optional[tuple[int, str]] = None + self._clock = clock + self._lock = threading.Lock() + self._records: dict[tuple[str, str], dict] = {} + self._dirty: set[tuple[str, str]] = set() + self._last_flush = clock() + self._maintained_on: Optional[str] = None + self._warn = _Throttle() + self.background = False + + @property + def writer(self) -> str: + """This process's writer id: fixed if given, else ``{pid}-{random}``.""" + if self._fixed_writer: + return self._fixed_writer + pid = os.getpid() + if self._writer is None or self._writer[0] != pid: + self._writer = (pid, f"{pid}-{uuid.uuid4().hex[:8]}") + return self._writer[1] + + def today(self) -> str: + """The current day, ISO format, in the configured timezone.""" + return datetime.fromtimestamp(self._clock(), tz=self._tz).date().isoformat() + + def record_view( + self, + app: str, + *, + path: str, + referrer: str, + language: str, + device: str, + ) -> None: + """Count one page view of ``app`` (a ``bot`` device counts as a bot hit).""" + with self._lock: + rec = self._record_for(app) + if device == "bot": + rec["bot_hits"] += 1 + else: + rec["pageviews"] += 1 + for dim, value in zip(DIMENSIONS, (path, referrer, language, device)): + self._bump(rec[dim], value) + if not self.background or self.flush_interval == 0: + self.maybe_flush() + + def record_bot_hit(self, app: str) -> None: + """Count a request from a crawler or scanner, and nothing else about it.""" + with self._lock: + self._record_for(app)["bot_hits"] += 1 + if not self.background or self.flush_interval == 0: + self.maybe_flush() + + def _record_for(self, app: str) -> dict: + key = (app, self.today()) + self._dirty.add(key) + if key not in self._records: + self._records[key] = _empty_record() + return self._records[key] + + def _bump(self, counts: dict, value: str) -> None: + if value not in counts and len(counts) >= self._max_values: + value = OTHER + counts[value] = counts.get(value, 0) + 1 + + def maybe_flush(self) -> None: + """Flush if the flush interval has elapsed since the last one.""" + if self._clock() - self._last_flush >= self.flush_interval: + self.flush() + + def flush(self) -> None: + """Write every changed record, and forget days that are over.""" + with self._lock: + today = self.today() + pending = {k: _copy_record(self._records[k]) for k in self._dirty} + self._dirty.clear() + self._last_flush = self._clock() + # A finished day's final state is in ``pending`` (or already written). + for key in [k for k in self._records if k[1] != today]: + del self._records[key] + for (app, day), rec in pending.items(): + try: + self.store[_record_key(app, day, self.writer)] = rec + except Exception: # analytics must never break serving + self._warn.warning( + "write", "analytics: could not write %s/%s", app, day + ) + + def tick(self) -> None: + """One background round: flush when due, maintain once per day.""" + self.maybe_flush() + today = self.today() + if self._maintained_on != today: + self.maintain(today=today) + + def maintain(self, *, today: Optional[str] = None) -> None: + """Purge expired records; compact finished days; sweep stale temp files. + + Purging is idempotent, so every worker may do it. Compaction is not, so + it runs only under the store's ``exclusive()`` lock, and only on days at + least two days old, which no writer touches any more. + """ + today = today or self.today() + try: + self.purge_expired(today=today) + lock = getattr(self.store, "exclusive", None) + if lock is not None: + with lock() as got_it: + if got_it: + before = date.fromisoformat(today) - timedelta(days=1) + self.compact(before=before.isoformat()) + sweep = getattr(self.store, "sweep_temp_files", None) + if sweep is not None: + sweep() + except Exception: + self._warn.warning("maintain", "analytics: daily maintenance failed") + return + self._maintained_on = today + + def purge_expired(self, *, today: Optional[str] = None) -> int: + """Delete records older than the retention window; return how many. + + The window is ``retention_days`` days, today included. + """ + today = today or self.today() + first_kept = date.fromisoformat(today) - timedelta(days=self.retention_days - 1) + cutoff = first_kept.isoformat() + removed = 0 + for key in list(self.store): + parts = _split_key(key) + if parts is not None and parts[1] < cutoff: + with suppress(KeyError): # another worker got there first + del self.store[key] + removed += 1 + return removed + + def compact(self, *, before: str) -> int: + """Fold each day's writer records (days before ``before``) into one. + + Every worker restart starts a new writer record, so a finished day can + hold dozens of small files; this leaves one per app per day. Crash-safe: + the merged record lists its ``sources``, so a source that survived an + interrupted run is deleted without being counted twice. Call only under + the store's exclusive lock. Returns how many records were folded. + """ + groups: dict[tuple[str, str], list[str]] = defaultdict(list) + for key in list(self.store): + parts = _split_key(key) + if parts is not None and parts[1] < before and parts[2] != MERGED: + groups[(parts[0], parts[1])].append(key) + folded = 0 + for (app, day), keys in groups.items(): + merged_key = _record_key(app, day, MERGED) + try: + merged = self.store[merged_key] + except KeyError: + merged = _empty_record() + sources = set(merged.get("sources", [])) + merged = {**_empty_record(), **merged} + for key in keys: + if key in sources: + continue + try: + _merge(merged, self.store[key]) + except KeyError: + continue + sources.add(key) + folded += 1 + merged["sources"] = sorted(sources) + self.store[merged_key] = merged + for key in keys: + with suppress(KeyError): + del self.store[key] + return folded + + async def run_background(self) -> None: + """The server-lifetime loop: :meth:`tick` in a thread, forever.""" + interval = self.flush_interval or _IDLE_TICK_SECONDS + self.background = True + try: + while True: + await asyncio.to_thread(self.tick) + await asyncio.sleep(interval) + finally: + self.background = False + + +# --------------------------------------------------------------------------- +# Attribution: which app does a path belong to? +# --------------------------------------------------------------------------- + + +class PageAttribution: + """Maps a request path to ``(app, app-relative path)``, or ``None``. + + Platform paths (``exclude_prefixes``) never count. Otherwise the longest + matching app mount (``/{name}/`` or its route prefix) wins, whether or not + that app opted in — a page of an app that did not must never fall through + to one that did. Anything else goes to the landing app, if it opted in. + """ + + def __init__( + self, + apps: Sequence["AppConfig"], + *, + landing_app: Optional[str] = None, + exclude_prefixes: Sequence[str] = (), + ): + self.enabled = frozenset(a.name for a in apps if a.analytics.enabled) + prefixes = { + (prefix, a.name) + for a in apps + for prefix in (f"/{a.name}/", a.route_prefix.rstrip("/") + "/") + if prefix != "/" + } + self._prefixes = sorted(prefixes, key=lambda p: -len(p[0])) + self._excluded = tuple(exclude_prefixes) + self._landing = landing_app if landing_app in self.enabled else None + # An app mounted at "/" (route = "/") is the fallback, like a landing app. + for a in apps: + if a.route_prefix.rstrip("/") == "" and a.name in self.enabled: + self._landing = self._landing or a.name + + def __call__(self, path: str) -> Optional[tuple[str, str]]: + if path.startswith(self._excluded): + return None + for prefix, name in self._prefixes: + if path.startswith(prefix): + if name not in self.enabled: + return None + return name, "/" + path[len(prefix) :] + if self._landing: + return self._landing, path + return None + + +# --------------------------------------------------------------------------- +# Collecting: the middleware +# --------------------------------------------------------------------------- + + +def _headers(raw: Sequence[tuple[bytes, bytes]]) -> dict[str, str]: + """Lowercased header dict; repeated headers joined (cookies with ``; ``).""" + out: dict[str, str] = {} + for k, v in raw: + name, value = k.decode("latin-1").lower(), v.decode("latin-1") + if name in out: + out[name] += ("; " if name == "cookie" else ", ") + value + else: + out[name] = value + return out + + +class PageViewMiddleware: + """Pure-ASGI middleware counting page views of the apps that opted in. + + It only observes: the request and response pass through unchanged, and + nothing is added to the page. It also owns the counter's lifetime: the + background flush/maintenance task starts with the server's lifespan and + the last flush happens at shutdown. + """ + + def __init__( + self, + app, + *, + counter: PageViewCounter, + attribute: Callable[[str], Optional[tuple[str, str]]], + settings: Optional[PlatformAnalyticsConfig] = None, + ): + self.app = app + self._counter = counter + self._attribute = attribute + self._settings = settings or PlatformAnalyticsConfig() + self._task: Optional[asyncio.Task] = None + + async def __call__(self, scope, receive, send): + if scope["type"] == "lifespan": + await self.app(scope, self._lifespan_receive(receive), send) + return + if scope["type"] != "http": + await self.app(scope, receive, send) + return + target = self._attribute(scope.get("path", "")) + headers = _headers(scope.get("headers", [])) if target else {} + if ( + target is None + or not is_page_request(scope.get("method", ""), headers) + or ( + self._settings.honor_opt_out_signals + and has_opted_out(headers, cookie_name=self._settings.opt_out_cookie) + ) + ): + await self.app(scope, receive, send) + return + + async def observing_send(message): + if message["type"] == "http.response.start": + self._maybe_count(target, headers, message) + await send(message) + + await self.app(scope, receive, observing_send) + + def _maybe_count(self, target, headers, message) -> None: + try: + if not is_page_response(message["status"], _headers(message["headers"])): + return + app, path = target + device = device_class( + headers.get("user-agent", ""), + client_hint_mobile=headers.get("sec-ch-ua-mobile", ""), + ) + if device == "bot" or looks_like_probe(path): + self._counter.record_bot_hit(app) + return + self._counter.record_view( + app, + path=normalize_path(path), + referrer=referrer_domain( + headers.get("referer", ""), host=headers.get("host", "") + ), + language=primary_language(headers.get("accept-language", "")), + device=device, + ) + except Exception: # counting must never break a page + _logger.warning("analytics: failed to count a page view", exc_info=True) + + def _lifespan_receive(self, receive): + async def wrapped(): + message = await receive() + kind = message.get("type") + if kind == "lifespan.startup" and self._task is None: + self._task = asyncio.get_running_loop().create_task( + self._counter.run_background() + ) + elif kind == "lifespan.shutdown": + await self._stop() + return message + + return wrapped + + async def _stop(self) -> None: + if self._task is not None: + self._task.cancel() + with suppress(asyncio.CancelledError): + await self._task + self._task = None + try: + await asyncio.to_thread(self._counter.flush) + except Exception: + _logger.warning("analytics: final flush failed", exc_info=True) + + +# --------------------------------------------------------------------------- +# The opt-out page +# --------------------------------------------------------------------------- + +_OPT_OUT_TEXT = { + "en": { + "title": "Audience statistics", + "about": "This site counts page views anonymously, on its own server: " + "no tracker, no third party, nothing stored about you.", + "out": "You have opted out: your visits are not counted.", + "in": "Your visits are counted, anonymously.", + "do_out": "Don't count my visits", + "do_in": "Count my visits again", + }, + "fr": { + "title": "Statistiques de fréquentation", + "about": "Ce site compte les pages vues de façon anonyme, sur son propre " + "serveur : aucun traceur, aucun tiers, rien n'est conservé sur vous.", + "out": "Vos visites ne sont plus comptées.", + "in": "Vos visites sont comptées, de façon anonyme.", + "do_out": "Ne plus compter mes visites", + "do_in": "Compter à nouveau mes visites", + }, +} + +#: How long the opt-out choice is remembered (13 months, the CNIL maximum). +_OPT_OUT_MAX_AGE = 13 * 30 * 24 * 3600 + + +def _add_opt_out_route(parent: "FastAPI", settings: PlatformAnalyticsConfig) -> None: + """``GET /_analytics/opt-out[?choice=out|in]``: a page a privacy notice links to. + + A plain page with one link, no script. Choosing "out" sets one first-party + cookie (exempt from consent: it stores the visitor's refusal); "in" removes + it. GET, so it works as a plain link and needs no CSRF token. The flip side: + a forged link can switch someone's choice either way; it can reveal nothing, + since the page shows only the visitor's own choice back to them. + """ + import html + + from fastapi import Request + from fastapi.responses import HTMLResponse + + cookie = settings.opt_out_cookie + base = f"{ANALYTICS_ROUTE_PREFIX}/opt-out" + + @parent.get(base, include_in_schema=False) + async def opt_out_page(request: Request, choice: str = "") -> HTMLResponse: + lang = primary_language(request.headers.get("accept-language", "")) + lang = lang if lang in _OPT_OUT_TEXT else "en" + text = {k: html.escape(v) for k, v in _OPT_OUT_TEXT[lang].items()} + opted_out = request.cookies.get(cookie) == "1" + if choice in ("out", "in"): + opted_out = choice == "out" + body = ( + f'' + '' + f"{text['title']}" + '' + f"

{text['title']}

{text['about']}

" + f"

{text['out'] if opted_out else text['in']}

" + f'

' + f"{text['do_in'] if opted_out else text['do_out']}

" + ) + response = HTMLResponse(body, headers={"Cache-Control": "no-store"}) + if choice == "out": + response.set_cookie( + cookie, + "1", + max_age=_OPT_OUT_MAX_AGE, + path="/", + httponly=True, + samesite="lax", + secure=request.url.scheme == "https", + ) + elif choice == "in": + response.delete_cookie(cookie, path="/") + return response + + +# --------------------------------------------------------------------------- +# Wiring (called by build_backend) +# --------------------------------------------------------------------------- + + +class Analytics: + """The analytics feature for one platform: counter, attribution, routes. + + ``collecting`` is False when no app opted in but old data exists: then only + the daily maintenance runs, so retention keeps being enforced after every + app has turned analytics off. + """ + + def __init__( + self, + config: "PlatformConfig", + counter: PageViewCounter, + *, + collecting: bool = True, + ): + self.config = config + self.settings = config.analytics + self.counter = counter + self.collecting = collecting + self.attribute = PageAttribution( + config.apps if collecting else (), + landing_app=config.landing_app, + exclude_prefixes=self.settings.exclude_prefixes, + ) + + def add_routes(self, parent: "FastAPI") -> None: + """Register the opt-out page. Call before any catch-all ``/`` mount.""" + parent.state.analytics = self + if self.collecting: + _add_opt_out_route(parent, self.settings) + + def add_middleware(self, parent: "FastAPI") -> None: + """Install the counting (and lifespan-owning) middleware.""" + parent.add_middleware( + PageViewMiddleware, + counter=self.counter, + attribute=self.attribute, + settings=self.settings, + ) + + +def _has_data(store: MutableMapping) -> bool: + try: + return next(iter(store), None) is not None + except Exception: + return False + + +def make_analytics( + config: "PlatformConfig", *, store: Optional[MutableMapping] = None +) -> Optional[Analytics]: + """The platform's analytics, or ``None`` when there is nothing to do. + + Nothing to do means no app opted in *and* the store holds no data. If data + remains after every app opted out, maintenance alone still runs, so it is + purged on schedule. ``store`` is the storage seam: any + ``MutableMapping[str, dict]``. Default: a :class:`JsonFileStore` at + ``[analytics].store_path`` (or :func:`default_store_path`). + """ + settings = config.analytics + if store is None: + store = JsonFileStore(settings.store_path or default_store_path()) + collecting = any(a.analytics.enabled for a in config.apps) + if not collecting and not _has_data(store): + return None + counter = PageViewCounter( + store, + retention_days=settings.retention_days, + timezone=settings.timezone, + flush_interval_seconds=settings.flush_interval_seconds, + max_values_per_dimension=settings.max_values_per_dimension, + ) + return Analytics(config, counter, collecting=collecting) + + +# --------------------------------------------------------------------------- +# Reading +# --------------------------------------------------------------------------- + + +def _index(store: MutableMapping) -> dict[str, dict[str, list[str]]]: + """``{app: {day: [keys to sum]}}`` from ONE pass over the store's keys. + + A day's merged record replaces the writer records it lists as ``sources``, + so a reader running during compaction never counts a record twice. + """ + index: dict[str, dict[str, list[str]]] = defaultdict(lambda: defaultdict(list)) + for key in store: + parts = _split_key(key) + if parts is not None: + index[parts[0]][parts[1]].append(key) + for days in index.values(): + for day, keys in days.items(): + merged = [k for k in keys if k.rsplit("/", 1)[-1] == MERGED] + if merged: + with suppress(KeyError): + sources = set(store[merged[0]].get("sources", [])) + days[day] = [k for k in keys if k not in sources] + return index + + +def apps_with_data(store: MutableMapping) -> list[str]: + """Names of the apps that have any analytics records.""" + return sorted(_index(store)) + + +def daily_counts( + store: MutableMapping, + app: str, + *, + days: int = 30, + today: Optional[str] = None, + timezone: str = "UTC", + index: Optional[dict] = None, +) -> list[dict]: + """One merged record per day for the last ``days`` days, oldest first. + + Days with no data are included, with zero counts, so the series has no + gaps. Each item is ``{"date": ..., "pageviews": ..., "bot_hits": ..., + "paths": {...}, "referrers": {...}, "languages": {...}, "devices": {...}}``. + Pass ``index`` (from one scan) when reading several apps. + """ + today = today or datetime.now(ZoneInfo(timezone)).date().isoformat() + last = date.fromisoformat(today) + wanted = [(last - timedelta(days=i)).isoformat() for i in range(days - 1, -1, -1)] + app_days = (index if index is not None else _index(store)).get(app, {}) + series = [] + for day in wanted: + record = {"date": day, **_empty_record()} + for key in app_days.get(day, ()): + with suppress(KeyError): # purged while we were reading + _merge(record, store[key]) + series.append(record) + return series + + +def summarize(daily: list[dict]) -> dict: + """Totals over a :func:`daily_counts` series: pageviews and each dimension.""" + total = _empty_record() + for day in daily: + _merge(total, day) + + def ranked(counts: dict) -> dict: + return dict(sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))) + + return { + "pageviews": total["pageviews"], + "bot_hits": total["bot_hits"], + **{dim: ranked(total[dim]) for dim in DIMENSIONS}, + } + + +def default_analytics_store(config: Optional["PlatformConfig"] = None) -> JsonFileStore: + """The default store a platform config points at.""" + settings = config.analytics if config is not None else PlatformAnalyticsConfig() + return JsonFileStore(settings.store_path or default_store_path()) + + +def analytics_report( + app: str = "", + *, + days: int = 30, + config: Optional["PlatformConfig"] = None, + store: Optional[MutableMapping] = None, +) -> dict: + """Per-app report for the last ``days`` days: daily series + totals. + + With no ``app``, reports every app that has data. ``config`` defaults to + ``platform.toml`` in the current directory (only its ``[analytics]`` table + is used; no app is imported). + """ + if config is None: + from enlace.base import PlatformConfig + + config = PlatformConfig.from_toml() + if store is None: + store = default_analytics_store(config) + index = _index(store) + names = [app] if app else sorted(index) + report = {} + for name in names: + series = daily_counts( + store, name, days=days, timezone=config.analytics.timezone, index=index + ) + report[name] = {"days": series, "totals": summarize(series)} + return report diff --git a/enlace/base.py b/enlace/base.py index 4e21c44..fac9aca 100644 --- a/enlace/base.py +++ b/enlace/base.py @@ -26,6 +26,7 @@ from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator +from enlace.analytics import AppAnalyticsConfig, PlatformAnalyticsConfig from enlace.appmeta import AppMetaConfig if sys.version_info >= (3, 11): @@ -155,6 +156,10 @@ class AppConfig(BaseModel): display_name: str = "" provenance: dict[str, str] = Field(default_factory=dict) + # Privacy-first page-view analytics (app.toml's [analytics] table). Off + # unless the app opts in with mode = "privacy". See enlace.analytics. + analytics: AppAnalyticsConfig = Field(default_factory=AppAnalyticsConfig) + # Set only when discovery ran with ``on_import_error="record"`` and this # app's entry module raised on import. ``None`` on every healthy app, and # on every app discovered under the default ``"raise"`` policy (which @@ -326,6 +331,9 @@ def _check_cors_origins(cls, origins: list[str]) -> list[str]: # and `store_path` are carried for the enlace_auth plugin (the editable # overlay's authz + persistence), which enlace core never interprets. app_meta: AppMetaConfig = Field(default_factory=AppMetaConfig) + # Platform-wide analytics settings (platform.toml [analytics]): storage, + # retention, timezone. Apps opt in individually; see enlace.analytics. + analytics: PlatformAnalyticsConfig = Field(default_factory=PlatformAnalyticsConfig) @model_validator(mode="after") def _normalize_dirs(self): @@ -389,6 +397,10 @@ def from_toml(cls, path: Path = Path("platform.toml")) -> "PlatformConfig": app_meta_data = data.get("app_meta") if app_meta_data is not None: platform_data["app_meta"] = app_meta_data + # [analytics] table — storage/retention for per-app analytics. + analytics_data = data.get("analytics") + if analytics_data is not None: + platform_data["analytics"] = analytics_data # Resolve relative path-like fields against the TOML file's own # directory (not the CWD), so the config is host-portable. Done @@ -414,6 +426,10 @@ def _resolve(value: Any) -> Path: if app_meta.get(key): app_meta[key] = _resolve(app_meta[key]) + analytics = platform_data.get("analytics") + if isinstance(analytics, dict) and analytics.get("store_path"): + analytics["store_path"] = _resolve(analytics["store_path"]) + # Environment variable overrides env_apps_dirs = os.environ.get("ENLACE_APPS_DIRS", "") if env_apps_dirs: diff --git a/enlace/compose.py b/enlace/compose.py index 021e5a2..ad9f24e 100644 --- a/enlace/compose.py +++ b/enlace/compose.py @@ -21,7 +21,7 @@ from contextlib import asynccontextmanager from datetime import datetime, timezone from pathlib import Path -from typing import Callable, Optional, Sequence +from typing import Callable, MutableMapping, Optional, Sequence from fastapi import FastAPI, Request from fastapi.responses import HTMLResponse, RedirectResponse, Response @@ -59,7 +59,12 @@ class EnlaceConfigError(RuntimeError): """ -def build_backend(config: PlatformConfig, *, plugins: Sequence[Plugin] = ()) -> FastAPI: +def build_backend( + config: PlatformConfig, + *, + plugins: Sequence[Plugin] = (), + analytics_store: Optional[MutableMapping] = None, +) -> FastAPI: """Compose all app backends into a single ASGI application. For each discovered app: @@ -71,6 +76,10 @@ def build_backend(config: PlatformConfig, *, plugins: Sequence[Plugin] = ()) -> Args: config: Platform configuration with apps already discovered. + plugins: Compose-time plugins, e.g. ``enlace_auth.plugin``. + analytics_store: Where page-view analytics go, for apps that opted in + (any ``MutableMapping[str, dict]``, e.g. a ``dol`` store). Default: + JSON files under ``[analytics].store_path``. See enlace.analytics. Returns: A FastAPI application with all sub-apps mounted. @@ -129,6 +138,15 @@ async def cascade_lifespan(app: FastAPI): # HTML index_page is disabled). _add_apps_listing_route(parent, config) + # Privacy-first analytics, only when some app opted in. Its routes (the + # opt-out page) go in now, before any catch-all "/" mount can shadow them; + # its middleware goes in with the others below. + from enlace.analytics import make_analytics + + analytics = make_analytics(config, store=analytics_store) + if analytics is not None: + analytics.add_routes(parent) + # Deploy manifest endpoints + response headers. Built once at startup; # cheap (one HTTP route per app + a small middleware). Endpoints are # registered BEFORE sub-app mounts so they win the path match. @@ -258,6 +276,12 @@ async def cascade_lifespan(app: FastAPI): is_protected=lambda app: app.access.startswith("protected"), ) + # Count page views of the apps that opted in. It only observes status and + # content type (nothing in the body), so its position is not load-bearing; + # it sits inside GZip only so it sees the same responses the others do. + if analytics is not None: + analytics.add_middleware(parent) + # Compress sizeable text/JSON responses. Added LAST so it is the OUTERMOST # middleware — it must wrap everything downstream (meta injection, sub-app # responses, static files) and see final bytes. diff --git a/enlace/data/skills/enlace/SKILL.md b/enlace/data/skills/enlace/SKILL.md index 65a3797..ae18ae4 100644 --- a/enlace/data/skills/enlace/SKILL.md +++ b/enlace/data/skills/enlace/SKILL.md @@ -59,6 +59,7 @@ enlace show-config --json # Machine-readable enlace show-config --verbose # Show where each value came from enlace check # Validate config, check route conflicts enlace list-apps # Table: name, route, type, access +enlace analytics # Page views per day/path (opted-in apps) ``` ## Creating an App @@ -183,8 +184,13 @@ access = "public" display_name = "My Custom App" entry_point = "application.py" app_attr = "my_app" + +[analytics] # optional: cookie-free, server-side page-view counts +mode = "privacy" # absent / "none" = nothing recorded ``` +Read analytics with `enlace analytics [--app-name X] [--days 30] [--json]`. Storage, retention and timezone live in `platform.toml`'s `[analytics]` table; see `misc/docs/privacy_analytics.md` in the enlace repo. + ### Override Precedence (lowest → highest) ``` diff --git a/enlace/discover.py b/enlace/discover.py index a96050c..829e708 100644 --- a/enlace/discover.py +++ b/enlace/discover.py @@ -433,6 +433,14 @@ def _overlay_toml_fields( fields["build"] = _parse_build_config(build_table, app_dir) provenance["build"] = "override: app.toml [build]" + # [analytics] — strict: a typo'd key or mode fails discovery loudly. + analytics_table = toml_data.get("analytics") + if analytics_table is not None: + from enlace.analytics import AppAnalyticsConfig + + fields["analytics"] = AppAnalyticsConfig.model_validate(analytics_table) + provenance["analytics"] = "override: app.toml [analytics]" + return fields, provenance diff --git a/enlace/tests/test_analytics.py b/enlace/tests/test_analytics.py new file mode 100644 index 0000000..dadedba --- /dev/null +++ b/enlace/tests/test_analytics.py @@ -0,0 +1,652 @@ +"""Privacy-first analytics (issue #55): opt-in per app, counted on the server. + +The acceptance line, piece by piece: + +- an app with ``[analytics] mode = "privacy"`` records a page view; +- an app without the key records nothing; +- the page is unchanged and sets no cookie (the browser-level half of that + check — no request to another origin, nothing in storage — is + ``tests/test_analytics_browser.py``, run in a headless browser); +- the owner can read per-path daily counts for the last 30 days. +""" + +import json +import logging +import os +import subprocess +import sys +import textwrap +import time +from pathlib import Path + +import pytest +from pydantic import ValidationError +from starlette.testclient import TestClient + +from enlace import analytics as an +from enlace.base import PlatformConfig +from enlace.compose import build_backend +from enlace.discover import discover_apps + +BROWSER = { + "accept": "text/html,application/xhtml+xml,*/*;q=0.8", + "sec-fetch-dest": "document", + "user-agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_0) Firefox/130.0", + "accept-language": "fr-FR,fr;q=0.9,en;q=0.8", +} + + +# --------------------------------------------------------------------------- +# Fixtures +# --------------------------------------------------------------------------- + + +def _frontend_app(apps_dir: Path, name: str, analytics_mode: str = "") -> Path: + d = apps_dir / name + (d / "frontend" / "assets").mkdir(parents=True) + (d / "frontend" / "index.html").write_text( + f"{name}" + f"{name}" + ) + (d / "frontend" / "assets" / "app.css").write_text("body{}") + if analytics_mode: + (d / "app.toml").write_text(f'[analytics]\nmode = "{analytics_mode}"\n') + return d + + +@pytest.fixture +def platform(tmp_path): + """Two frontend apps: ``kids`` opted in, ``plain`` not. Instant flushes.""" + apps_dir = tmp_path / "apps" + _frontend_app(apps_dir, "kids", "privacy") + _frontend_app(apps_dir, "plain") + store = an.JsonFileStore(tmp_path / "analytics") + config = discover_apps( + PlatformConfig( + apps_dirs=[apps_dir], + analytics={"flush_interval_seconds": 0}, + ) + ) + return config, store + + +def _client(config, store): + return TestClient(build_backend(config, analytics_store=store)) + + +def _report(store, app, days=30): + return an.summarize(an.daily_counts(store, app, days=days)) + + +# --------------------------------------------------------------------------- +# The acceptance line +# --------------------------------------------------------------------------- + + +def test_opted_in_app_records_a_page_view(platform): + config, store = platform + client = _client(config, store) + assert client.get("/kids/", headers=BROWSER).status_code == 200 + totals = _report(store, "kids") + assert totals["pageviews"] == 1 + assert totals["paths"] == {"/": 1} + assert totals["languages"] == {"fr": 1} + assert totals["devices"] == {"desktop": 1} + assert totals["referrers"] == {an.DIRECT: 1} + + +def test_app_without_the_key_records_nothing(platform): + config, store = platform + client = _client(config, store) + assert client.get("/plain/", headers=BROWSER).status_code == 200 + assert "plain" not in an.apps_with_data(store) + assert _report(store, "plain")["pageviews"] == 0 + + +def test_no_app_opted_in_means_no_analytics_at_all(tmp_path): + """Not even the opt-out route or the middleware exist.""" + _frontend_app(tmp_path / "apps", "plain") + config = discover_apps(PlatformConfig(apps_dirs=[tmp_path / "apps"])) + store = an.JsonFileStore(tmp_path / "analytics") + backend = build_backend(config, analytics_store=store) + assert not hasattr(backend.state, "analytics") + client = TestClient(backend) + assert client.get("/plain/", headers=BROWSER).status_code == 200 + assert client.get("/_analytics/opt-out").status_code == 404 + assert list(store) == [] + + +def test_page_is_unchanged_and_sets_no_cookie(platform, tmp_path): + """Counting adds nothing to the response: same bytes, same headers, no cookie.""" + config, store = platform + with_analytics = _client(config, store).get("/kids/", headers=BROWSER) + plain_config = config.model_copy(deep=True) + for app in plain_config.apps: + app.analytics = an.AppAnalyticsConfig() + without = TestClient(build_backend(plain_config)).get("/kids/", headers=BROWSER) + assert "set-cookie" not in with_analytics.headers + assert with_analytics.content == without.content + assert dict(with_analytics.headers) == dict(without.headers) + + +def test_owner_reads_per_path_daily_counts_for_30_days(platform): + config, store = platform + client = _client(config, store) + for path in ("/kids/", "/kids/lesson/1", "/kids/lesson/1?utm_source=x"): + client.get(path, headers=BROWSER) + series = an.daily_counts(store, "kids", days=30) + assert len(series) == 30 + assert series[-1]["paths"] == {"/": 1, "/lesson/1": 2} # query string dropped + assert all(day["pageviews"] == 0 for day in series[:-1]) + + +# --------------------------------------------------------------------------- +# What is (not) a page view +# --------------------------------------------------------------------------- + + +def test_assets_api_calls_and_prefetches_are_not_page_views(platform): + config, store = platform + client = _client(config, store) + client.get("/kids/assets/app.css", headers={**BROWSER, "sec-fetch-dest": "style"}) + client.get("/kids/", headers={**BROWSER, "sec-fetch-dest": "empty"}) # fetch() + client.get("/kids/", headers={**BROWSER, "sec-purpose": "prefetch"}) + client.head("/kids/", headers=BROWSER) + client.get("/_apps", headers=BROWSER) + assert _report(store, "kids")["pageviews"] == 0 + + +def test_opt_out_signals_are_honoured(platform): + config, store = platform + client = _client(config, store) + client.get("/kids/", headers={**BROWSER, "dnt": "1"}) + client.get("/kids/", headers={**BROWSER, "sec-gpc": "1"}) + client.get("/kids/", headers={**BROWSER, "cookie": "enlace_analytics_opt_out=1"}) + assert _report(store, "kids")["pageviews"] == 0 + + +def test_bots_are_counted_apart(platform): + config, store = platform + client = _client(config, store) + client.get("/kids/", headers={**BROWSER, "user-agent": "Googlebot/2.1"}) + totals = _report(store, "kids") + assert totals["pageviews"] == 0 + assert totals["bot_hits"] == 1 + assert totals["devices"] == {} + + +def test_referrer_is_reduced_to_its_domain(platform): + config, store = platform + client = _client(config, store) + client.get( + "/kids/", + headers={**BROWSER, "referer": "https://www.ecole.fr/classe/42?eleve=bob"}, + ) + client.get("/kids/", headers={**BROWSER, "referer": "http://testserver/kids/"}) + assert _report(store, "kids")["referrers"] == { + "www.ecole.fr": 1, + an.INTERNAL: 1, + } + + +def test_redirects_and_errors_are_not_page_views(platform): + """Only the page the visitor finally sees counts: a 307 hop does not.""" + config, store = platform + client = _client(config, store) + client.get("/kids", headers=BROWSER) # bare prefix: 307 → /kids/ + assert _report(store, "kids")["pageviews"] == 1 + assert not an.is_page_response(404, {"content-type": "text/html"}) + assert not an.is_page_response(307, {}) + + +def test_missing_image_is_not_a_page_view(platform): + config, store = platform + client = _client(config, store) + client.get("/kids/missing.png", headers={**BROWSER, "sec-fetch-dest": "image"}) + assert _report(store, "kids")["pageviews"] == 0 + + +def test_landing_app_gets_root_pages_but_not_platform_or_other_apps(tmp_path): + apps_dir = tmp_path / "apps" + _frontend_app(apps_dir, "home", "privacy") + _frontend_app(apps_dir, "plain") + config = discover_apps( + PlatformConfig( + apps_dirs=[apps_dir], + landing_app="home", + analytics={"flush_interval_seconds": 0}, + ) + ) + store = an.JsonFileStore(tmp_path / "analytics") + client = _client(config, store) + client.get("/", headers=BROWSER) + client.get("/plain/", headers=BROWSER) # another app, not opted in + client.get("/_analytics/opt-out", headers=BROWSER) # platform page + assert _report(store, "home")["paths"] == {"/": 1} + + +# --------------------------------------------------------------------------- +# The opt-out page +# --------------------------------------------------------------------------- + + +def test_opt_out_page_sets_a_cookie_only_when_asked(platform): + config, store = platform + client = _client(config, store) + page = client.get("/_analytics/opt-out", headers=BROWSER) + assert page.status_code == 200 + assert "set-cookie" not in page.headers + assert "Statistiques" in page.text # French for a French browser + assert "", an.UNKNOWN), + ], +) +def test_primary_language(header, expected): + assert an.primary_language(header) == expected + + +# --------------------------------------------------------------------------- +# The CLI +# --------------------------------------------------------------------------- + + +def test_cli_reports_per_path_daily_counts(tmp_path): + store = an.JsonFileStore(tmp_path / "analytics") + c = an.PageViewCounter(store, writer="w", flush_interval_seconds=0) + _view(c, path="/") + _view(c, path="/lesson/1") + (tmp_path / "platform.toml").write_text('[analytics]\nstore_path = "analytics"\n') + runner = textwrap.dedent( + """ + import sys + from enlace.__main__ import main + sys.argv = ["enlace", *sys.argv[1:]] + main() + """ + ) + out = subprocess.run( + [sys.executable, "-c", runner, "analytics", "--json"], + cwd=tmp_path, + capture_output=True, + text=True, + timeout=120, + ) + assert out.returncode == 0, out.stderr + report = json.loads(out.stdout) + assert report["kids"]["totals"]["paths"] == {"/": 1, "/lesson/1": 1} + assert len(report["kids"]["days"]) == 30 + + text = subprocess.run( + [sys.executable, "-c", runner, "analytics"], + cwd=tmp_path, + capture_output=True, + text=True, + timeout=120, + ).stdout + assert "kids: 2 page views in the last 30 days" in text + assert "/lesson/1" in text + + +# --------------------------------------------------------------------------- +# Review findings: background writes, maintenance, compaction, sanitizing +# --------------------------------------------------------------------------- + + +def test_background_task_writes_a_lone_view_without_waiting_for_another(platform): + """A single view reaches the store on the timer, not on the next view.""" + config, store = platform + config.analytics.flush_interval_seconds = 0.05 + with _client(config, store) as client: + client.get("/kids/", headers=BROWSER) + deadline = time.time() + 5 + while not list(store) and time.time() < deadline: + time.sleep(0.02) + assert _report(store, "kids")["pageviews"] == 1 + + +def test_retention_is_enforced_with_no_traffic_and_no_app_opted_in(tmp_path): + """Old data is purged even after every app turned analytics off.""" + _frontend_app(tmp_path / "apps", "plain") + store = an.JsonFileStore(tmp_path / "analytics") + store["kids/2020-01-01/w"] = {"pageviews": 3} + config = discover_apps(PlatformConfig(apps_dirs=[tmp_path / "apps"])) + backend = build_backend(config, analytics_store=store) + assert backend.state.analytics.collecting is False + with TestClient(backend) as client: + assert client.get("/_analytics/opt-out").status_code == 404 + deadline = time.time() + 5 + while list(store) and time.time() < deadline: + time.sleep(0.02) + assert list(store) == [] + + +def test_compaction_folds_writers_once_and_survives_an_interrupted_run(tmp_path): + store = an.JsonFileStore(tmp_path) + old = _day(-3) + store[f"kids/{old}/a"] = {"pageviews": 2, "paths": {"/": 2}} + store[f"kids/{old}/b"] = {"pageviews": 1, "paths": {"/x": 1}} + store[f"kids/{_day(-1)}/a"] = {"pageviews": 5} # too recent to compact + c = an.PageViewCounter(store, clock=_Clock(T0)) + assert c.compact(before=_day(-1)) == 2 + assert sorted(store) == sorted([f"kids/{old}/merged", f"kids/{_day(-1)}/a"]) + + # A crash between writing the merged record and deleting a source must not + # count that source twice, for readers or for the next compaction. + store[f"kids/{old}/a"] = {"pageviews": 2, "paths": {"/": 2}} + [day] = an.daily_counts(store, "kids", days=1, today=old) + assert day["pageviews"] == 3 + c.compact(before=_day(-1)) + [day] = an.daily_counts(store, "kids", days=1, today=old) + assert day["pageviews"] == 3 + assert day["paths"] == {"/": 2, "/x": 1} + + +def test_maintenance_lock_lets_one_worker_compact(tmp_path): + store = an.JsonFileStore(tmp_path) + with store.exclusive() as first: + with an.JsonFileStore(tmp_path).exclusive() as second: + assert (first, second) == (True, False) + + +def test_stale_temp_files_are_swept(tmp_path): + store = an.JsonFileStore(tmp_path) + (tmp_path / "kids").mkdir() + stale = tmp_path / "kids" / ".tmp-abc.json" + stale.write_text("{") + os.utime(stale, (0, 0)) + assert list(store) == [] # never read as a record + assert store.sweep_temp_files() == 1 + assert not stale.exists() + + +def test_app_names_outside_the_key_charset_are_stored_and_read(tmp_path): + apps_dir = tmp_path / "apps" + _frontend_app(apps_dir, "my app", "privacy") + config = discover_apps( + PlatformConfig(apps_dirs=[apps_dir], analytics={"flush_interval_seconds": 0}) + ) + store = an.JsonFileStore(tmp_path / "analytics") + _client(config, store).get("/my app/", headers=BROWSER) + assert an.apps_with_data(store) == ["my app"] + assert _report(store, "my app")["pageviews"] == 1 + + +def test_scanner_probes_count_as_bot_hits_not_paths(platform): + config, store = platform + client = _client(config, store) + for probe in ("/kids/wp-admin/setup.php", "/kids/.env", "/kids/.git/config"): + client.get(probe, headers=BROWSER) + client.get("/kids/lesson/1", headers=BROWSER) + client.get("/kids/index.html", headers=BROWSER) + totals = _report(store, "kids") + assert totals["bot_hits"] == 3 + assert totals["paths"] == {"/lesson/1": 1, "/": 1} + + +@pytest.mark.parametrize( + "raw, stored", + [ + ("/u/alice@example.com/reset", "/u/:id/reset"), + ("/doc/3f2b8c1e-9a4d-4e21-b7c3-0d5e6f7a8b9c", "/doc/:id"), + ("/share/aZ9kQ2xLm8Rt4Vb7Np", "/share/:id"), + ("/lesson/12/fractions-intro", "/lesson/12/fractions-intro"), + ("/\x1b[31mRED\x1b[0m", "/[31mRED[0m"), + ], +) +def test_stored_paths_are_redacted_and_printable(raw, stored): + assert an.normalize_path(raw) == stored + + +@pytest.mark.parametrize( + "referer, expected", + [ + ("http://203.0.113.7/x", an.IP), + ("http://[2001:db8::1]/x", an.IP), + ("https:///", an.UNKNOWN), + ("https://École.fr/", "xn--cole-9oa.fr"), + ], +) +def test_referrer_hosts_are_plain_names_or_placeholders(referer, expected): + assert an.referrer_domain(referer, host="kids.example") == expected + + +def test_unknown_timezone_is_a_config_error_not_a_boot_crash(): + with pytest.raises(ValidationError, match="timezone"): + PlatformConfig(analytics={"timezone": "Europe/Pariss"}) + + +def test_platform_paths_never_count_even_under_an_app_mounted_at_root(tmp_path): + apps_dir = tmp_path / "apps" + d = _frontend_app(apps_dir, "site") + (d / "app.toml").write_text('route = "/"\n[analytics]\nmode = "privacy"\n') + config = discover_apps(PlatformConfig(apps_dirs=[apps_dir])) + attribute = an.PageAttribution( + config.apps, exclude_prefixes=config.analytics.exclude_prefixes + ) + assert attribute("/_analytics/opt-out") is None + assert attribute("/auth/login") is None + assert attribute("/about") == ("site", "/about") + + +def test_writer_id_follows_the_process(): + """Workers forked from one preloaded app must not share a writer id.""" + c = an.PageViewCounter({}) + assert c.writer.startswith(f"{os.getpid()}-") + assert c.writer == c.writer + + +def test_a_broken_store_logs_once_not_on_every_flush(caplog): + class Broken(dict): + def __setitem__(self, key, value): + raise OSError("disk full") + + c = an.PageViewCounter(Broken(), writer="w", flush_interval_seconds=0) + with caplog.at_level(logging.WARNING, logger="enlace.analytics"): + for _ in range(5): + _view(c) + assert len([r for r in caplog.records if "could not write" in r.message]) == 1 diff --git a/misc/docs/privacy_analytics.md b/misc/docs/privacy_analytics.md new file mode 100644 index 0000000..4622ddc --- /dev/null +++ b/misc/docs/privacy_analytics.md @@ -0,0 +1,98 @@ +# Privacy-first analytics + +Design record for [i2mint/enlace#55](https://github.com/i2mint/enlace/issues/55). Module: `enlace/analytics.py`. + +## In one paragraph + +An app turns it on in its own `app.toml` with `[analytics] mode = "privacy"`. The enlace server that already serves the app counts that app's HTML page views as it serves them: no JavaScript, no beacon, no request to any other origin, nothing written to the visitor's device, and the page bytes are unchanged. It keeps only daily aggregates per app, with each dimension (path, referrer domain, primary language, device class) stored as its own count and never crossed with another. No IP address is read and no identifier is stored. The owner reads the counts with `enlace analytics` on the serving host. + +## Why server-side, and not a self-hosted Plausible / Umami / GoatCounter + +The issue asked for this to be weighed first. All three are good tools, and each would be a worse seam here: + +- **They need a script in the page.** Plausible, Umami and GoatCounter all count through a JS snippet, so the page would have to change. Either the app edits its own HTML, which breaks "apps should not need to change", or enlace injects the snippet, which makes enlace a tracker-injector. A children's site promising that nothing runs besides its own code would then be running someone else's JS, even if self-hosted. Counting server-side avoids the question: there is nothing to block and nothing to explain. +- **They are another service to run.** Plausible needs PostgreSQL + ClickHouse. Umami needs PostgreSQL or MySQL. GoatCounter is the light one: a single Go binary with SQLite. Each is still another process, port, backup and upgrade on a host where disk is the binding constraint (see [i2mint/enlace#38](https://github.com/i2mint/enlace/issues/38)). enlace already sees every request, so counting costs one pure-ASGI middleware and a few small JSON files. +- **The seam stays open.** If a real analytics product is wanted later, the collector already produces `(app, path, referrer, language, device)` events at one place, `PageViewCounter.record_view`. Forwarding those to GoatCounter's HTTP API, or writing to a different store (`build_backend(..., analytics_store=...)` takes any `MutableMapping`), is an addition at an existing boundary, not a rewrite. + +## What counts as a page view + +A `GET` that the browser marks as a document navigation (`Sec-Fetch-Dest: document`, or `Accept: text/html` for older browsers), answered with a `200` HTML body or a `304` revalidation. Assets, `fetch()`/XHR calls, `HEAD`, prefetches/prerenders, redirects and errors are not page views. + +For an SPA, a navigation to any unknown path is a view, because the SPA fallback serves `index.html`: the visitor did see a page. Client-side route changes after the first load are not seen by the server and are not counted. That is the price of having no script; a same-origin `navigator.sendBeacon` endpoint could add them later without changing the storage. + +Attribution goes to the longest matching app mount (`/{name}/` or the app's route prefix). Any other path goes to the landing app, if the landing app opted in. Platform paths (`/_…`, `/auth/…`) are never attributed. A page of an app that did not opt in never falls through to an app that did. + +Bots (by User-Agent, including an empty one and `HeadlessChrome`) are counted only as `bot_hits`, never in the dimensions. So are scanner probes: a path with a dot-segment (`/.env`, `/.git/…`) or a non-HTML file name (`/wp-admin/setup.php`), which an SPA would otherwise answer with its `index.html` like a real page. + +Stored values are sanitized, best effort. In paths, non-printable characters are dropped, and segments that look like an email, a UUID or a long token become `:id`. Referrers are reduced to a plain host name: `(ip)` for an address, since it may be a person's own machine, and `(unknown)` for anything that isn't a host name. An app that puts personal data in its URL paths should still not turn analytics on. + +## What is stored, and where + +Key `{app}/{YYYY-MM-DD}/{writer}` → `{"pageviews", "bot_hits", "paths", "referrers", "languages", "devices"}`. The app name is percent-encoded in the key, so any directory name works. + +- **One writer per worker process.** Production runs several workers, so each writes only its own record for the day, overwriting its own cumulative totals. No worker ever read-modify-writes a shared record. Readers sum the writers. The writer id is derived from the process id at first use, so workers forked from one preloaded app (gunicorn `--preload`) still get distinct ids. +- **Written off the event loop.** A background task, started with the server's lifespan, flushes each worker's buffer every `flush_interval_seconds` (default 10) in a thread, and flushes again at shutdown. A lone view is saved on the timer, without waiting for another view. A crash can lose up to one interval of counts. Where no lifespan runs, the flush happens inline on the next view. +- **Bounded.** Each dimension keeps at most `max_values_per_dimension` distinct values per day and writer; the rest go under `(other)`. Paths are also truncated to 200 characters. +- **Compacted.** Every worker start opens a new writer record, so a finished day can accumulate many small files, which matters on a host where disk is short. Once a day, one worker (holding the store's `exclusive()` lock) folds each day that is at least two days old into a single `{app}/{day}/merged` record. The merged record lists its `sources`, so neither an interrupted compaction nor a reader running during one counts anything twice. Stale temp files from killed writes are swept at the same time. +- **Retention.** Records older than `retention_days` (default 395, at most 750, which is under the CNIL's 25 months) are purged daily by the same background task, whether or not anything was viewed. The purge continues after every app has turned analytics off: if the store still holds data, enlace keeps running the maintenance until the data has aged out. +- **Location.** The default is `$XDG_DATA_HOME/enlace/analytics`, i.e. `~/.local/share/enlace/analytics`: outside any app directory, as the "an app dir contains only code + build output" rule requires. Set it with `[analytics] store_path` in `platform.toml`. +- **Day boundary.** `[analytics] timezone` (default `UTC`) sets whose midnight starts a new day. An unknown zone is a config error, caught by `enlace check`, not a crash at boot. + +## Deliberately not counted: unique visitors + +The Plausible/GoatCounter method is a hash of IP + User-Agent with a daily salt that is never stored. With several worker processes, the salt has to be shared for the count to mean anything. That is only possible by storing it somewhere for the day, or by deriving it from a long-lived secret. Deriving it would let anyone holding the secret recompute past salts, which defeats the rotation. The version without sharing (one in-memory salt per worker) over-counts by up to the number of workers. Neither was good enough to ship under a "privacy-first" name, so v1 counts page views only. If uniques are wanted, the decision is where the day's salt may live (a root-only file deleted at midnight is the obvious candidate). That needs an owner's call, not an implementation detail. + +## Opting out + +- `DNT: 1` and `Sec-GPC: 1` are honoured (`honor_opt_out_signals`, on by default). +- `GET /_analytics/opt-out` is a small no-script page, in English or French depending on the browser, meant to be linked from the privacy notice. Choosing "don't count my visits" sets one first-party cookie, `enlace_analytics_opt_out=1` (13 months, `HttpOnly`, `SameSite=Lax`). That cookie stores the visitor's refusal, and the CNIL exempts such cookies from consent. It is set only when the visitor asks. Choosing to be counted again deletes it. + +## CNIL audience-measurement exemption: checklist + +Researched 2026-09-25 against the texts in the references. This maps the design onto the conditions. It is not legal advice, and the operator of each site should confirm it. + +**Is it in scope at all?** No CNIL text says outright whether purely server-side analytics, reading and storing nothing on the device, falls under Art. 82 of the Loi Informatique et Libertés (ePrivacy Art. 5(3)). The EDPB's Guidelines 2/2023 read "gaining access" broadly. They say that relying on protocol headers such as `Accept` or the User-Agent "can lead to the application of Article 5(3)" (§§42–43), and they treat an IP that originates from the terminal the same way (§55) [1]. The safe reading is to meet the CNIL exemption conditions in full, which makes the scope question moot. GDPR applies either way; the legal basis is legitimate interest [2 §52; 3]. + +| Condition (CNIL guidelines 2020-091 §51 [2], recommendation §5 [4], self-assessment tool [5]) | Status | +|---|---| +| Purpose limited to measuring the site's own audience, for the publisher alone | Met: first-party, no marketing features | +| Produces anonymous statistics only; no combination of criteria may isolate one user | Met: dimensions stored as separate marginal counts, never crossed; no identifiers | +| No tracking across sites or apps; no cross-site identifier | Met: no identifier at all | +| Data minimised; headers reduced | Met: device class and primary language subtag only; referrer reduced to its host | +| No campaign/CRM identifiers imported from URLs | Met: query string and fragment dropped from the stored path; identifier-like path segments redacted (best effort) | +| IP used at most for city-level location, then truncated | Met: the client IP is never read; an IP-literal referrer is stored as `(ip)` | +| No session replay, no following one user's navigation | Met | +| Tracker lifetime ≤ 13 months | Met (not applicable): nothing placed on the device; the opt-out cookie lasts 13 months | +| Data retention ≤ 25 months | Met: `retention_days` validated ≤ 750 days, default 395 (13 months); purged daily regardless of traffic | +| No transfer to third parties; processor terms if a vendor is used | Met: no vendor. The hosting provider's own Art. 28 terms are the operator's | +| Users informed, and able to object | Met by enlace + operator: DNT/GPC honoured and `/_analytics/opt-out` exists; the operator must mention the processing in the privacy notice and link that page | +| No cross-referencing with other processing | Operator: do not join analytics with account/auth data or with access logs | + +**What the site operator still has to do.** This is required even when the exemption applies [4 §5; 7]: + +- Add a privacy-notice section: purpose (audience statistics), legal basis (legitimate interest), what is derived (path, referrer domain, language, device class; no IP stored), retention period, controller contact, the right to object and how to exercise it (link `/_analytics/opt-out`; DNT/GPC), and the right to complain to the CNIL. +- Add the processing to the Art. 30 record. +- Govern the other logs separately. The reverse proxy's and Uvicorn's access logs record raw IPs and full URLs. They are a separate processing that this feature neither creates nor controls. + +**Sites for children.** The CNIL has no rule specific to audience measurement on children's sites. General rules apply [3; 8; 9]: + +- The legitimate-interest balancing test weighs "in particular where the data subject is a child" (GDPR Art. 6(1)(f)) and should be written down. +- Information addressed to children must be clear and plain (Art. 12(1)). The CNIL recommends short sentences and icons. +- No profiling. This design does none. + +**Platforms that also run `enlace_auth`.** Its CSRF middleware currently sets an `enlace_csrf` cookie on every safe request that arrives without one, including public pages. A site that promises "no cookie" has to be served without that plugin, or wait for the CSRF cookie to be scoped to where it is needed. Tracked in [i2mint/enlace_auth#31](https://github.com/i2mint/enlace_auth/issues/31). + +**Pending law, not verified.** The Commission's Digital Omnibus proposal (2025/0360(COD)) would add a GDPR Art. 88a exempting aggregated first-party audience measurement from consent, and an Art. 88b making browser-level signals binding [10]. Whether it has been adopted was not confirmed as of 2026-09-25. + +## REFERENCES + +1. EDPB. [Guidelines 2/2023 on Technical Scope of Art. 5(3) of ePrivacy Directive, v2.0](https://www.edpb.europa.eu/system/files/documents/2024-10/edpb_guidelines_202302_technical_scope_art_53_eprivacydirective_v2_en_0.pdf). Adopted 7 Oct 2024; §§32–34, 42–43, 54–56. +2. CNIL. [Délibération n° 2020-091 du 17 septembre 2020 (lignes directrices cookies et autres traceurs)](https://www.cnil.fr/sites/default/files/atoms/files/lignes_directrices_de_la_cnil_sur_les_cookies_et_autres_traceurs.pdf). Art. 5, §§50–52. +3. [Regulation (EU) 2016/679 (GDPR)](https://eur-lex.europa.eu/eli/reg/2016/679/oj). Arts. 6(1)(f), 12(1), 13, 21, 30. +4. CNIL. [Recommandation « cookies et autres traceurs », version consolidée du 16 janvier 2026](https://www.cnil.fr/sites/default/files/2026-01/recommandation_cookies_consolidee.pdf). §5. +5. CNIL. [Outil d'auto-évaluation — mesure d'audience exemptée de consentement](https://www.cnil.fr/sites/default/files/2025-07/outil_d_auto-evaluation_mesure_d_audience.pdf). July 2025. +6. CNIL. [Cookies : solutions pour les outils de mesure d'audience](https://www.cnil.fr/fr/cookies-et-autres-traceurs/regles/cookies-solutions-pour-les-outils-de-mesure-daudience). 4 Jul 2025. +7. CNIL. [Questions-réponses sur les lignes directrices et la recommandation « cookies et autres traceurs »](https://www.cnil.fr/fr/cookies-et-autres-traceurs/regles/cookies/FAQ). Q10 (objection), Q18 (information). +8. CNIL. [Recommandation 6 : renforcer l'information et les droits des mineurs par le design](https://www.cnil.fr/fr/recommandation-6-renforcer-linformation-et-les-droits-des-mineurs-par-le-design). 9 Jun 2021. +9. CNIL. [Recommandation 8 : prévoir des garanties spécifiques pour protéger l'intérêt de l'enfant](https://www.cnil.fr/fr/recommandation-8-prevoir-des-garanties-specifiques-pour-proteger-linteret-de-lenfant). 9 Jun 2021. +10. Council of the EU. [Digital Omnibus, 2025/0360 (COD), working document of 19 Mar 2026](https://data.consilium.europa.eu/doc/document/WK-3736-2026-INIT/en/pdf). Secondary sources only; final text not verified. diff --git a/tests/test_analytics_browser.py b/tests/test_analytics_browser.py new file mode 100644 index 0000000..2dc4ad3 --- /dev/null +++ b/tests/test_analytics_browser.py @@ -0,0 +1,111 @@ +"""Issue #55's acceptance check, in a real (headless) browser. + +A page of an app with ``[analytics] mode = "privacy"`` is loaded in Chromium +from a live enlace server. The page view must be recorded, while the page +makes no request to any other origin and leaves nothing on the device: no +cookie, no localStorage, no sessionStorage, no IndexedDB. + +Skipped unless Playwright and its Chromium are installed +(``pip install playwright && playwright install chromium``). CI does not +install them; run it locally when touching ``enlace.analytics``. +""" + +import socket +import threading +import time + +import pytest + +pytest.importorskip("playwright.sync_api") + +import uvicorn # noqa: E402 +from playwright.sync_api import Error as PlaywrightError # noqa: E402 +from playwright.sync_api import sync_playwright # noqa: E402 + +from enlace import analytics as an # noqa: E402 +from enlace.base import PlatformConfig # noqa: E402 +from enlace.compose import build_backend # noqa: E402 +from enlace.discover import discover_apps # noqa: E402 + +# Not "HeadlessChrome": that user agent is (rightly) classified as a bot. +_USER_AGENT = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_0) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/130.0 Safari/537.36" +) + + +def _free_port() -> int: + with socket.socket() as s: + s.bind(("127.0.0.1", 0)) + return s.getsockname()[1] + + +def _serve(app, port): + server = uvicorn.Server( + uvicorn.Config(app, host="127.0.0.1", port=port, log_level="warning") + ) + thread = threading.Thread(target=server.run, daemon=True) + thread.start() + deadline = time.time() + 10 + while not server.started and time.time() < deadline: + time.sleep(0.05) + return server, thread + + +def test_page_view_recorded_with_no_third_party_request_and_nothing_stored(tmp_path): + app_dir = tmp_path / "apps" / "kids" + (app_dir / "frontend").mkdir(parents=True) + (app_dir / "frontend" / "index.html").write_text( + 'Kids' + '' + '

Bonjour

' + ) + (app_dir / "frontend" / "style.css").write_text("h1{color:teal}") + (app_dir / "frontend" / "app.js").write_text("document.title += ' ok';") + (app_dir / "app.toml").write_text('[analytics]\nmode = "privacy"\n') + + config = discover_apps( + PlatformConfig( + apps_dirs=[tmp_path / "apps"], analytics={"flush_interval_seconds": 0} + ) + ) + store = an.JsonFileStore(tmp_path / "analytics") + port = _free_port() + server, thread = _serve(build_backend(config, analytics_store=store), port) + origin = f"http://127.0.0.1:{port}" + try: + with sync_playwright() as p: + try: + browser = p.chromium.launch() + except PlaywrightError as e: # browser binary not installed + pytest.skip(f"Chromium unavailable: {e}") + context = browser.new_context(user_agent=_USER_AGENT, locale="fr-FR") + page = context.new_page() + requested: list[str] = [] + page.on("request", lambda r: requested.append(r.url)) + page.goto(f"{origin}/kids/", wait_until="networkidle") + assert page.title() == "Kids ok" # the page (and its script) ran + + storage = page.evaluate( + """async () => ({ + local: localStorage.length, + session: sessionStorage.length, + idb: indexedDB.databases + ? (await indexedDB.databases()).length : 0, + })""" + ) + cookies = context.cookies() + browser.close() + finally: + server.should_exit = True + thread.join(timeout=10) + + assert requested, "the browser made no request at all?" + assert all(url.startswith(origin + "/") for url in requested), requested + assert cookies == [] + assert storage == {"local": 0, "session": 0, "idb": 0} + + totals = an.summarize(an.daily_counts(store, "kids", days=1)) + assert totals["pageviews"] == 1 # the page, not its CSS or JS + assert totals["paths"] == {"/": 1} + assert totals["languages"] == {"fr": 1}