From b2816b5fe75e62c892d6bb8c242fd061badc2512 Mon Sep 17 00:00:00 2001 From: Ntege Daniel Date: Wed, 30 Sep 2026 13:57:44 +0300 Subject: [PATCH 1/3] feat: add repository and contributor detail views Open a detail view for any repository or contributor from its name in the dashboard's tables and charts, backed by period-accurate activity computed from the persisted events and published through the data API. * Add the entity_activity pipeline: tracked actions per repository, contributor and (repository, contributor) pair for Week, 1 month, 1 year and all time, each counted from its own events, plus monthly trends * Publish per-org entity indexes and lazily fetched detail documents (additive to API v1) with source, freshness, population, methodology and window dates; empty periods are zeros * Join releases, governance, HIP engagement, onboarding and security into a repository's view, naming tables a run did not produce and linking each figure back to its dashboard section * Link repository and contributor names in tables and charts to their view, keeping a separate GitHub link; keep navigation in the URL so shared links, Back and Forward work * Hold a section jump on its target while charts above it load * Cover calculations, documents, contract, navigation and rendering with tests; document the data model in docs/entity-views.md Fixes #503 Fixes #357 Signed-off-by: Ntege Daniel --- README.md | 1 + docs/architecture.md | 5 +- docs/entity-views.md | 184 +++++ docs/snapshots.md | 6 +- .../analysis/entity_activity.py | 198 +++++ .../dashboard_spec/entities.py | 191 +++++ src/hiero_analytics/export/chart_data.py | 43 +- src/hiero_analytics/export/data_api.py | 89 ++- src/hiero_analytics/export/entity_views.py | 484 ++++++++++++ src/hiero_analytics/pipelines/__init__.py | 7 + .../pipelines/entity_activity.py | 74 ++ tests/analysis/test_entity_activity.py | 172 ++++ tests/contracts/test_output_contract.py | 43 +- tests/export/test_chart_data.py | 16 + tests/export/test_entity_views.py | 327 ++++++++ tests/pipelines/test_entity_activity.py | 75 ++ web/README.md | 9 + web/e2e/entities.spec.ts | 84 ++ web/src/App.tsx | 90 ++- web/src/api.ts | 127 +++ web/src/components/ContributorCell.tsx | 26 +- web/src/components/DataTable.tsx | 28 +- web/src/components/EntityLink.tsx | 57 ++ web/src/components/EntityView.tsx | 743 ++++++++++++++++++ web/src/components/RepoCell.tsx | 26 +- web/src/components/SectionTable.tsx | 12 +- web/src/components/charts/EntityTick.tsx | 50 ++ web/src/components/charts/EventsView.tsx | 35 +- web/src/components/charts/MatrixView.tsx | 9 +- web/src/components/charts/NetworkView.tsx | 11 +- web/src/components/charts/SeriesView.tsx | 101 ++- web/src/entities.ts | 226 ++++++ web/src/printing.tsx | 11 +- web/src/test/entities.test.tsx | 388 +++++++++ web/src/test/entityFixtures.ts | 292 +++++++ 35 files changed, 4168 insertions(+), 72 deletions(-) create mode 100644 docs/entity-views.md create mode 100644 src/hiero_analytics/analysis/entity_activity.py create mode 100644 src/hiero_analytics/dashboard_spec/entities.py create mode 100644 src/hiero_analytics/export/entity_views.py create mode 100644 src/hiero_analytics/pipelines/entity_activity.py create mode 100644 tests/analysis/test_entity_activity.py create mode 100644 tests/export/test_entity_views.py create mode 100644 tests/pipelines/test_entity_activity.py create mode 100644 web/e2e/entities.spec.ts create mode 100644 web/src/components/EntityLink.tsx create mode 100644 web/src/components/EntityView.tsx create mode 100644 web/src/components/charts/EntityTick.tsx create mode 100644 web/src/entities.ts create mode 100644 web/src/test/entities.test.tsx create mode 100644 web/src/test/entityFixtures.ts diff --git a/README.md b/README.md index d89da3979..bed6b73fd 100644 --- a/README.md +++ b/README.md @@ -144,6 +144,7 @@ Available pipelines: | `contributor_profiles` | Per-contributor profiles | | `maintainer_pipeline` | Maintainer pipeline by governance role | | `contributor_activity` | Org-wide contributor activity tables | +| `entity_activity` | Per-repository and per-contributor activity (by period and month) behind the dashboard's detail views | | `contributor_heatmap` | Contributor activity heatmaps | | `role_coverage` | Governance roles vs. real activity per repo | | `affiliation` | Contributor affiliation mapping | diff --git a/docs/architecture.md b/docs/architecture.md index b0c56c15a..353abee9d 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -99,7 +99,10 @@ contract for everything downstream of the pipelines (the web dashboard, notebooks, external tools), and it enforces the producer↔spec agreement: a produced table missing a spec-declared column fails the emit, so a renamed output is a red build rather than a silently blank dashboard column. Breaking -shape changes bump the version directory; `v1` is additive-only. +shape changes bump the version directory; `v1` is additive-only. Each org also +publishes repository and contributor detail documents (an index of each, plus one +lazily fetched document per entity) from the `entity_activity` pipeline's tables; +see [entity-views.md](entity-views.md). **The web dashboard.** `web/` is a static Vite + React app deployed at the Pages site root with `data/api/` and `charts/` nested beneath it. It is manifest-driven: diff --git a/docs/entity-views.md b/docs/entity-views.md new file mode 100644 index 000000000..d3c59e083 --- /dev/null +++ b/docs/entity-views.md @@ -0,0 +1,184 @@ +# Repository and contributor detail views + +Every repository and every person with tracked activity in an organisation has +a detail view on the dashboard. It opens from their name in any table or chart. + +## What is counted + +Tracked activity is five kinds of GitHub event: + +- pull requests opened; +- reviews submitted; +- pull requests merged, credited to the person who merged; +- issues opened; +- labels applied (label removals are not counted). + +It does not measure commits, comments, reactions or anything else, and it is not +a measure of individual performance. Every view says so above its numbers. + +The counts use the same events and definitions as the Contributors tab's +profiles (`analysis/contributor_activity_profile.combined_activity_events`). +Automation accounts are excluded by `domain/bots.is_bot_login`, the same filter +the rest of the dashboard uses. + +## Data model + +``` +persisted datasets (contributor_activity__all, issue_label_events__all) + → analysis/entity_activity.py pure, per-window counts from the events + → pipelines/entity_activity.py writes 5 org-level CSVs (+ .meta.json) + → export/entity_views.py (data_api) indexes + one detail document per entity +``` + +**Windows.** Week, 1 month and 1 year are the 7, 30 and 365 days before the +pipeline ran, plus all time. Each window is computed from the events inside it, +not by summing or filtering the all-time profiles. So a repository's _active +contributors_ in a week is the number of distinct people with an action there +that week. The per-repository contributor profiles the dashboard had before were +all-time only; these tables add the windowed versions. + +**Tables** (`dashboard_spec/entities.py` names them; each is long form, with one +row per entity and window that had activity): + +| File | One row per | Columns | +| ----------------------------------------------------------- | ---------------------------------- | --------------------------------------------------------------------------------------------- | +| `entity_repo_activity.csv` | repository × window | the 5 counts, `total_actions`, `active_contributors`, first/last active | +| `entity_contributor_activity.csv` | contributor × window | the 5 counts, `total_actions`, `repos_touched`, work-mix counts and shares, first/last active | +| `entity_repo_contributor_activity.csv` | (repository, contributor) × window | the 5 counts, `total_actions`, work-mix shares, first/last active | +| `entity_repo_monthly.csv`, `entity_contributor_monthly.csv` | entity × month (UTC) | the 5 counts | + +The three activity tables also carry: + +- `window_end`, the moment every window ends; +- `data_through`, the latest tracked event in the organisation. A view warns + when there is no event after the Week window's start, because a quiet week + then can't be told apart from activity data that was not refreshed. + +**Joined tables.** A repository's view also joins the optional per-repository +tables other pipelines write, declared in `dashboard_spec.entities.REPO_RELATED`: + +| Section | Tables | +| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Releases | `release_repo_summary.csv`, `release_timeline.csv` | +| Governance | `repo_activity_overview.csv` (role counts, active permission-holders), `repo_affiliation_diversity.csv`, `review_load_share.csv`, `repo_wise_codeowner_status.csv`, `role_coverage_all.csv` | +| HIP engagement | `hip_repo_engagement.csv` | +| Onboarding | `difficulty_by_repo.csv` (including issues without a difficulty label) | +| Security | `org_scorecard_checks.csv` | + +**Links back to sources.** Each section links to the dashboard sections its +tables feed (`links: [{macro, id, title}]`). `data_api` builds the map from what +it actually published for the org, so a link never leads to a section the org +lacks. Following one closes the detail view and jumps to that section, the same +way a shared section link does. While charts above the target load, the jump +holds the target in place for a few seconds, and stops as soon as the reader +scrolls. Shared section links benefit too. + +Repositories match by bare or `owner/repo` name. A table this run did not +produce is listed as unavailable, for example when an offline run skips the +network-only pipelines. A table that was produced but has no row for the +repository shows as an empty section, which the view states. + +## Data API (v1, additive) + +Each org's manifest entry gains `entities`: + +```json +"entities": { + "repositories": {"path": "/entities/repositories.json", "count": 40}, + "contributors": {"path": "/entities/contributors.json", "count": 1457} +} +``` + +**Indexes.** Each index (`kind: repositories-index` / `contributors-index`, +`schema_version: 1`) lists every entity with its `id`, its names, all-time +`total_actions` and `last_active`. Its `detail_path` template locates each +entity's document, with `{id}` substituted. + +**Ids.** Ids are lower-case and safe to use as file names. A repository name that +starts with a dot gets an underscore instead (`.github` → `_github`), so no +document is a hidden file. Clients find ids through the index, not by deriving +them from names. + +**Detail documents.** Each detail document (`kind: repository` / `contributor`, +`schema_version: 1`) is fetched only when a reader opens it. It carries: + +- **Provenance:** `source` (the CSVs it was built from), `generated_at` and + `stale` (from the CSV sidecar, with the same 132-hour rule as sections); +- **Explanation:** `scope`, `population`, `methodology` and `limits`; +- **Dates:** `window` (end, `data_through`, and each window's start), + `first_active` and `last_active`; +- **Counts:** `summary` and `mix`, keyed by `all`, `365d`, `30d` and `7d`. A + window with no activity is all zeros, never missing; +- **Trend:** a standard monthly timeseries chart document (the same format + `chart_data` publishes); +- **Breakdown table:** `contributors` (repository) or `repositories` + (contributor), shaped like a section with all-time `rows` and `periods`; +- **Repositories only:** `related` and `unavailable`. + +The `entities/` tree is rewritten on every emit, so an entity that leaves the +data does not leave a stale document behind. + +**Size.** hiero-ledger publishes about 1,500 detail documents (about 18 MB; +median 9 KB per contributor), which brings the API to about 24 MB across +about 1,660 files, or 1.3 MB gzipped. See `docs/snapshots.md`. + +## Navigation + +- **Opening a view.** `entity=repo:` or `entity=contributor:` in the hash + opens a view over the current tab. The tab, org and focus keys are kept, so + "Back to " and browser Back return exactly where the reader was. Links are + real hash hrefs: a click adds a history entry, and Forward, "open in new tab" + and copied links all reopen the view. +- **Which names link.** Every well-formed repository name and GitHub login links + once the org's indexes have loaded (they are fetched once per org, after the + manifest), so names behave the same everywhere. A name with no tracked activity + in the org opens a short view saying so, with a "View on GitHub" button. An org + that publishes no indexes keeps plain GitHub links. +- **Where names link.** Table cells (repositories and people), heatmap rows, + ranking and release-timeline axis labels, chart Data views, and the network's + focus panel all link. Each keeps a separate GitHub link: an icon beside the + name, or the view's "View on GitHub". +- **Focus.** In a chart's Data view, the dashboard-focus toggle moves to a + crosshair beside the name and keeps its label ("Focus on …"). Names without a + detail view keep their old behaviour. A detail view's "Focus the dashboard on …" + sets the focus and returns to the tab in one history entry. +- **Switching.** Choosing a tab closes an open view. Switching organisation + clears both the focus and the open view, since both belong to one org. +- **Reuse.** A view is built from the dashboard's own pieces: section cards and + groups (with the sidebar's "On this page"), the period-tabbed table with its + filter and CSV export, the interactive chart with its Data view and CSV, the + loading, error and retry states, and "Print page", which uses the tab print + path. + +## Limits of the source data + +- **Capped reads.** At most 100 reviews per pull request and 100 label events + per issue are read. +- **Merge credit.** A merge counts for the person who merged, who may not be + the author. +- **Unattributable events.** Events whose author GitHub no longer reports, and + label events without an actor, are dropped. +- **Window timing.** Windows end when the pipeline runs, not when the data was + fetched. With stale datasets the recent windows undercount, and the view warns + about it (see `data_through` above). +- **Bot filtering.** Automation accounts that `is_bot_login` does not recognise + are counted as people. `stepsecurity-app` in hiero-ledger is one example. +- **Other orgs.** Only the organisation's own repositories are counted. A person's + activity in other organisations is not. + +## Tests + +- **Calculations:** `tests/analysis/test_entity_activity.py` covers window + boundaries, distinct active contributors, merge credit, work mix, bots and + label removals. +- **Pipeline:** `tests/pipelines/test_entity_activity.py`. +- **Documents:** `tests/export/test_entity_views.py` covers ids, indexes, + zero-filled windows, the trend document, joined and missing tables, freshness, + the manifest entry and stale-document cleanup. +- **Contract:** `tests/contracts/test_output_contract.py` checks, in the full + synthetic run, that every org's index resolves to documents and that every + joined table is actually produced. +- **Frontend:** `web/src/test/entities.test.tsx` covers the URL, table and chart + links, Back/Forward, the rendering, focus, org and tab switches, and the + missing and failed cases. `web/e2e/entities.spec.ts` covers real browser + history and printing, in Chromium and Firefox. diff --git a/docs/snapshots.md b/docs/snapshots.md index 0003ddf8c..5502949d0 100644 --- a/docs/snapshots.md +++ b/docs/snapshots.md @@ -66,7 +66,11 @@ version before parsing anything older than the current release. - **Deletions propagate.** The tree is cleared before each snapshot is written, so a section removed from the spec disappears from the latest snapshot rather than lingering as an orphan. Earlier commits keep it, which is the point. -- **No retention policy yet.** One snapshot is ~3.5 MB of JSON across ~20 files, +- **No retention policy yet.** The repository and contributor detail documents + ([entity-views.md](entity-views.md)) grew a snapshot to ~24 MB across ~1,660 + files (1.3 MB gzipped), and most of them change on every refresh because they + carry their generation time, so re-measure the packed history after a few + snapshots. Before them, one snapshot was ~3.5 MB of JSON across ~20 files, but consecutive snapshots are near-identical and git packs the deltas: a rehearsal of two snapshots (3.2 MB each on disk) packed to 632 KB of history in total. At the 5-day cadence that is a few MB a year, so there is nothing to diff --git a/src/hiero_analytics/analysis/entity_activity.py b/src/hiero_analytics/analysis/entity_activity.py new file mode 100644 index 000000000..f1a5dbbfc --- /dev/null +++ b/src/hiero_analytics/analysis/entity_activity.py @@ -0,0 +1,198 @@ +"""Per-repository and per-contributor activity for the dashboard's detail views. + +Pure transforms over the combined activity event frame +(:func:`~hiero_analytics.analysis.contributor_activity_profile.combined_activity_events`): +one row per tracked action — a pull request authored, a review submitted, a pull +request merged (credited to whoever merged it), an issue opened, or a label +applied. Every count here is a count of those events, with the same definitions +the contributor profiles use, so the detail views agree with the Contributors tab. + +Each table is computed once per window from the events themselves (never by +summing per-contributor or org-wide summary rows), so a repository's +*active contributors* in the last week is the number of distinct people with a +tracked action in that repository in that week, not a sum of overlapping counts. + +Windows are ``(key, cutoff)`` pairs: ``cutoff`` is the inclusive lower bound, or +``None`` for all recorded time. The caller decides the windows and the moment +they end; nothing here reads the clock. +""" + +from __future__ import annotations + +from collections.abc import Sequence +from datetime import datetime + +import pandas as pd + +from hiero_analytics.analysis.contributor_activity_profile import CONTRIB_COUNT_FIELDS + +# Which tracked activity type each count field counts (the profile definitions). +COUNT_OF_ACTIVITY: dict[str, str] = { + "authored_pull_request": "prs_opened", + "reviewed_pull_request": "reviews_given", + "merged_pull_request": "merges_done", + "authored_issue": "issues_opened", + "labeled_issue": "labels_applied", +} + +# The three neutral work families, as sums of the count fields. +FAMILY_FIELDS: dict[str, tuple[str, ...]] = { + "building_and_fixing": ("prs_opened",), + "reviewing_and_guiding": ("reviews_given", "merges_done"), + "organizing_and_answering": ("issues_opened", "labels_applied"), +} +SHARE_OF_FAMILY = { + "building_and_fixing": "building_share", + "reviewing_and_guiding": "reviewing_share", + "organizing_and_answering": "organizing_share", +} + +Window = tuple[str, datetime | None] + +REPO_ACTIVITY_COLUMNS = [ + "repo", + "period", + *CONTRIB_COUNT_FIELDS, + "total_actions", + "active_contributors", + "first_active", + "last_active", +] +CONTRIBUTOR_ACTIVITY_COLUMNS = [ + "contributor", + "period", + *CONTRIB_COUNT_FIELDS, + "total_actions", + "repos_touched", + *FAMILY_FIELDS, + *SHARE_OF_FAMILY.values(), + "first_active", + "last_active", +] +PAIR_ACTIVITY_COLUMNS = [ + "repo", + "contributor", + "period", + *CONTRIB_COUNT_FIELDS, + "total_actions", + *SHARE_OF_FAMILY.values(), + "first_active", + "last_active", +] +MONTHLY_COLUMNS = ["month", *CONTRIB_COUNT_FIELDS] + + +def _counted(events: pd.DataFrame) -> pd.DataFrame: + """The events that count toward a field, tagged with that field. + + Activity types outside the five tracked ones (none today) are dropped rather + than silently folded into a total. + """ + frame = events.assign(field=events["activity_type"].map(COUNT_OF_ACTIVITY)) + return frame[frame["field"].notna() & frame["occurred_at"].notna()] + + +def _in_window(events: pd.DataFrame, cutoff: datetime | None) -> pd.DataFrame: + return events if cutoff is None else events[events["occurred_at"] >= cutoff] + + +def _field_counts(events: pd.DataFrame, keys: list[str]) -> pd.DataFrame: + """One row per key combination with an integer column per count field.""" + counts = events.groupby([*keys, "field"]).size().unstack("field", fill_value=0) + counts = counts.reindex(columns=list(CONTRIB_COUNT_FIELDS), fill_value=0).astype(int) + counts["total_actions"] = counts[list(CONTRIB_COUNT_FIELDS)].sum(axis=1) + return counts + + +def _span(events: pd.DataFrame, keys: list[str]) -> pd.DataFrame: + grouped = events.groupby(keys)["occurred_at"] + return pd.DataFrame({"first_active": grouped.min(), "last_active": grouped.max()}) + + +def _share(part: pd.Series, whole: pd.Series) -> pd.Series: + """Integer percentages, 0 where the whole is 0 (matches the profile tables).""" + return (part / whole.where(whole > 0) * 100).round().fillna(0).astype(int) + + +def _add_families(frame: pd.DataFrame, *, keep_counts: bool) -> pd.DataFrame: + for family, fields in FAMILY_FIELDS.items(): + total = frame[list(fields)].sum(axis=1) + if keep_counts: + frame[family] = total + frame[SHARE_OF_FAMILY[family]] = _share(total, frame["total_actions"]) + return frame + + +def _per_window( + events: pd.DataFrame, + windows: Sequence[Window], + columns: list[str], + build, +) -> pd.DataFrame: + """Concatenate ``build(window_events)`` for each window, tagged with its key.""" + counted = _counted(events) + frames = [] + for key, cutoff in windows: + scoped = _in_window(counted, cutoff) + if scoped.empty: + continue + frame = build(scoped) + frame.insert(0, "period", key) + frames.append(frame.reset_index()) + if not frames: + return pd.DataFrame(columns=columns) + return pd.concat(frames, ignore_index=True)[columns] + + +def repo_activity(events: pd.DataFrame, windows: Sequence[Window]) -> pd.DataFrame: + """Each repository's tracked actions and distinct active contributors per window. + + A repository with no tracked action in a window has no row for it; the + detail view reads a missing row as zero activity. + """ + + def build(scoped: pd.DataFrame) -> pd.DataFrame: + frame = _field_counts(scoped, ["repo"]) + frame["active_contributors"] = scoped.groupby("repo")["contributor"].nunique() + return frame.join(_span(scoped, ["repo"])) + + return _per_window(events, windows, REPO_ACTIVITY_COLUMNS, build) + + +def contributor_activity(events: pd.DataFrame, windows: Sequence[Window]) -> pd.DataFrame: + """Each contributor's tracked actions, work mix and repositories touched per window.""" + + def build(scoped: pd.DataFrame) -> pd.DataFrame: + frame = _field_counts(scoped, ["contributor"]) + frame["repos_touched"] = scoped.groupby("contributor")["repo"].nunique() + frame = _add_families(frame, keep_counts=True) + return frame.join(_span(scoped, ["contributor"])) + + return _per_window(events, windows, CONTRIBUTOR_ACTIVITY_COLUMNS, build) + + +def repo_contributor_activity(events: pd.DataFrame, windows: Sequence[Window]) -> pd.DataFrame: + """Each (repository, contributor) pair's tracked actions and work mix per window. + + The one table behind both a repository's contributor list and a + contributor's per-repository breakdown, so the two can never disagree. + """ + + def build(scoped: pd.DataFrame) -> pd.DataFrame: + frame = _add_families(_field_counts(scoped, ["repo", "contributor"]), keep_counts=False) + return frame.join(_span(scoped, ["repo", "contributor"])) + + return _per_window(events, windows, PAIR_ACTIVITY_COLUMNS, build) + + +def monthly_activity(events: pd.DataFrame, key: str) -> pd.DataFrame: + """Tracked actions per calendar month (UTC) per ``key`` (``repo`` or ``contributor``). + + Months without activity are absent; the chart export completes the calendar. + """ + counted = _counted(events) + columns = [key, *MONTHLY_COLUMNS] + if counted.empty: + return pd.DataFrame(columns=columns) + frame = _field_counts(counted, [key, "month"]).drop(columns="total_actions").reset_index() + return frame.sort_values([key, "month"], ignore_index=True)[columns] diff --git a/src/hiero_analytics/dashboard_spec/entities.py b/src/hiero_analytics/dashboard_spec/entities.py new file mode 100644 index 000000000..58283f629 --- /dev/null +++ b/src/hiero_analytics/dashboard_spec/entities.py @@ -0,0 +1,191 @@ +"""Repository and contributor detail views: what they read, publish and say. + +Pure data, like the family modules, but not a tab: the detail views open from a +repository or contributor name anywhere on the dashboard. The ``entity_activity`` +pipeline writes the ``*_FILE`` tables; ``export/entity_views`` joins them with +the optional per-repository tables below and publishes, per org: + +- ``/entities/repositories.json`` and ``/entities/contributors.json``, + the indexes the dashboard loads to know which names have a detail view, and +- ``/entities/repositories/.json`` and + ``/entities/contributors/.json``, one detail document each, fetched + only when a reader opens one. +""" + +from __future__ import annotations + +from hiero_analytics.config.analysis import ROLE_ACTIVE_DAYS + +# Tables the entity_activity pipeline writes into each org's data directory. +REPO_ACTIVITY_FILE = "entity_repo_activity.csv" +CONTRIBUTOR_ACTIVITY_FILE = "entity_contributor_activity.csv" +PAIR_ACTIVITY_FILE = "entity_repo_contributor_activity.csv" +REPO_MONTHLY_FILE = "entity_repo_monthly.csv" +CONTRIBUTOR_MONTHLY_FILE = "entity_contributor_monthly.csv" +ENTITY_FILES = ( + REPO_ACTIVITY_FILE, + CONTRIBUTOR_ACTIVITY_FILE, + PAIR_ACTIVITY_FILE, + REPO_MONTHLY_FILE, + CONTRIBUTOR_MONTHLY_FILE, +) + +# Stated on every detail view: what the counts are, and what they are not. +SCOPE = ( + "Tracked activity covers pull requests opened, reviews submitted, pull requests " + "merged, issues opened and labels applied. It does not measure commits, comments, " + "reactions or anything else, and it is not a measure of individual performance." +) + +COUNT_LABELS = { + "prs_opened": "PRs opened", + "reviews_given": "Reviews", + "merges_done": "Merges", + "issues_opened": "Issues opened", + "labels_applied": "Labels applied", +} + +FAMILY_LABELS = { + "building_and_fixing": "Building & fixing", + "reviewing_and_guiding": "Reviewing & guiding", + "organizing_and_answering": "Organizing & answering", +} + +REPOSITORY_POPULATION = ( + "Every tracked action in this repository by a person (automation accounts are " + "excluded). A person active in several repositories is counted in each." +) +CONTRIBUTOR_POPULATION = ( + "Every tracked action by this person across the organisation's repositories. " + "Merges are credited to the person who merged, not the pull request's author." +) + +METHODOLOGY = [ + "Read the persisted GitHub activity for the organisation: pull requests with their " + "reviews and merges, issues, and issue label events.", + "Count one action per event: a pull request opened (at creation), a review " + "submitted, a pull request merged (by the merger), an issue opened, and a label " + "applied (label removals are not counted).", + "Drop automation accounts and events whose author GitHub no longer reports.", + "Count each window from the events inside it: Week, 1 month and 1 year are the " + "7, 30 and 365 days before the analysis ran; All time is everything recorded.", + "Active contributors are distinct people with at least one action in the window, " + "counted from the events rather than summed from other tables.", + "Work mix splits the same actions into building & fixing (PRs opened), reviewing & " + "guiding (reviews and merges) and organizing & answering (issues and labels).", +] + +LIMITS = [ + "At most 100 reviews per pull request and 100 label events per issue are read, so " + "unusually long threads can undercount.", + "A pull request's merge counts for the person who merged it, which may not be its author.", +] + +# Optional per-repository tables a repository's detail view joins, by section. +# Each names its org-level CSV (the repository is matched in its ``repo`` +# column, bare or owner/repo) and the columns published. A table that was not +# produced — an offline run skips the network-only pipelines — is listed as +# unavailable on the view. Each section links to the dashboard sections its +# tables feed, so every figure leads back to its evidence. +REPO_RELATED = { + "releases": { + "title": "Releases", + "file": "release_repo_summary.csv", + "columns": [ + ("latest_release", "Latest release", "date"), + ("days_since_last_release", "Days since last release", "number"), + ("median_gap_days", "Median days between releases", "number"), + ("release_status", "Release cadence", "status"), + ], + "list": { + "file": "release_timeline.csv", + "title": "Recent releases", + "sort": "published_at", + "limit": 10, + "columns": [ + ("tag_name", "Tag"), + ("published_at", "Published (UTC)", "date"), + ("is_prerelease", "Pre-release", "flag"), + ], + }, + }, + "governance": { + "title": "Governance", + "file": "repo_activity_overview.csv", + "columns": [ + ("maintainers", "Maintainers", "number"), + ("committers", "Committers", "number"), + ("triage", "Triage", "number"), + ("active_recent", f"Permission-holders active in the last {ROLE_ACTIVE_DAYS} days", "number"), + ], + "extra": [ + { + "file": "repo_affiliation_diversity.csv", + "columns": [ + ("distinct_orgs", "Maintainer organisations", "number"), + ("top_org", "Largest organisation"), + ("top_org_pct", "Largest organisation's share", "percent"), + ], + }, + { + "file": "review_load_share.csv", + "columns": [ + ("mergers", "People merging recently", "number"), + ("top_carrier", "Largest review load"), + ("top_pct", "Their share of recent merges", "percent"), + ], + }, + { + "file": "repo_wise_codeowner_status.csv", + "columns": [("status", "CODEOWNERS file", "presence")], + }, + ], + "list": { + "file": "role_coverage_all.csv", + "title": "Role holders", + "sort": "granted_role", + "columns": [ + ("user", "Person"), + ("granted_role", "Role"), + ("status", "Status", "status"), + ("last_active", "Last active (UTC)", "date"), + ], + }, + }, + "hips": { + "title": "HIP engagement", + "file": "hip_repo_engagement.csv", + "columns": [ + ("distinct_hips_merged", "Distinct HIPs with merged PRs", "number"), + ("matched_prs", "PRs referencing a HIP", "number"), + ("total_prs", "PRs checked", "number"), + ], + }, + "onboarding": { + "title": "Onboarding", + "file": "difficulty_by_repo.csv", + "columns": [ + ("Good First Issue", "Open good first issues", "number"), + ("Beginner", "Open beginner issues", "number"), + ("Intermediate", "Open intermediate issues", "number"), + ("Advanced", "Open advanced issues", "number"), + ("Unknown", "Open issues without a difficulty label", "number"), + ], + }, + "security": { + "title": "Security", + "file": "org_scorecard_checks.csv", + "columns": [ + ("score", "OpenSSF Scorecard", "number"), + ("date", "Scored on", "date"), + ("Maintained", "Maintained", "number"), + ("Code-Review", "Code review", "number"), + ("Branch-Protection", "Branch protection", "number"), + ("Token-Permissions", "Token permissions", "number"), + ("Pinned-Dependencies", "Pinned dependencies", "number"), + ("Dangerous-Workflow", "Dangerous workflow", "number"), + ("Security-Policy", "Security policy", "number"), + ("Signed-Releases", "Signed releases", "number"), + ], + }, +} diff --git a/src/hiero_analytics/export/chart_data.py b/src/hiero_analytics/export/chart_data.py index 3f040bbf2..d6f653f08 100644 --- a/src/hiero_analytics/export/chart_data.py +++ b/src/hiero_analytics/export/chart_data.py @@ -89,11 +89,16 @@ def _series(source: dict, frame: pd.DataFrame, category: str, group: str | None) return series -def _read(source: dict, csv_path: Path, org: str) -> tuple[pd.DataFrame, list[dict], list[dict]]: - """Read and validate the CSV columns a source declares.""" +def _read( + source: dict, csv_path: Path, org: str, table: pd.DataFrame | None = None +) -> tuple[pd.DataFrame, list[dict], list[dict]]: + """Read and validate the CSV columns a source declares (or an already-built ``table``).""" category = source["category"] group = source.get("group", {}).get("key") - frame = pd.read_csv(csv_path, dtype={category: str, **({group: str} if group else {})}) + if table is None: + frame = pd.read_csv(csv_path, dtype={category: str, **({group: str} if group else {})}) + else: + frame = table.astype({category: str, **({group: str} if group else {})}) series = _series(source, frame, category, group) details = [{"format": "number", **detail} for detail in source.get("details", [])] required = [category, *([group] if group else []), *(s["key"] for s in series), *(d["key"] for d in details)] @@ -160,9 +165,11 @@ def _bucket_labels(first: datetime, last: datetime, frequency: str) -> list[str] return [stamp.strftime(fmt) for stamp in stamps] or [first.strftime(fmt)] -def _timeseries(source: dict, csv_path: Path, org: str, generated_at: str | None) -> dict: +def _timeseries( + source: dict, csv_path: Path, org: str, generated_at: str | None, table: pd.DataFrame | None = None +) -> dict: """Rows in time order, with calendar gaps completed and partial buckets flagged.""" - frame, series, details = _read(source, csv_path, org) + frame, series, details = _read(source, csv_path, org, table) category, frequency = source["category"], source["frequency"] if source.get("normalize_dates"): frame[category] = pd.to_datetime(frame[category], utc=True).dt.strftime(FORMATS[frequency].removesuffix("-%u")) @@ -193,9 +200,11 @@ def _timeseries(source: dict, csv_path: Path, org: str, generated_at: str | None } -def _categories(source: dict, csv_path: Path, org: str, generated_at: str | None) -> dict: +def _categories( + source: dict, csv_path: Path, org: str, generated_at: str | None, table: pd.DataFrame | None = None +) -> dict: """Category rows in the source's order (the analysis's ranking or sequence).""" - frame, series, details = _read(source, csv_path, org) + frame, series, details = _read(source, csv_path, org, table) category = source["category"] group = source.get("group") if group: @@ -428,11 +437,25 @@ def _comparison(rows: list[dict], frequency: str) -> dict | None: OTHER_KINDS = {"matrix": _matrix, "network": _network, "events": _events} -def chart_document(source: dict, csv_path: Path, org: str, generated_at: str | None = None) -> dict: - """Build the interactive document a dashboard spec source declares.""" +def chart_document( + source: dict, + csv_path: Path, + org: str, + generated_at: str | None = None, + *, + table: pd.DataFrame | None = None, +) -> dict: + """Build the interactive document a dashboard spec source declares. + + ``table`` supplies a timeseries or categories dataset already in memory (one + slice of a larger table, such as one repository's monthly activity); the + document still names ``csv_path`` as its source. + """ kind = source.get("kind") if kind not in SERIES_KINDS and kind not in OTHER_KINDS: raise ValueError(f"Unknown interactive chart kind: {kind}") + if table is not None and kind not in SERIES_KINDS: + raise ValueError(f"An in-memory table can only build a timeseries or categories chart, not {kind}") document = { "schema_version": 1, "id": csv_path.stem, @@ -458,7 +481,7 @@ def chart_document(source: dict, csv_path: Path, org: str, generated_at: str | N mark = source.get("mark", "bar") if mark not in MARKS and not (kind == "categories" and mark in CATEGORY_MARKS): raise ValueError(f"Unknown chart mark for {kind}: {mark}") - body = SERIES_KINDS[kind](source, csv_path, org, generated_at) + body = SERIES_KINDS[kind](source, csv_path, org, generated_at, table) if mark in CATEGORY_MARKS and len(body["series"]) != 1: raise ValueError(f"{csv_path.name}: a {mark} draws exactly one series") document = { diff --git a/src/hiero_analytics/export/data_api.py b/src/hiero_analytics/export/data_api.py index 29c657a24..077e8cd01 100644 --- a/src/hiero_analytics/export/data_api.py +++ b/src/hiero_analytics/export/data_api.py @@ -38,6 +38,7 @@ import importlib import json import logging +import shutil from datetime import UTC, datetime, timedelta from pathlib import Path @@ -66,6 +67,7 @@ from hiero_analytics.domain.periods import ACTIVITY_PERIODS from hiero_analytics.export.chart_data import chart_document from hiero_analytics.export.csv_safety import sanitize_csv_text +from hiero_analytics.export.entity_views import build_entity_documents from hiero_analytics.export.macro_metrics import macro_metrics from hiero_analytics.provenance import resolve_provenance @@ -108,26 +110,30 @@ def _read_meta(csv_path: Path) -> dict: return payload if isinstance(payload, dict) else {} -def _stamp_freshness(document: dict, csv_path: Path) -> None: - """Attach ``generated_at``/``stale`` from the source CSV's sidecar, if any.""" +def _freshness(csv_path: Path) -> dict: + """``generated_at``/``stale`` from the source CSV's sidecar; {} when it has none.""" generated_at = _read_meta(csv_path).get("generated_at") if not generated_at: - return - document["generated_at"] = generated_at + return {} try: generated = datetime.fromisoformat(generated_at) except ValueError: # Ship the raw stamp without a staleness verdict, but say so — a sidecar # that stops parsing should show up in the run log, not vanish. logger.warning("Unparseable generated_at %r in sidecar for %s", generated_at, csv_path) - return + return {"generated_at": generated_at} # Sidecars are written UTC-aware, but a hand-edited or legacy one may be # naive; assume UTC rather than letting the subtraction raise TypeError and # fail the entire emit over one stamp. if generated.tzinfo is None: logger.warning("Naive generated_at %r in sidecar for %s; assuming UTC", generated_at, csv_path) generated = generated.replace(tzinfo=UTC) - document["stale"] = datetime.now(UTC) - generated > STALE_AFTER + return {"generated_at": generated_at, "stale": datetime.now(UTC) - generated > STALE_AFTER} + + +def _stamp_freshness(document: dict, csv_path: Path) -> None: + """Attach ``generated_at``/``stale`` from the source CSV's sidecar, if any.""" + document.update(_freshness(csv_path)) def _rows(frame: pd.DataFrame) -> list[dict]: @@ -458,6 +464,67 @@ def _org_views(org: str, org_data_dir: Path, org_dir: Path) -> list[dict]: return refs +def _source_sections( + org: str, table_sources: list[tuple[str, dict]], chart_sections: list[dict] +) -> dict[str, list[dict]]: + """Each table file the org published -> the dashboard sections it feeds. + + Tables by their documents' ``source`` (``table_sources`` pairs a file with the + card that shows it); chart cards by the CSVs their charts and downloads read + (from the spec). Only sections emitted for this org are named, so a link + never leads to a section the org lacks. + """ + index: dict[str, list[dict]] = {} + + def add(name: str | None, entry: dict) -> None: + if name and entry not in index.setdefault(name, []): + index[name].append(entry) + + for name, ref in table_sources: + add(name, {"macro": ref["macro"], "id": ref["id"], "title": ref["title"]}) + emitted = {section["id"]: section for section in chart_sections} + for macro in CHART_MACROS: + for spec in macro["charts"].get(org) or macro["charts"].get("*", []): + section = emitted.get(spec["id"]) + if section is None: + continue + entry = {"macro": section["macro"], "id": section["id"], "title": section["title"]} + for source in spec.get("interactive_sources", {}).values(): + add(source.get("file"), entry) + add(source.get("edges_file"), entry) + csv = spec.get("csv") + for name in csv.values() if isinstance(csv, dict) else [csv]: + add(name, entry) + return index + + +def _org_entities( + org: str, + org_data_dir: Path, + org_dir: Path, + sources: dict[str, list[dict]] | None = None, +) -> dict | None: + """Emit the org's repository and contributor documents; their manifest entry, or None. + + The whole ``entities/`` tree is rewritten each emit, so a repository or + person no longer in the data does not leave a stale document behind. + """ + entities_dir = org_dir / "entities" + if entities_dir.exists(): + shutil.rmtree(entities_dir) + built = build_entity_documents(org, org_data_dir, _freshness, sources) + if built is None: + return None + api_dir = org_dir.parent + for relative, document in built.documents.items(): + path = api_dir / relative + path.parent.mkdir(parents=True, exist_ok=True) + # Compact: hundreds of small documents, each fetched on demand. + path.write_text(json.dumps(document, separators=(",", ":"), allow_nan=False), encoding="utf-8") + logger.info("Entity documents for %s: %d", org, len(built.documents)) + return built.manifest_entry + + def _metric_tiles(family, org_data_dir: Path) -> list[dict]: """The macro's headline tiles as JSON objects, [] when none apply. @@ -529,6 +596,8 @@ def emit_data_api() -> Path: org_dir.mkdir(parents=True, exist_ok=True) org_data_dir = paths.ORG_DATA_DIR / org sections = [] + # (source CSV, the card showing it): where an entity view's figures link back to. + table_sources: list[tuple[str, dict]] = [] for family in TABLE_FAMILIES.values(): group_of = family.SECTION_GROUP_OF # SECTION_ORDER, not SECTION_SPECS: the order groups sections @@ -542,6 +611,8 @@ def emit_data_api() -> Path: continue document["macro"] = family.CHART_MACRO["name"] sections.append(_write_section(document, org, org_dir)) + card = sections[-1] + table_sources.extend((variant["source"], card) for variant in document.get("variants", [document])) # A role-tabbed card absorbs what used to be sibling sections. # Each absorbed variant keeps its own document *and* its own # manifest entry, tagged with the card that now renders it: @@ -559,6 +630,7 @@ def emit_data_api() -> Path: sections.append(_write_section(absorbed, org, org_dir)) chart_sections = _org_chart_sections(org, org_data_dir, org_dir) views = _org_views(org, org_data_dir, org_dir) + entities = _org_entities(org, org_data_dir, org_dir, _source_sections(org, table_sources, chart_sections)) if sections or chart_sections or views: metrics = { family.CHART_MACRO["name"]: tiles @@ -571,6 +643,11 @@ def emit_data_api() -> Path: "views": views, "metrics": metrics, } + # The repository and contributor detail views: an index of each, + # whose rows name their lazily fetched documents. Additive: absent + # when the entity tables were not produced. + if entities: + manifest["orgs"][org]["entities"] = entities manifest_path = api_dir / "manifest.json" manifest_path.write_text(json.dumps(manifest, indent=1), encoding="utf-8") diff --git a/src/hiero_analytics/export/entity_views.py b/src/hiero_analytics/export/entity_views.py new file mode 100644 index 000000000..2fadd2093 --- /dev/null +++ b/src/hiero_analytics/export/entity_views.py @@ -0,0 +1,484 @@ +"""The repository and contributor detail documents of the data API. + +Built from the ``entity_activity`` pipeline's tables +(:mod:`hiero_analytics.dashboard_spec.entities` names them) plus, for a +repository, the optional per-repository tables other pipelines write. Nothing is +re-aggregated here: every count was computed from the underlying events by the +pipeline, per window. This module only selects each entity's rows, completes +periods with no activity as zeros, and shapes the documents: + +- two **indexes** per org, listing every repository and contributor with a detail + document, so the dashboard knows which names to link, and +- one **detail document** per repository and per contributor, fetched only when a + reader opens it. + +Documents carry ``schema_version: 1``. Like the rest of v1 they only ever gain +fields; a breaking change moves to a new API version. +""" + +from __future__ import annotations + +import json +import logging +import re +from collections.abc import Callable +from dataclasses import dataclass, field +from datetime import datetime, timedelta +from pathlib import Path + +import pandas as pd + +from hiero_analytics.analysis.entity_activity import FAMILY_FIELDS +from hiero_analytics.dashboard_spec import entities as spec +from hiero_analytics.domain.periods import ACTIVITY_PERIODS +from hiero_analytics.domain.repos import bare_repo +from hiero_analytics.export.chart_data import chart_document + +logger = logging.getLogger(__name__) + +SCHEMA_VERSION = 1 +ALL_TIME = "all" +COUNT_FIELDS = tuple(spec.COUNT_LABELS) + +# Ids become file names and URL values: lower-case GitHub login / repository +# characters only, never a path separator. A leading dot (``.github``) becomes +# an underscore so no document is a hidden file; the dashboard finds ids through +# the index, so they need not be derivable from the name. +_ID = re.compile(r"^[a-z0-9_][a-z0-9._-]*$") + +Freshness = Callable[[Path], dict] +# A published table's file name -> the dashboard sections it feeds ({macro, id, title}). +SourceSections = dict[str, list[dict]] + + +@dataclass +class EntityDocuments: + """The org's entity documents keyed by API-relative path, and its manifest entry.""" + + manifest_entry: dict + documents: dict[str, dict] = field(default_factory=dict) + + +def entity_id(name: str) -> str | None: + """The id a repository (bare or ``owner/repo``) or login is published under, or None.""" + candidate = bare_repo(str(name)).strip().lower() + if candidate.startswith("."): + candidate = f"_{candidate.lstrip('.')}" + return candidate if _ID.match(candidate) and candidate != "_" else None + + +def _read(path: Path) -> pd.DataFrame | None: + if not path.exists(): + return None + return pd.read_csv(path, dtype={"repo": str, "contributor": str, "period": str, "month": str}) + + +def _grouped(frame: pd.DataFrame | None, key: str) -> dict[str, pd.DataFrame]: + """A table's rows split by ``key`` once, rather than filtered per entity.""" + return {} if frame is None or frame.empty else {str(name): rows for name, rows in frame.groupby(key)} + + +def _json_rows(frame: pd.DataFrame) -> list[dict]: + """JSON-safe records: NaN -> null, numpy scalars -> Python.""" + return json.loads(frame.to_json(orient="records", date_format="iso") or "[]") + + +def _windows(window_end: str | None, data_through: str | None = None) -> dict: + """The dates each published window covers; ``end`` is when the analysis ran. + + ``data_through`` is the latest tracked event in the organisation: a reader can + tell a quiet week from activity datasets that had not been refreshed. + """ + end = datetime.fromisoformat(window_end) if window_end else None + periods = [ + { + "key": period.key, + "label": period.label, + "days": period.days, + "start": (end - timedelta(days=period.days)).isoformat() if end and period.days else None, + } + for period in ACTIVITY_PERIODS + ] + return { + "end": window_end, + "data_through": data_through, + "periods": [*periods, {"key": ALL_TIME, "label": "All time", "days": None}], + } + + +def _summary(rows: pd.DataFrame, extra: tuple[str, ...]) -> dict: + """Counts per window, zero-filled for windows with no tracked activity.""" + fields = (*COUNT_FIELDS, "total_actions", *extra) + by_period = rows.set_index("period") + summary = {} + for key in (ALL_TIME, *(period.key for period in ACTIVITY_PERIODS)): + row = by_period.loc[key] if key in by_period.index else None + summary[key] = {name: int(row[name]) if row is not None else 0 for name in fields} + return summary + + +def _mix(summary: dict) -> dict: + """The work mix per window: each family's actions and integer share of all actions.""" + mix = {} + for key, counts in summary.items(): + total = counts["total_actions"] + mix[key] = {} + for family, fields in FAMILY_FIELDS.items(): + count = sum(counts[name] for name in fields) + mix[key][family] = {"count": count, "share": round(count / total * 100) if total else 0} + return mix + + +def _trend(rows: pd.DataFrame | None, key: str, source: Path, org: str, freshness: dict) -> dict | None: + """One entity's monthly activity (its rows of a monthly table) as a stacked timeseries chart.""" + if rows is None or rows.empty: + return None + rows = rows.drop(columns=key) + document = chart_document( + { + "kind": "timeseries", + "category": "month", + "frequency": "month", + "category_label": "Month (UTC)", + "series": [{"key": field_name, "label": label} for field_name, label in spec.COUNT_LABELS.items()], + "metric": "tracked_actions", + "unit": "Tracked actions", + "population": spec.REPOSITORY_POPULATION if key == "repo" else spec.CONTRIBUTOR_POPULATION, + "mark": "bar", + "stacked": True, + }, + source, + org, + freshness.get("generated_at"), + table=rows.reset_index(drop=True), + ) + # One id for every entity's trend: a reader's chart settings (style, span, + # hidden series) carry from one detail view to the next. + document["id"] = f"{'repository' if key == 'repo' else 'contributor'}-trend" + return {**document, **freshness} + + +def _activity_table( + pairs: pd.DataFrame, + *, + table_id: str, + title: str, + description: str, + first: tuple[str, str], + columns: list[tuple], + source: str, + freshness: dict, +) -> dict: + """A section-shaped table (all-time rows plus period variants) the dashboard's table renders.""" + keys = [first[0], *(column[0] for column in columns)] + + def rows(period: str) -> list[dict]: + scoped = pairs[pairs["period"] == period].sort_values("last_active", ascending=False) + return _json_rows(scoped[keys]) + + document = { + "id": table_id, + "title": title, + "description": description, + "source": source, + "columns": [ + {"key": first[0], "label": first[1]}, + *({"key": key, "label": label, **({"format": fmt[0]} if fmt else {})} for key, label, *fmt in columns), + ], + "rows": rows(ALL_TIME), + "periods": {period.key: rows(period.key) for period in ACTIVITY_PERIODS}, + **freshness, + } + document["row_count"] = len(document["rows"]) + return document + + +_COUNT_COLUMNS = [(name, label, "number") for name, label in spec.COUNT_LABELS.items()] +_TOTAL_COLUMN = ("total_actions", "All tracked actions", "number") +_LAST_ACTIVE_COLUMN = ("last_active", "Last active (UTC)", "date") + + +def _related_row(frame: pd.DataFrame, repo_id: str) -> pd.DataFrame: + """The rows of a per-repository table for one repository (bare or owner/repo names).""" + if "repo" not in frame.columns: + return frame.iloc[0:0] + ids = frame["repo"].astype(str).map(entity_id) + return frame[ids == repo_id] + + +class _Tables: + """Each optional table read once per emit (None when it was not produced).""" + + def __init__(self, org_data_dir: Path, freshness: Freshness) -> None: + self._dir, self._freshness, self._frames, self._fresh = org_data_dir, freshness, {}, {} + + def frame(self, name: str) -> pd.DataFrame | None: + if name not in self._frames: + self._frames[name] = _read(self._dir / name) + return self._frames[name] + + def freshness(self, name: str) -> dict: + if name not in self._fresh: + self._fresh[name] = self._freshness(self._dir / name) + return self._fresh[name] + + +def _links(files: list[str], sources: SourceSections) -> list[dict]: + """The published dashboard sections these tables feed, each once, in file order.""" + links: list[dict] = [] + for name in files: + for section in sources.get(name, []): + if all(link["id"] != section["id"] for link in links): + links.append(section) + return links + + +def _related(repo_id: str, tables: _Tables, sources: SourceSections) -> tuple[list[dict], list[str]]: + """The optional release, governance, onboarding and security sections for one repository. + + Returns ``(sections, unavailable)``: a section whose table was not produced by + this run is named in ``unavailable`` instead; one produced without a row for + this repository is published with no fields, which the view states. + """ + sections, unavailable = [], [] + for section_id, declared in spec.REPO_RELATED.items(): + frame = tables.frame(declared["file"]) + if frame is None: + unavailable.append(declared["title"]) + continue + files = [declared["file"]] + fields = [] + for part in [{"file": declared["file"], "columns": declared["columns"]}, *declared.get("extra", [])]: + part_frame = tables.frame(part["file"]) + if part_frame is None: + continue + if part["file"] not in files: + files.append(part["file"]) + match = _related_row(part_frame, repo_id) + if match.empty: + continue + row = _json_rows(match.head(1))[0] + for key, label, *fmt in part["columns"]: + if key in row and row[key] is not None: + fields.append( + {"key": key, "label": label, "value": row[key], **({"format": fmt[0]} if fmt else {})} + ) + section = { + "id": section_id, + "title": declared["title"], + "source": files, + "fields": fields, + **tables.freshness(declared["file"]), + } + if listing := declared.get("list"): + list_frame = tables.frame(listing["file"]) + if list_frame is not None: + matches = _related_row(list_frame, repo_id) + keys = [column[0] for column in listing["columns"] if column[0] in matches.columns] + if not matches.empty and keys: + ascending = listing["sort"] != "published_at" + matches = matches.sort_values(listing["sort"], ascending=ascending) + if limit := listing.get("limit"): + matches = matches.head(limit) + section["list"] = { + "title": listing["title"], + "source": listing["file"], + "columns": [ + {"key": key, "label": label, **({"format": fmt[0]} if fmt else {})} + for key, label, *fmt in listing["columns"] + if key in keys + ], + "rows": _json_rows(matches[keys]), + **tables.freshness(listing["file"]), + } + if listing["file"] not in files: + files.append(listing["file"]) + # Every figure leads back to the dashboard section it came from. + section["links"] = _links(files, sources) + sections.append(section) + return sections, unavailable + + +def _head(kind: str, org: str, entity: str, freshness: dict, window: dict) -> dict: + return { + "schema_version": SCHEMA_VERSION, + "kind": kind, + "org": org, + "id": entity, + **freshness, + "scope": spec.SCOPE, + "methodology": spec.METHODOLOGY, + "limits": spec.LIMITS, + "window": window, + } + + +def build_entity_documents( + org: str, + org_data_dir: Path, + freshness: Freshness, + sources: SourceSections | None = None, +) -> EntityDocuments | None: + """Every entity document for ``org``, or None when the entity tables were not produced. + + ``sources`` maps each table the org published to the dashboard sections it + feeds, so a repository's joined figures can link back to their evidence. + """ + repos = _read(org_data_dir / spec.REPO_ACTIVITY_FILE) + contributors = _read(org_data_dir / spec.CONTRIBUTOR_ACTIVITY_FILE) + pairs = _read(org_data_dir / spec.PAIR_ACTIVITY_FILE) + if repos is None or contributors is None or pairs is None: + return None + tables = _Tables(org_data_dir, freshness) + repo_monthly = _grouped(tables.frame(spec.REPO_MONTHLY_FILE), "repo") + contributor_monthly = _grouped(tables.frame(spec.CONTRIBUTOR_MONTHLY_FILE), "contributor") + pairs_by_repo = _grouped(pairs, "repo") + pairs_by_contributor = _grouped(pairs, "contributor") + repo_freshness = freshness(org_data_dir / spec.REPO_ACTIVITY_FILE) + contributor_freshness = freshness(org_data_dir / spec.CONTRIBUTOR_ACTIVITY_FILE) + pair_freshness = freshness(org_data_dir / spec.PAIR_ACTIVITY_FILE) + + def stamp(column: str) -> str | None: + return next(iter(repos[column].dropna()), None) if column in repos else None + + window = _windows(stamp("window_end"), stamp("data_through")) + base = f"{org}/entities" + out = EntityDocuments(manifest_entry={}) + + repo_index, seen = [], set() + for repo_key, rows in repos.groupby("repo"): + full_name = str(repo_key) + repo_id = entity_id(full_name) + if repo_id is None or repo_id in seen: + logger.warning("Skipping repository %r: no unique, safe id", full_name) + continue + seen.add(repo_id) + name = bare_repo(full_name) + summary = _summary(rows, ("active_contributors",)) + all_time = rows[rows["period"] == ALL_TIME].iloc[0] + repo_pairs = pairs_by_repo.get(full_name, pairs.iloc[0:0]) + related, unavailable = _related(repo_id, tables, sources or {}) + document = { + **_head("repository", org, repo_id, repo_freshness, window), + "name": name, + "full_name": full_name, + "github_url": f"https://github.com/{full_name}", + "population": spec.REPOSITORY_POPULATION, + "source": [spec.REPO_ACTIVITY_FILE, spec.PAIR_ACTIVITY_FILE, spec.REPO_MONTHLY_FILE], + "first_active": all_time["first_active"], + "last_active": all_time["last_active"], + "summary": summary, + "mix": _mix(summary), + "trend": _trend( + repo_monthly.get(full_name), + "repo", + org_data_dir / spec.REPO_MONTHLY_FILE, + org, + tables.freshness(spec.REPO_MONTHLY_FILE), + ), + "contributors": _activity_table( + repo_pairs, + table_id="repository-contributors", + title="Active contributors", + description="Everyone with a tracked action in this repository in the window, most recently active first.", + first=("contributor", "Contributor"), + columns=[ + *_COUNT_COLUMNS, + _TOTAL_COLUMN, + ("building_share", "Building & fixing", "percent"), + ("reviewing_share", "Reviewing & guiding", "percent"), + ("organizing_share", "Organizing & answering", "percent"), + _LAST_ACTIVE_COLUMN, + ], + source=spec.PAIR_ACTIVITY_FILE, + freshness=pair_freshness, + ), + "related": related, + "unavailable": unavailable, + } + out.documents[f"{base}/repositories/{repo_id}.json"] = document + repo_index.append( + { + "id": repo_id, + "name": name, + "full_name": full_name, + "total_actions": summary[ALL_TIME]["total_actions"], + "active_contributors": summary[ALL_TIME]["active_contributors"], + "last_active": all_time["last_active"], + } + ) + + contributor_index, seen = [], set() + for login_key, rows in contributors.groupby("contributor"): + login = str(login_key) + contributor_id = entity_id(login) + if contributor_id is None or contributor_id in seen: + logger.warning("Skipping contributor %r: no unique, safe id", login) + continue + seen.add(contributor_id) + summary = _summary(rows, ("repos_touched",)) + all_time = rows[rows["period"] == ALL_TIME].iloc[0] + document = { + **_head("contributor", org, contributor_id, contributor_freshness, window), + "login": login, + "github_url": f"https://github.com/{login}", + "population": spec.CONTRIBUTOR_POPULATION, + "source": [spec.CONTRIBUTOR_ACTIVITY_FILE, spec.PAIR_ACTIVITY_FILE, spec.CONTRIBUTOR_MONTHLY_FILE], + "first_active": all_time["first_active"], + "last_active": all_time["last_active"], + "summary": summary, + "mix": _mix(summary), + "trend": _trend( + contributor_monthly.get(login), + "contributor", + org_data_dir / spec.CONTRIBUTOR_MONTHLY_FILE, + org, + tables.freshness(spec.CONTRIBUTOR_MONTHLY_FILE), + ), + "repositories": _activity_table( + pairs_by_contributor.get(login, pairs.iloc[0:0]), + table_id="contributor-repositories", + title="Repositories", + description="Each repository with a tracked action by this person in the window, most recent first.", + first=("repo", "Repository"), + columns=[ + *_COUNT_COLUMNS, + _TOTAL_COLUMN, + ("first_active", "First active (UTC)", "date"), + _LAST_ACTIVE_COLUMN, + ], + source=spec.PAIR_ACTIVITY_FILE, + freshness=pair_freshness, + ), + } + out.documents[f"{base}/contributors/{contributor_id}.json"] = document + contributor_index.append( + { + "id": contributor_id, + "login": login, + "total_actions": summary[ALL_TIME]["total_actions"], + "repos_touched": summary[ALL_TIME]["repos_touched"], + "last_active": all_time["last_active"], + } + ) + + for kind, rows, freshness_of, population in ( + ("repositories", repo_index, repo_freshness, spec.REPOSITORY_POPULATION), + ("contributors", contributor_index, contributor_freshness, spec.CONTRIBUTOR_POPULATION), + ): + path = f"{base}/{kind}.json" + out.documents[path] = { + "schema_version": SCHEMA_VERSION, + "kind": f"{kind}-index", + "org": org, + **freshness_of, + "source": spec.REPO_ACTIVITY_FILE if kind == "repositories" else spec.CONTRIBUTOR_ACTIVITY_FILE, + "scope": spec.SCOPE, + "population": population, + "window": window, + # Where each row's detail document lives: substitute its id. + "detail_path": f"{base}/{kind}/{{id}}.json", + "rows": sorted(rows, key=lambda entry: entry["id"]), + } + out.manifest_entry[kind] = {"path": path, "count": len(rows)} + return out diff --git a/src/hiero_analytics/pipelines/__init__.py b/src/hiero_analytics/pipelines/__init__.py index dd331b999..ac2d051ea 100644 --- a/src/hiero_analytics/pipelines/__init__.py +++ b/src/hiero_analytics/pipelines/__init__.py @@ -27,6 +27,13 @@ Pipeline("contributor_profiles", "Analyze contributor profiles", args=("org", "repo")), Pipeline("maintainer_pipeline", "Run maintainer analytics pipeline", args=("org",), offline=True), Pipeline("contributor_activity", "Run contributor activity analysis", args=("org",), offline=True, extra_orgs=True), + Pipeline( + "entity_activity", + "Build per-repository and per-contributor activity for the detail views", + args=("org",), + offline=True, + extra_orgs=True, + ), Pipeline("contributor_heatmap", "Generate contributor activity heatmaps", args=("org",), offline=True), Pipeline("role_coverage", "Analyze role coverage for organization", args=("org",), offline=True), Pipeline("affiliation", "Map contributor affiliations", args=("org",), offline=True), diff --git a/src/hiero_analytics/pipelines/entity_activity.py b/src/hiero_analytics/pipelines/entity_activity.py new file mode 100644 index 000000000..ead172a83 --- /dev/null +++ b/src/hiero_analytics/pipelines/entity_activity.py @@ -0,0 +1,74 @@ +"""Build the tables behind the repository and contributor detail views. + +Reads the persisted org-wide activity datasets (pull requests, reviews, merges, +issues and label events — the same records the contributor profiles use) and +writes, per org, each repository's and each contributor's tracked actions for +the Week, 1 month and 1 year windows and all time, the activity of every +(repository, contributor) pair, and monthly trends. Every window is computed +from the events inside it, so counts such as distinct active contributors are +exact rather than summed from all-time profiles. + +Offline-capable: it only reads datasets other pipelines persisted. The +``data_api`` step publishes these tables as the detail documents +(``export/entity_views``). +""" + +from __future__ import annotations + +import logging +from datetime import UTC, datetime + +from hiero_analytics.analysis.contributor_activity_profile import combined_activity_events +from hiero_analytics.analysis.entity_activity import ( + contributor_activity, + monthly_activity, + repo_activity, + repo_contributor_activity, +) +from hiero_analytics.config.paths import ORG +from hiero_analytics.dashboard_spec import entities as spec +from hiero_analytics.domain.periods import ACTIVITY_PERIODS +from hiero_analytics.export.save import save_dataframe +from hiero_analytics.pipelines._shared import load_contributor_activity, load_issue_label_events, org_context + +logger = logging.getLogger(__name__) + +# All time, then the shared activity periods. "all" names the all-time rows in +# these long-form tables only; the published documents keep the API's +# convention of all-time as the base with period variants beside it. +ALL_TIME = "all" + + +def main(org: str = ORG) -> None: + """Write the entity-activity tables for ``org``.""" + client, org_data_dir, _ = org_context(org) + records = load_contributor_activity(client, org) + label_events = load_issue_label_events(client, org) + events = combined_activity_events(records, label_events) + logger.info("Entity activity for %s from %d tracked events", org, len(events)) + + # One moment ends every window, and it is written into the tables so the + # published documents can state the exact dates each window covers. + now = datetime.now(UTC) + windows = [(ALL_TIME, None), *((period.key, period.cutoff(now)) for period in ACTIVITY_PERIODS)] + tables = { + spec.REPO_ACTIVITY_FILE: repo_activity(events, windows), + spec.CONTRIBUTOR_ACTIVITY_FILE: contributor_activity(events, windows), + spec.PAIR_ACTIVITY_FILE: repo_contributor_activity(events, windows), + } + # The latest event recorded anywhere in the org: when it is well before the + # window end, the activity datasets are behind and recent windows undercount. + latest = events["occurred_at"].max() if not events.empty else None + stamps = {"window_end": now.isoformat(), "data_through": latest.isoformat() if latest is not None else None} + for name, frame in tables.items(): + save_dataframe(frame.assign(**stamps), org_data_dir / name) + save_dataframe(monthly_activity(events, "repo"), org_data_dir / spec.REPO_MONTHLY_FILE) + save_dataframe(monthly_activity(events, "contributor"), org_data_dir / spec.CONTRIBUTOR_MONTHLY_FILE) + + repos = tables[spec.REPO_ACTIVITY_FILE] + contributors = tables[spec.CONTRIBUTOR_ACTIVITY_FILE] + logger.info( + "Entity activity: %d repositories, %d contributors", + repos.loc[repos["period"] == ALL_TIME, "repo"].nunique(), + contributors.loc[contributors["period"] == ALL_TIME, "contributor"].nunique(), + ) diff --git a/tests/analysis/test_entity_activity.py b/tests/analysis/test_entity_activity.py new file mode 100644 index 000000000..3e1e0ac45 --- /dev/null +++ b/tests/analysis/test_entity_activity.py @@ -0,0 +1,172 @@ +"""Per-window repository and contributor activity, counted from the events.""" + +from __future__ import annotations + +from datetime import UTC, datetime, timedelta + +import pandas as pd +import pytest + +from hiero_analytics.analysis.contributor_activity_profile import combined_activity_events +from hiero_analytics.analysis.entity_activity import ( + contributor_activity, + monthly_activity, + repo_activity, + repo_contributor_activity, +) +from hiero_analytics.data_sources.models import ContributorActivityRecord, IssueTimelineEventRecord + +NOW = datetime(2026, 7, 1, 12, tzinfo=UTC) +WINDOWS = [("all", None), ("7d", NOW - timedelta(days=7)), ("30d", NOW - timedelta(days=30))] + + +def _act(repo, actor, activity_type, days_ago, number=1, target_author=None): + return ContributorActivityRecord( + repo=f"org/{repo}", + activity_type=activity_type, + actor=actor, + occurred_at=NOW - timedelta(days=days_ago), + target_type="issue" if activity_type == "authored_issue" else "pull_request", + target_number=number, + target_author=target_author or actor, + ) + + +def _label(repo, actor, days_ago, event_type="labeled"): + return IssueTimelineEventRecord( + repo=f"org/{repo}", + issue_number=9, + event_type=event_type, + occurred_at=NOW - timedelta(days=days_ago), + label="bug", + actor=actor, + ) + + +@pytest.fixture +def events() -> pd.DataFrame: + """Tracked events across two repositories, with a bot and a label removal mixed in.""" + records = [ + # alice: a PR 2 days ago in sdk, an issue 20 days ago in sdk, a PR 100 days ago in docs. + _act("sdk", "alice", "authored_pull_request", 2, number=1), + _act("sdk", "alice", "authored_issue", 20, number=5), + _act("docs", "alice", "authored_pull_request", 100, number=2), + # bob reviews and merges alice's sdk PR: the merge is bob's, not alice's. + _act("sdk", "bob", "reviewed_pull_request", 1, number=1, target_author="alice"), + _act("sdk", "bob", "merged_pull_request", 1, number=1, target_author="alice"), + # A bot's action is never a contributor's. + _act("sdk", "dependabot[bot]", "authored_pull_request", 1, number=3), + ] + labels = [ + _label("sdk", "carol", 3), + # Removing a label is not applying one. + _label("sdk", "carol", 3, event_type="unlabeled"), + _label("docs", "carol", 40), + ] + return combined_activity_events(records, labels) + + +def _row(frame: pd.DataFrame, **match) -> dict: + selected = frame + for key, value in match.items(): + selected = selected[selected[key] == value] + assert len(selected) == 1, match + return selected.iloc[0].to_dict() + + +def test_repository_counts_each_window_from_its_own_events(events): + """Each window counts only the events inside it, by type.""" + table = repo_activity(events, WINDOWS) + + sdk_all = _row(table, repo="org/sdk", period="all") + assert (sdk_all["prs_opened"], sdk_all["reviews_given"], sdk_all["merges_done"]) == (1, 1, 1) + assert (sdk_all["issues_opened"], sdk_all["labels_applied"], sdk_all["total_actions"]) == (1, 1, 5) + assert sdk_all["active_contributors"] == 3 # alice, bob, carol; not the bot + + sdk_week = _row(table, repo="org/sdk", period="7d") + # The issue (20 days ago) is outside the week; the label (3 days) is inside. + assert (sdk_week["issues_opened"], sdk_week["labels_applied"], sdk_week["total_actions"]) == (0, 1, 4) + assert sdk_week["active_contributors"] == 3 + + # docs had a PR 100 days ago and a label 40 days ago: nothing in the last month. + assert table[(table["repo"] == "org/docs") & (table["period"] == "30d")].empty + assert _row(table, repo="org/docs", period="all")["active_contributors"] == 2 + + +def test_active_contributors_are_distinct_people_not_a_sum(events): + """Active contributors are distinct people, while totals still add up per person.""" + # Two actions by the same person in one repository count one active contributor. + table = repo_activity(events, WINDOWS) + assert _row(table, repo="org/sdk", period="30d")["active_contributors"] == 3 + per_person = repo_contributor_activity(events, WINDOWS) + scoped = per_person[(per_person["repo"] == "org/sdk") & (per_person["period"] == "30d")] + assert scoped["total_actions"].sum() == _row(table, repo="org/sdk", period="30d")["total_actions"] + + +def test_merges_are_credited_to_the_merger(events): + """A merge counts for whoever merged, not the pull request's author.""" + table = contributor_activity(events, WINDOWS) + assert _row(table, contributor="bob", period="all")["merges_done"] == 1 + assert _row(table, contributor="alice", period="all")["merges_done"] == 0 + + +def test_contributor_work_mix_and_repositories_per_window(events): + """Work mix, shares, repositories touched and the active span follow the window.""" + table = contributor_activity(events, WINDOWS) + + alice_all = _row(table, contributor="alice", period="all") + assert alice_all["repos_touched"] == 2 + assert (alice_all["building_and_fixing"], alice_all["organizing_and_answering"]) == (2, 1) + assert (alice_all["building_share"], alice_all["organizing_share"], alice_all["reviewing_share"]) == (67, 33, 0) + + alice_week = _row(table, contributor="alice", period="7d") + assert (alice_week["repos_touched"], alice_week["total_actions"], alice_week["building_share"]) == (1, 1, 100) + + bob = _row(table, contributor="bob", period="7d") + assert (bob["reviewing_and_guiding"], bob["reviewing_share"]) == (2, 100) + + # The first and last tracked action span the window's own events. + assert alice_all["first_active"] == NOW - timedelta(days=100) + assert alice_all["last_active"] == NOW - timedelta(days=2) + assert "dependabot[bot]" not in set(table["contributor"]) + + +def test_pairs_break_a_contributor_down_by_repository(events): + """The pair table splits one person's actions by repository.""" + table = repo_contributor_activity(events, WINDOWS) + alice_docs = _row(table, repo="org/docs", contributor="alice", period="all") + assert (alice_docs["prs_opened"], alice_docs["total_actions"], alice_docs["building_share"]) == (1, 1, 100) + assert table[(table["contributor"] == "alice") & (table["period"] == "7d")]["repo"].tolist() == ["org/sdk"] + carol_sdk = _row(table, repo="org/sdk", contributor="carol", period="all") + assert (carol_sdk["labels_applied"], carol_sdk["organizing_share"]) == (1, 100) + + +def test_monthly_activity_by_repository_and_contributor(events): + """Monthly counts per repository and per contributor, by type.""" + by_repo = monthly_activity(events, "repo") + sdk = by_repo[by_repo["repo"] == "org/sdk"].set_index("month") + assert list(sdk.index) == ["2026-06"] + assert sdk.loc[ + "2026-06", ["prs_opened", "reviews_given", "merges_done", "issues_opened", "labels_applied"] + ].tolist() == [ + 1, + 1, + 1, + 1, + 1, + ] + by_person = monthly_activity(events, "contributor") + assert by_person[by_person["contributor"] == "alice"]["month"].tolist() == ["2026-03", "2026-06"] + + +def test_no_events_give_empty_tables_with_their_columns(): + """No events still yield every table, empty but with its columns.""" + empty = combined_activity_events([], []) + for table in ( + repo_activity(empty, WINDOWS), + contributor_activity(empty, WINDOWS), + repo_contributor_activity(empty, WINDOWS), + monthly_activity(empty, "repo"), + ): + assert table.empty + assert len(table.columns) > 3 diff --git a/tests/contracts/test_output_contract.py b/tests/contracts/test_output_contract.py index 51c15a8bc..d96578faa 100644 --- a/tests/contracts/test_output_contract.py +++ b/tests/contracts/test_output_contract.py @@ -34,6 +34,7 @@ import hiero_analytics.pipelines.contributor_profiles as profiles_mod import hiero_analytics.pipelines.difficulty as difficulty_mod import hiero_analytics.pipelines.difficulty_over_time as difficulty_time_mod +import hiero_analytics.pipelines.entity_activity as entity_mod import hiero_analytics.pipelines.hiero_hackers as hackers_mod import hiero_analytics.pipelines.hip_implementation as hip_mod import hiero_analytics.pipelines.maintainer_pipeline as maintainer_mod @@ -44,6 +45,7 @@ import hiero_analytics.pipelines.run_all as run_all import hiero_analytics.pipelines.scorecard as scorecard_mod from hiero_analytics.dashboard_spec import CHART_MACROS, TABLE_FAMILIES, table_variants +from hiero_analytics.dashboard_spec import entities as entity_spec from hiero_analytics.data_sources.models import ( CodeOwnersRecord, ContributorActivityRecord, @@ -371,9 +373,9 @@ def outputs_root(tmp_path_factory) -> Path: mp.setattr(activity_mod, "fetch_org_merged_pr_difficulty_graphql", lambda _c, _org, **_k: REPO_PRS) for mod in (maintainer_mod, heatmap_mod, role_coverage_mod, affiliation_mod): mp.setattr(mod, "fetch_governance_config", lambda *_a, **_k: GOVERNANCE) - for mod in (maintainer_mod, heatmap_mod, role_coverage_mod, affiliation_mod, activity_mod): + for mod in (maintainer_mod, heatmap_mod, role_coverage_mod, affiliation_mod, activity_mod, entity_mod): mp.setattr(mod, "load_contributor_activity", lambda _c, org: _org_activity(org)) - for mod in (role_coverage_mod, activity_mod): + for mod in (role_coverage_mod, activity_mod, entity_mod): mp.setattr(mod, "load_issue_label_events", lambda _c, _org: TIMELINE) mp.setattr(affiliation_mod, "load_affiliations", lambda: AFFILIATIONS) mp.setattr(affiliation_mod, "load_manual_logins", set) @@ -604,6 +606,8 @@ def test_no_orphan_org_level_outputs(outputs_root: Path): for name in (source["file"], source.get("edges_file")) if name ) + # The detail views' tables are published as entity documents, not sections. + spec_csvs.update(entity_spec.ENTITY_FILES) period_suffixes = tuple(f"_{period.key}.csv" for period in ACTIVITY_PERIODS) orphans = [] @@ -673,3 +677,38 @@ def test_every_produced_chart_has_interactive_data(outputs_root: Path): assert variant.get("interactive"), variant["file"] target = outputs_root / "data/api/v1" / variant["interactive"]["path"] assert target.exists() + + +def test_data_api_publishes_entity_indexes_and_their_documents(outputs_root: Path): + """Every org's entity indexes resolve, row by row, to a detail document of their kind.""" + api_dir = outputs_root / "data" / "api" / "v1" + manifest = json.loads((api_dir / "manifest.json").read_text()) + for org in (PRIMARY, HACKERS): + entry = manifest["orgs"][org]["entities"] + for kind, detail_kind in (("repositories", "repository"), ("contributors", "contributor")): + index = json.loads((api_dir / entry[kind]["path"]).read_text()) + assert index["rows"], f"{org} publishes no {kind}" + assert entry[kind]["count"] == len(index["rows"]) + for row in index["rows"]: + document = json.loads((api_dir / index["detail_path"].format(id=row["id"])).read_text()) + assert (document["schema_version"], document["kind"], document["id"]) == (1, detail_kind, row["id"]) + assert set(document["summary"]) == {"all", "7d", "30d", "365d"} + assert document["scope"] and document["methodology"] and document["population"] + assert "generated_at" in document + + +def test_repository_views_join_tables_the_pipelines_produce(outputs_root: Path): + """Each optional table a repository view joins is one the run actually writes. + + A renamed release, governance, onboarding or security table would otherwise + quietly turn into "unavailable" on every repository. + """ + org_data = outputs_root / "data" / "org" / PRIMARY + declared = set() + for section in entity_spec.REPO_RELATED.values(): + declared.add(section["file"]) + declared.update(part["file"] for part in section.get("extra", [])) + if listing := section.get("list"): + declared.add(listing["file"]) + missing = sorted(name for name in declared if not (org_data / name).exists()) + assert not missing, f"repository views join tables no pipeline produced: {missing}" diff --git a/tests/export/test_chart_data.py b/tests/export/test_chart_data.py index b6b6df9ae..3dffc44b0 100644 --- a/tests/export/test_chart_data.py +++ b/tests/export/test_chart_data.py @@ -461,3 +461,19 @@ def test_median_reference_is_computed_from_the_published_rows(tmp_path): assert chart_document({**source, "value_max": 10}, path, "org")["value_max"] == 10 fixed = {"value": 50, "label": "Majority"} assert chart_document({**source, "reference": fixed}, path, "org")["reference"] == fixed + + +def test_an_in_memory_table_builds_the_same_chart_as_its_csv(tmp_path): + """A slice of a larger table charts exactly like the same rows read from a CSV.""" + path = write_counts(tmp_path, "month", ["2026-01", "2026-03"], [4, 7]) + from_file = chart_document(roles("month"), path, "org", "2026-03-10T00:00:00+00:00") + from_table = chart_document( + roles("month"), path, "org", "2026-03-10T00:00:00+00:00", table=pd.read_csv(path, dtype={"month": str}) + ) + assert from_table == from_file + + +def test_an_in_memory_table_only_builds_series_charts(tmp_path): + """Matrix, network and events documents still read their own files.""" + with pytest.raises(ValueError, match="only build a timeseries or categories"): + chart_document({"kind": "matrix"}, tmp_path / "x.csv", "org", table=pd.DataFrame()) diff --git a/tests/export/test_entity_views.py b/tests/export/test_entity_views.py new file mode 100644 index 000000000..b92f61c36 --- /dev/null +++ b/tests/export/test_entity_views.py @@ -0,0 +1,327 @@ +"""Repository and contributor documents, as published through the data API.""" + +from __future__ import annotations + +import json +from datetime import UTC, datetime, timedelta +from pathlib import Path + +import pandas as pd +import pytest + +import hiero_analytics.pipelines.entity_activity as runner +from hiero_analytics.dashboard_spec import entities as spec +from hiero_analytics.data_sources.models import ContributorActivityRecord, IssueTimelineEventRecord +from hiero_analytics.export import data_api +from hiero_analytics.export.data_api import API_VERSION, emit_data_api +from hiero_analytics.export.entity_views import build_entity_documents, entity_id + +ORG = "test-org" + + +def _act(repo, actor, activity_type, days_ago, target_author=None): + return ContributorActivityRecord( + repo=f"{ORG}/{repo}", + activity_type=activity_type, + actor=actor, + occurred_at=datetime.now(UTC) - timedelta(days=days_ago), + target_type="pull_request", + target_number=1, + target_author=target_author or actor, + ) + + +RECORDS = [ + _act("hiero-sdk-js", "Alice", "authored_pull_request", 2), + _act("hiero-sdk-js", "bob", "reviewed_pull_request", 2, target_author="Alice"), + _act("hiero-sdk-js", "bob", "merged_pull_request", 2, target_author="Alice"), + _act("hiero-sdk-js", "Alice", "authored_issue", 200), + _act(".github", "bob", "authored_pull_request", 800), +] +LABELS = [ + IssueTimelineEventRecord( + repo=f"{ORG}/hiero-sdk-js", + issue_number=4, + event_type="labeled", + occurred_at=datetime.now(UTC) - timedelta(days=10), + label="good first issue", + actor="carol", + ) +] + + +@pytest.fixture +def org_data(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + """The entity tables the pipeline writes for RECORDS, in a sandboxed output tree.""" + org_dir = tmp_path / "data" / "org" / ORG + org_dir.mkdir(parents=True) + monkeypatch.setattr(runner, "org_context", lambda _org: (None, org_dir, tmp_path / "charts")) + monkeypatch.setattr(runner, "load_contributor_activity", lambda _client, _org: RECORDS) + monkeypatch.setattr(runner, "load_issue_label_events", lambda _client, _org: LABELS) + runner.main(ORG) + return org_dir + + +def _build(org_data: Path): + built = build_entity_documents(ORG, org_data, data_api._freshness) + assert built is not None + return built + + +def _related_tables(org_data: Path) -> None: + pd.DataFrame( + { + "repo": [f"{ORG}/hiero-sdk-js"], + "latest_release": ["2026-07-01"], + "days_since_last_release": [4], + "median_gap_days": [12], + "release_status": ["released"], + } + ).to_csv(org_data / "release_repo_summary.csv", index=False) + Path(f"{org_data / 'release_repo_summary.csv'}.meta.json").write_text( + json.dumps({"generated_at": "2026-07-25T10:00:00+00:00"}) + ) + pd.DataFrame( + { + "repo": [f"{ORG}/hiero-sdk-js"] * 3, + "tag_name": ["v1", "v3", "v2"], + "published_at": ["2026-01-01", "2026-07-01", "2026-04-01"], + "is_prerelease": [False, False, True], + } + ).to_csv(org_data / "release_timeline.csv", index=False) + # Governance keys repositories by bare name; both spellings match. + pd.DataFrame({"repo": ["hiero-sdk-js"], "Good First Issue": [3], "Beginner": [1]}).to_csv( + org_data / "difficulty_by_repo.csv", index=False + ) + pd.DataFrame( + {"repo": ["hiero-sdk-js"], "user": ["bob"], "granted_role": ["maintainer"], "status": ["active"]} + ).to_csv(org_data / "role_coverage_all.csv", index=False) + pd.DataFrame({"repo": ["hiero-sdk-js"], "distinct_orgs": [2], "top_org": ["Hashgraph"]}).to_csv( + org_data / "repo_affiliation_diversity.csv", index=False + ) + pd.DataFrame( + {"repo": ["hiero-sdk-js"], "maintainers": [4], "committers": [9], "triage": [2], "active_recent": [6]} + ).to_csv(org_data / "repo_activity_overview.csv", index=False) + pd.DataFrame( + {"repo": [f"{ORG}/hiero-sdk-js"], "distinct_hips_merged": [7], "matched_prs": [30], "total_prs": [900]} + ).to_csv(org_data / "hip_repo_engagement.csv", index=False) + + +def test_ids_are_safe_file_names(): + """Entity ids are lower-case, path-safe file names.""" + assert entity_id("hiero-ledger/hiero-sdk-js") == "hiero-sdk-js" + assert entity_id("Alice") == "alice" + assert entity_id("hiero-ledger/.github") == "_github" + # Only the last path segment is ever used, and it can never be "." or "..". + assert entity_id("../etc") == "etc" + assert entity_id("..") is None + assert entity_id(".") is None + assert entity_id("a/b c") is None + + +def test_indexes_list_every_entity_with_its_detail_path(org_data): + """Each index lists its entities and where their documents live.""" + built = _build(org_data) + assert built.manifest_entry == { + "repositories": {"path": f"{ORG}/entities/repositories.json", "count": 2}, + "contributors": {"path": f"{ORG}/entities/contributors.json", "count": 3}, + } + index = built.documents[f"{ORG}/entities/contributors.json"] + assert index["schema_version"] == 1 + assert index["detail_path"] == f"{ORG}/entities/contributors/{{id}}.json" + assert [row["id"] for row in index["rows"]] == ["alice", "bob", "carol"] + assert index["rows"][0]["login"] == "Alice" + assert "generated_at" in index and index["stale"] is False + repos = built.documents[f"{ORG}/entities/repositories.json"]["rows"] + assert {row["id"]: row["name"] for row in repos} == {"_github": ".github", "hiero-sdk-js": "hiero-sdk-js"} + for row in repos: + assert f"{ORG}/entities/repositories/{row['id']}.json" in built.documents + + +def test_repository_document_counts_windows_from_events(org_data): + """A repository's windows, work mix and contributor table come from its events.""" + document = _build(org_data).documents[f"{ORG}/entities/repositories/hiero-sdk-js.json"] + assert (document["kind"], document["full_name"]) == ("repository", f"{ORG}/hiero-sdk-js") + assert document["github_url"] == f"https://github.com/{ORG}/hiero-sdk-js" + week, year, all_time = (document["summary"][key] for key in ("7d", "365d", "all")) + assert (week["prs_opened"], week["reviews_given"], week["merges_done"], week["active_contributors"]) == (1, 1, 1, 2) + assert (year["issues_opened"], year["labels_applied"], year["active_contributors"]) == (1, 1, 3) + assert all_time["total_actions"] == 5 + assert document["mix"]["7d"]["reviewing_and_guiding"] == {"count": 2, "share": 67} + contributors = document["contributors"] + assert {row["contributor"] for row in contributors["periods"]["7d"]} == {"Alice", "bob"} + assert {row["contributor"] for row in contributors["rows"]} == {"Alice", "bob", "carol"} + assert contributors["columns"][0] == {"key": "contributor", "label": "Contributor"} + + +def test_documents_state_scope_source_window_and_methodology(org_data): + """Every detail document says what it counts, from where, over which dates.""" + document = _build(org_data).documents[f"{ORG}/entities/contributors/alice.json"] + assert "does not measure commits, comments, reactions" in document["scope"] + assert "individual performance" in document["scope"] + assert document["methodology"] and document["population"] + assert spec.PAIR_ACTIVITY_FILE in document["source"] + periods = {period["key"]: period for period in document["window"]["periods"]} + assert set(periods) == {"7d", "30d", "365d", "all"} + end = datetime.fromisoformat(document["window"]["end"]) + assert datetime.fromisoformat(periods["7d"]["start"]) == end - timedelta(days=7) + assert document["window"]["data_through"] + assert "generated_at" in document + + +def test_a_window_without_activity_is_zeros_not_missing(org_data): + """A quiet window is published as zeros, and its tables as empty lists.""" + document = _build(org_data).documents[f"{ORG}/entities/repositories/_github.json"] + assert document["summary"]["30d"] == { + "prs_opened": 0, + "reviews_given": 0, + "merges_done": 0, + "issues_opened": 0, + "labels_applied": 0, + "total_actions": 0, + "active_contributors": 0, + } + assert document["mix"]["30d"]["building_and_fixing"] == {"count": 0, "share": 0} + assert document["contributors"]["periods"]["7d"] == [] + + +def test_contributor_document_breaks_activity_down_by_repository(org_data): + """A contributor's document lists each repository they acted in, per window.""" + document = _build(org_data).documents[f"{ORG}/entities/contributors/bob.json"] + assert (document["login"], document["summary"]["all"]["repos_touched"]) == ("bob", 2) + assert document["summary"]["all"]["merges_done"] == 1 + repos = document["repositories"] + assert [row["repo"] for row in repos["rows"]] == [f"{ORG}/hiero-sdk-js", f"{ORG}/.github"] + assert [row["repo"] for row in repos["periods"]["30d"]] == [f"{ORG}/hiero-sdk-js"] + assert document["first_active"] < document["last_active"] + + +def test_trend_is_a_valid_monthly_chart_document(org_data): + """The trend is a gap-filled monthly timeseries the dashboard's charts accept.""" + trend = _build(org_data).documents[f"{ORG}/entities/contributors/alice.json"]["trend"] + assert (trend["schema_version"], trend["kind"], trend["frequency"]) == (1, "timeseries", "month") + assert [series["key"] for series in trend["series"]] == list(spec.COUNT_LABELS) + buckets = [row["bucket"] for row in trend["rows"]] + # Gap-filled from the first active month to the generation month. + assert buckets == sorted(buckets) and len(buckets) >= 7 + assert sum(row["prs_opened"] for row in trend["rows"]) == 1 + assert trend["rows"][-1]["partial"] is True + + +def test_related_sections_join_available_tables_and_name_the_missing(org_data): + """Optional per-repository tables join by name; an unproduced one is named.""" + _related_tables(org_data) + document = _build(org_data).documents[f"{ORG}/entities/repositories/hiero-sdk-js.json"] + related = {section["id"]: section for section in document["related"]} + releases = related["releases"] + assert {field["key"]: field["value"] for field in releases["fields"]}["days_since_last_release"] == 4 + assert [row["tag_name"] for row in releases["list"]["rows"]] == ["v3", "v2", "v1"] + assert releases["generated_at"] == "2026-07-25T10:00:00+00:00" + governance = related["governance"] + # Role counts with the active share lead; affiliation facts join from their own table. + assert [field["key"] for field in governance["fields"]] == [ + "maintainers", + "committers", + "triage", + "active_recent", + "distinct_orgs", + "top_org", + ] + assert governance["list"]["rows"] == [{"user": "bob", "granted_role": "maintainer", "status": "active"}] + hips = {field["key"]: field["value"] for field in related["hips"]["fields"]} + assert hips == {"distinct_hips_merged": 7, "matched_prs": 30, "total_prs": 900} + onboarding = {field["key"]: field["value"] for field in related["onboarding"]["fields"]} + assert onboarding == {"Good First Issue": 3, "Beginner": 1} + # The scorecard pipeline did not run: said so, never silently blank. + assert document["unavailable"] == ["Security"] + # A repository with no row in a produced table gets the section with no fields. + other = _build(org_data).documents[f"{ORG}/entities/repositories/_github.json"] + assert {section["id"]: section["fields"] for section in other["related"]}["releases"] == [] + + +def test_no_entity_tables_means_no_documents(tmp_path): + """Without the pipeline's tables there is nothing to publish.""" + assert build_entity_documents(ORG, tmp_path, data_api._freshness) is None + + +def test_emit_publishes_entity_documents_and_lists_them_in_the_manifest( + org_data, tmp_path: Path, monkeypatch: pytest.MonkeyPatch +): + """The emit writes every document, lists the indexes and clears stale ones.""" + monkeypatch.setattr(data_api.paths, "DATA_DIR", tmp_path / "data") + monkeypatch.setattr(data_api.paths, "ORG_DATA_DIR", tmp_path / "data" / "org") + monkeypatch.setattr(data_api.paths, "ORG_CHARTS_DIR", tmp_path / "charts" / "org") + monkeypatch.setattr(data_api.paths, "ORG", ORG) + widgets = { + "id": "widgets", + "file": "widgets.csv", + "title": "Widgets", + "description": "All widgets.", + "columns": [("name", "widget")], + } + family = type( + "Family", + (), + { + "SECTION_SPECS": [widgets], + "SECTION_ORDER": ["widgets"], + "SECTION_GROUP_OF": {"widgets": "A group"}, + "CHART_MACRO": {"name": "Testing", "charts": {}}, + }, + ) + monkeypatch.setattr(data_api, "TABLE_FAMILIES", {"Testing": family}) + monkeypatch.setattr(data_api, "CHART_MACROS", [{"name": "Testing", "charts": {}}]) + monkeypatch.setattr(data_api, "CUSTOM_VIEW_MODULES", {}) + pd.DataFrame({"name": ["a"]}).to_csv(org_data / "widgets.csv", index=False) + api_dir = tmp_path / "data" / "api" / API_VERSION + # A document left by an earlier run for a repository no longer in the data. + stale = api_dir / ORG / "entities" / "repositories" / "retired.json" + stale.parent.mkdir(parents=True) + stale.write_text("{}") + + manifest = json.loads(emit_data_api().read_text()) + + entry = manifest["orgs"][ORG]["entities"] + assert entry["repositories"]["count"] == 2 + index = json.loads((api_dir / entry["contributors"]["path"]).read_text()) + for row in index["rows"]: + detail = json.loads((api_dir / index["detail_path"].format(id=row["id"])).read_text()) + assert detail["id"] == row["id"] + assert not stale.exists() + + +def test_related_sections_link_back_to_the_dashboard_sections_they_summarise(org_data): + """Each joined figure names the published sections its tables feed, and only those.""" + _related_tables(org_data) + sources = { + "repo_activity_overview.csv": [{"macro": "Governance", "id": "repoactivity", "title": "Repository activity"}], + "role_coverage_all.csv": [{"macro": "Governance", "id": "repo", "title": "Roles by repo"}], + "hip_repo_engagement.csv": [{"macro": "HIPs", "id": "hip-repo-engagement", "title": "HIP engagement"}], + } + built = build_entity_documents(ORG, org_data, data_api._freshness, sources) + related = { + section["id"]: section + for section in built.documents[f"{ORG}/entities/repositories/hiero-sdk-js.json"]["related"] + } + assert [link["id"] for link in related["governance"]["links"]] == ["repoactivity", "repo"] + assert related["hips"]["links"] == [{"macro": "HIPs", "id": "hip-repo-engagement", "title": "HIP engagement"}] + # A table no published section shows links nowhere. + assert related["onboarding"]["links"] == [] + + +def test_source_sections_name_only_what_the_org_published(): + """Tables map to the card that shows them (absorbed variants to their card); charts by their CSVs.""" + card = {"id": "affiliations", "macro": "Governance", "title": "Organisation affiliations", "row_count": 2} + index = data_api._source_sections( + "hiero-ledger", + [("affiliations.csv", card), ("committer_affiliations.csv", card)], + [{"id": "hip-repo-engagement", "macro": "HIPs", "title": "Which repositories engage with HIPs"}], + ) + entry = {"macro": "Governance", "id": "affiliations", "title": "Organisation affiliations"} + assert index["affiliations.csv"] == [entry] + assert index["committer_affiliations.csv"] == [entry] + assert index["hip_repo_engagement.csv"][0]["id"] == "hip-repo-engagement" + # Chart cards that were not emitted for the org are never linked. + assert all( + link["id"] == "hip-repo-engagement" for links in index.values() for link in links if link["macro"] == "HIPs" + ) diff --git a/tests/pipelines/test_entity_activity.py b/tests/pipelines/test_entity_activity.py new file mode 100644 index 000000000..c2773f324 --- /dev/null +++ b/tests/pipelines/test_entity_activity.py @@ -0,0 +1,75 @@ +"""The entity_activity pipeline writes the detail views' tables from persisted activity.""" + +from __future__ import annotations + +from datetime import UTC, datetime, timedelta + +import pandas as pd + +import hiero_analytics.pipelines.entity_activity as runner +from hiero_analytics.dashboard_spec import entities as spec +from hiero_analytics.data_sources.models import ContributorActivityRecord, IssueTimelineEventRecord + + +def _act(repo: str, actor: str, activity_type: str, days_ago: int) -> ContributorActivityRecord: + return ContributorActivityRecord( + repo=f"test-org/{repo}", + activity_type=activity_type, + actor=actor, + occurred_at=datetime.now(UTC) - timedelta(days=days_ago), + target_type="pull_request", + target_number=1, + target_author=actor, + ) + + +def _patch(monkeypatch, stub_pipeline_context, records, labels): + _, data_dir, _ = stub_pipeline_context(runner) + monkeypatch.setattr(runner, "load_contributor_activity", lambda _client, _org: records) + monkeypatch.setattr(runner, "load_issue_label_events", lambda _client, _org: labels) + return data_dir + + +def test_writes_every_table_with_windows_and_a_freshness_sidecar(monkeypatch, stub_pipeline_context): + """Every table is written per window, with its sidecar and window stamps.""" + records = [ + _act("sdk", "alice", "authored_pull_request", 2), + _act("sdk", "bob", "merged_pull_request", 40), + _act("docs", "alice", "authored_issue", 400), + ] + labels = [ + IssueTimelineEventRecord( + repo="test-org/sdk", + issue_number=3, + event_type="labeled", + occurred_at=datetime.now(UTC) - timedelta(days=1), + label="bug", + actor="carol", + ) + ] + data_dir = _patch(monkeypatch, stub_pipeline_context, records, labels) + + runner.main("test-org") + + for name in spec.ENTITY_FILES: + assert (data_dir / name).exists(), name + assert (data_dir / f"{name}.meta.json").exists(), name + repos = pd.read_csv(data_dir / spec.REPO_ACTIVITY_FILE) + assert set(repos["period"]) == {"all", "7d", "30d", "365d"} + sdk = repos[repos["repo"] == "test-org/sdk"].set_index("period") + assert sdk.loc["all", "active_contributors"] == 3 + assert sdk.loc["7d", "active_contributors"] == 2 # alice's PR and carol's label; bob merged 40 days ago + assert "30d" in sdk.index and "7d" in sdk.index + # docs was last active 400 days ago: only its all-time row exists. + assert repos[repos["repo"] == "test-org/docs"]["period"].tolist() == ["all"] + # One window end for every row, and the org's latest tracked event beside it. + assert repos["window_end"].nunique() == 1 + assert repos["data_through"].nunique() == 1 + + +def test_an_org_with_no_activity_writes_empty_tables(monkeypatch, stub_pipeline_context): + """An org with no tracked activity still gets its (empty) tables.""" + data_dir = _patch(monkeypatch, stub_pipeline_context, [], []) + runner.main("test-org") + assert pd.read_csv(data_dir / spec.REPO_ACTIVITY_FILE).empty + assert pd.read_csv(data_dir / spec.CONTRIBUTOR_MONTHLY_FILE).empty diff --git a/web/README.md b/web/README.md index d75d94d53..d6cef02b6 100644 --- a/web/README.md +++ b/web/README.md @@ -56,6 +56,15 @@ npm run dev # the app, on http://localhost:5173 `'self'` only, so it ships in the bundle via `@fontsource-variable`). Numbers that line up in columns use `tabular-nums`. +## Repository and contributor views + +Names in tables and charts open a detail view (`src/entities.ts`; the data model +is in `docs/entity-views.md`); a name with no tracked activity opens one that +says so. Render a +repository or person name with `RepoCell`/`ContributorCell` in a table, or +`EntityLink`/`EntityTick` in a chart, rather than a bare GitHub link, so every +name behaves the same: an in-dashboard link, with GitHub one separate icon away. + ## Third-party requests The CSP in `index.html` keeps everything on `'self'` except images from diff --git a/web/e2e/entities.spec.ts b/web/e2e/entities.spec.ts new file mode 100644 index 000000000..a5765657e --- /dev/null +++ b/web/e2e/entities.spec.ts @@ -0,0 +1,84 @@ +import type { Page } from '@playwright/test'; +import { ENTITY_ROUTES } from '../src/test/entityFixtures'; +import { test, expect } from './browser'; + +/** Serve the entity fixtures over the staged site's manifest and API tree. */ +async function serveEntities(page: Page) { + // The component suite's role table (alice, bob, carol) rather than the print suite's large one. + const served = new Set(['manifest.json', 'hiero-ledger/roles.json']); + for (const [path, body] of Object.entries(ENTITY_ROUTES)) { + if (!served.has(path) && !path.includes('/entities/')) continue; + await page.route(`**/data/api/v1/${path}`, (route) => route.fulfill({ json: body })); + } +} + +test('a table name opens its detail view, and browser history steps through it', async ({ + page, + browserErrors, +}) => { + await serveEntities(page); + await page.goto('./#tab=Governance'); + const roles = page.locator('#roles'); + await roles.getByRole('link', { name: 'alice', exact: true }).click(); + + await expect(page.getByRole('heading', { level: 1, name: 'alice' })).toBeVisible(); + await expect(page.getByText(/does not measure commits, comments, reactions/)).toBeVisible(); + await expect(page).toHaveURL(/entity=contributor%3Aalice/); + await expect(page.getByRole('region', { name: 'Activity by period' })).toBeVisible(); + + // The repository in alice's list opens that repository: a second history entry. + await page + .getByRole('region', { name: 'Repositories' }) + .getByRole('link', { name: 'hiero-ledger/hiero-sdk-js', exact: true }) + .click(); + await expect(page.getByRole('heading', { level: 1, name: 'hiero-sdk-js' })).toBeVisible(); + await expect( + page.getByText('Not available in this data: Security.', { exact: false }), + ).toBeVisible(); + + await page.goBack(); + await expect(page.getByRole('heading', { level: 1, name: 'alice' })).toBeVisible(); + await page.goBack(); + await expect(page.getByRole('heading', { level: 1, name: 'Governance' })).toBeVisible(); + await expect(roles).toBeVisible(); + await page.goForward(); + await expect(page.getByRole('heading', { level: 1, name: 'alice' })).toBeVisible(); + expect(browserErrors).toEqual([]); +}); + +test('a detail view prints as a document, without navigation or controls', async ({ page }) => { + await serveEntities(page); + await page.goto('./#tab=Governance&entity=repo:hiero-sdk-js'); + await expect(page.getByRole('region', { name: 'Activity by period' })).toBeVisible(); + await page.evaluate(() => window.dispatchEvent(new Event('beforeprint'))); + await page.emulateMedia({ media: 'print' }); + await expect(page.locator('html')).toHaveAttribute('data-printing', 'true'); + + await expect(page.getByRole('heading', { level: 1, name: 'hiero-sdk-js' })).toBeVisible(); + await expect(page.getByRole('region', { name: 'Active contributors' })).toBeVisible(); + for (const control of [ + page.getByRole('link', { name: /Back to Governance/, includeHidden: true }), + page.getByRole('button', { name: 'Print page', includeHidden: true }), + page.locator('[data-slot="sidebar"]'), + ]) { + for (const element of await control.all()) await expect(element).toBeHidden(); + } + await page.emulateMedia({ media: 'screen' }); + await page.evaluate(() => window.dispatchEvent(new Event('afterprint'))); +}); + +test('a figure on a repository view links back to its section, landing on it', async ({ page }) => { + await serveEntities(page); + await page.goto('./#tab=Contributors&entity=repo:hiero-sdk-js'); + await page + .getByRole('region', { name: 'HIP engagement' }) + .getByRole('link', { name: /Role holders/ }) + .click(); + const roles = page.locator('#roles'); + await expect(page).toHaveURL(/tab=Governance&.*widget=roles|widget=roles&.*tab=Governance/); + await expect(page.getByRole('heading', { level: 1, name: 'Governance' })).toBeVisible(); + // Held in place while the charts above it load, then left to the reader. + await expect(roles).toBeInViewport(); + await page.waitForTimeout(1500); + await expect(roles).toBeInViewport(); +}); diff --git a/web/src/App.tsx b/web/src/App.tsx index fe7b40d51..58c8cc80d 100644 --- a/web/src/App.tsx +++ b/web/src/App.tsx @@ -3,7 +3,7 @@ * header, a sidebar of tabs, and the active tab's tiles, glossary and section groups. */ -import { useEffect, useMemo, useRef, useState } from 'react'; +import { lazy, Suspense, useContext, useEffect, useMemo, useRef, useState } from 'react'; import { CircleAlertIcon, RotateCwIcon } from 'lucide-react'; import { Alert, AlertDescription, AlertTitle } from '@/components/ui/alert'; import { Button } from '@/components/ui/button'; @@ -31,6 +31,7 @@ import { tocEntries, type Group, type TocEntry } from './toc'; import { useHashState } from './useHashState'; import { writeParams } from './urlState'; import { FocusBar } from './components/FocusBar'; +import { EntityDirectoryContext, useEntity, useEntityDirectory } from './entities'; import { useSectionDocs } from './useSectionDocs'; import { useViewDocs } from './useViewDocs'; import { ViewCards } from './components/ViewCards'; @@ -39,7 +40,13 @@ import { PrintControls, PrintProvider } from './printing'; import { stamp } from './format'; import './print.css'; +// Loaded when a reader first opens a repository or contributor, with its charts. +const EntityView = lazy(() => import('./components/EntityView')); + const FLASH_MS = 1800; // shared link jump: flash the target for this long, then remove the highlight +// Charts above a jump target load after the tab settles and push it down mid-scroll; +// the jump keeps the target in place while the page grows, for at most this long. +const HOLD_MS = 4000; function OrgPanel({ org, @@ -110,12 +117,29 @@ function OrgPanel({ const flashTimer = useRef(undefined); useEffect(() => { if (!settled || !widget) return; + let hold: ResizeObserver | undefined; + let holdTimer: number | undefined; + // The reader taking over (scrolling, a key, a tap) ends the hold at once. + const release = () => { + hold?.disconnect(); + window.clearTimeout(holdTimer); + for (const name of ['wheel', 'touchstart', 'keydown'] as const) { + window.removeEventListener(name, release); + } + }; const jump = () => { const target = document.getElementById(absorbedInto[widget] ?? widget); target?.scrollIntoView({ block: 'start', behavior: 'smooth' }); target?.classList.add('flash'); if (flashTimer.current) window.clearTimeout(flashTimer.current); flashTimer.current = window.setTimeout(() => target?.classList.remove('flash'), FLASH_MS); + if (!target || typeof ResizeObserver === 'undefined') return; + hold = new ResizeObserver(() => target.scrollIntoView({ block: 'start' })); + hold.observe(document.body); + holdTimer = window.setTimeout(release, HOLD_MS); + for (const name of ['wheel', 'touchstart', 'keydown'] as const) { + window.addEventListener(name, release, { passive: true }); + } }; const canRequestFrame = typeof window.requestAnimationFrame === 'function'; const raf = canRequestFrame ? window.requestAnimationFrame(jump) : undefined; @@ -125,6 +149,7 @@ function OrgPanel({ window.cancelAnimationFrame(raf as number); } if (flashTimer.current) window.clearTimeout(flashTimer.current); + release(); document.getElementById(absorbedInto[widget] ?? widget)?.classList.remove('flash'); }; }, [settled, widget, absorbedInto]); @@ -241,6 +266,39 @@ function Dashboard({ const { activeMacro, shownOrg, orgHasMacro } = nav; const glossary = orgHasMacro ? manifest.macro_glossaries?.[activeMacro] : undefined; const dataAsOf = manifest.provenance.data_as_of; + const directory = useContext(EntityDirectoryContext); + const entity = useEntity(); + const footer = ( + // One footer bar: WIP notice left, provenance right — same rule, same baseline. +
+ {manifest.wip !== false && } + +
+ ); + + if (entity) { + // A repository or contributor in place of the tab; the tab (and its focus) waits underneath. + return ( + +

+ {entity.kind === 'repo' ? 'Repository' : 'Contributor'} · {shownOrg} +

+ + }> + + + + {footer} +
+ ); + } return ( @@ -300,11 +358,7 @@ function Dashboard({ )} - {/* One footer bar: WIP notice left, provenance right — same rule, same baseline. */} -
- {manifest.wip !== false && } - -
+ {footer}
); } @@ -339,6 +393,17 @@ export default function App() { const nav = manifest ? navModel(manifest, macro, org) : null; const [toc, setToc] = useState([]); + const entity = useEntity(); + // Every table and chart links the names the shown org has detail views for. + const directory = useEntityDirectory( + nav?.shownOrg ?? '', + nav ? manifest?.orgs[nav.shownOrg]?.entities : undefined, + ); + // Choosing a tab closes an open detail view, as one history entry. + const onTab = (next: string) => + entity ? writeParams({ tab: next, entity: null }, { push: true }) : setMacro(next); + // A detail view has its own table of contents, whatever the tab behind it holds. + const shownToc = nav?.orgHasMacro || entity ? toc : []; return ( // The header renders in every state; only the content beneath it changes shape. @@ -350,16 +415,17 @@ export default function App() { { - writeParams({ focus: null }); + writeParams({ focus: null, entity: null }); setOrg(next); }} - toc={nav?.orgHasMacro ? toc : []} - onTab={setMacro} + toc={shownToc} + onTab={onTab} dataAsOf={manifest?.provenance.data_as_of} />
- + {/* min-w-0: otherwise a wide table or nowrap stamp widens the page sideways. */}
@@ -374,7 +440,9 @@ export default function App() { ) : ( - + + + )}
diff --git a/web/src/api.ts b/web/src/api.ts index 3588e49b9..dafe4aad3 100644 --- a/web/src/api.ts +++ b/web/src/api.ts @@ -336,6 +336,14 @@ export interface OrgEntry { chart_sections: ChartSection[]; views?: ViewRef[]; metrics: Record; + /** The repository and contributor detail views' indexes; absent when not produced. */ + entities?: EntityRefs; +} + +/** Where an org's entity indexes live, and how many entities each lists. */ +export interface EntityRefs { + repositories?: { path: string; count: number }; + contributors?: { path: string; count: number }; } export interface Glossary { @@ -416,6 +424,119 @@ export interface SectionDoc { variants?: SectionVariant[]; } +/** A table shaped like a section: all-time rows, plus period variants (export/entity_views.py). */ +export type TableDoc = Omit; + +/** The five tracked action counts, their total and the entity's own extra count. */ +export interface EntityCounts { + prs_opened: number; + reviews_given: number; + merges_done: number; + issues_opened: number; + labels_applied: number; + total_actions: number; + /** Repositories only: distinct people with a tracked action in the window. */ + active_contributors?: number; + /** Contributors only: repositories with a tracked action by them in the window. */ + repos_touched?: number; +} + +export type WorkFamily = + 'building_and_fixing' | 'reviewing_and_guiding' | 'organizing_and_answering'; + +/** The windows a detail document counts: Week, 1 month, 1 year (their start dates) and all time. */ +export interface EntityWindow { + /** When the analysis ran; every window ends here. */ + end: string | null; + /** The latest tracked event in the organisation's data. */ + data_through?: string | null; + periods: { key: string; label: string; days: number | null; start?: string | null }[]; +} + +interface EntityDocumentBase { + schema_version: 1; + org: string; + id: string; + generated_at?: string; + stale?: boolean; + /** What the counts cover, and what they do not. */ + scope: string; + population: string; + methodology: string[]; + limits: string[]; + source: string[]; + window: EntityWindow; + github_url: string; + first_active: string | null; + last_active: string | null; + /** Counts keyed by window: `all`, `365d`, `30d`, `7d`; a quiet window is all zeros. */ + summary: Record; + mix: Record>; + /** Tracked actions per month, as a chart document; null without any. */ + trend: TimeseriesDocument | null; +} + +/** An optional per-repository table joined into a repository's view. */ +export interface RelatedSection { + id: string; + title: string; + source: string[]; + /** The dashboard sections these figures come from, each a jump target (`#tab=…&widget=…`). */ + links?: { macro: string; id: string; title: string }[]; + fields: { key: string; label: string; value: unknown; format?: ColumnFormat }[]; + generated_at?: string; + stale?: boolean; + list?: { + title: string; + source: string; + columns: ColumnSpec[]; + rows: Row[]; + generated_at?: string; + stale?: boolean; + }; +} + +export interface RepositoryDocument extends EntityDocumentBase { + kind: 'repository'; + name: string; + full_name: string; + contributors: TableDoc; + related: RelatedSection[]; + /** Titles of the optional sections whose tables this run did not produce. */ + unavailable: string[]; +} + +export interface ContributorDocument extends EntityDocumentBase { + kind: 'contributor'; + login: string; + repositories: TableDoc; +} + +export type EntityDocument = RepositoryDocument | ContributorDocument; + +export interface EntityIndexRow { + id: string; + /** Repositories: the bare and `owner/repo` names. */ + name?: string; + full_name?: string; + /** Contributors: the GitHub login as GitHub spells it. */ + login?: string; + total_actions: number; + last_active: string | null; +} + +export interface EntityIndex { + schema_version: 1; + kind: 'repositories-index' | 'contributors-index'; + org: string; + generated_at?: string; + stale?: boolean; + /** Where a row's document lives: `{id}` is replaced by its id. */ + detail_path: string; + window: EntityWindow; + rows: EntityIndexRow[]; +} + /** Deploy-relative roots: the app and the API ship together. */ const BASE = import.meta.env.BASE_URL; export const API_ROOT = `${BASE}data/api/v1`; @@ -437,6 +558,12 @@ export const fetchSection = (ref: SectionRef): Promise => export const fetchView = (ref: ViewRef): Promise => getJson(`${API_ROOT}/${ref.path}`); +export const fetchEntityIndex = (path: string): Promise => + getJson(`${API_ROOT}/${path}`); + +export const fetchEntityDocument = (path: string): Promise => + getJson(`${API_ROOT}/${path}`); + /** Raw text of a file shipped inside the API tree (e.g. a chart's CSV). */ export const fetchApiText = (path: string): Promise => fetch(`${API_ROOT}/${path}`).then((response) => { diff --git a/web/src/components/ContributorCell.tsx b/web/src/components/ContributorCell.tsx index 3926379f0..387d7a4e7 100644 --- a/web/src/components/ContributorCell.tsx +++ b/web/src/components/ContributorCell.tsx @@ -1,6 +1,12 @@ import { useState } from 'react'; +import { useEntityLink } from '../entities'; +import { GitHubLink } from './EntityLink'; -/** A GitHub login with its avatar; initials stand in when the image fails. */ +/** + * A GitHub login with its avatar; initials stand in when the image fails. With a + * detail view the name opens it, and a separate icon links to GitHub; otherwise + * the name links to GitHub. + */ export function ContributorCell({ login, compact = false, @@ -10,12 +16,15 @@ export function ContributorCell({ compact?: boolean; }) { const [failed, setFailed] = useState(false); + const detail = useEntityLink('contributor', login); if (!/^[a-zA-Z0-9][a-zA-Z0-9-]{0,38}(\[bot\])?$/.test(login)) return <>{login}; - return ( + const github = `https://github.com/${encodeURIComponent(login)}`; + const person = ( ); + if (!detail) return person; + return ( + + {person} + + + ); } diff --git a/web/src/components/DataTable.tsx b/web/src/components/DataTable.tsx index ba4b42a3c..fbbedf911 100644 --- a/web/src/components/DataTable.tsx +++ b/web/src/components/DataTable.tsx @@ -47,12 +47,18 @@ export function DataTable({ table, controls, actions, + printColumns, + printRowLimit = PRINT_ROW_LIMIT, }: { table: DataTableInstance; /** Beside the search: the switches that choose which rows (a time range). */ controls?: ReactNode; /** At the toolbar's end: what a reader does with the rows (download, an external link). */ actions?: ReactNode; + /** A detail view may print a concise set of columns while its CSV keeps the complete table. */ + printColumns?: string[]; + /** Maximum paper rows for this table; the full selection remains downloadable as CSV. */ + printRowLimit?: number; }) { const printing = usePrintMode(); const scrollRef = useRef(null); @@ -76,16 +82,19 @@ export function DataTable({ : 0; const visibleRows = printing ? [ - ...rows.slice(0, PRINT_ROW_LIMIT), + ...rows.slice(0, printRowLimit), // The screen viewport beyond the cap stays mounted (hidden) so its focused links survive. ...virtualRows - .filter((item) => item.index >= PRINT_ROW_LIMIT) + .filter((item) => item.index >= printRowLimit) .map((item) => rows[item.index]), ] : virtualized ? virtualRows.map((item) => rows[item.index]) : rows; - const columnCount = table.getVisibleFlatColumns().length; + const columnCount = + printing && printColumns + ? table.getVisibleFlatColumns().filter((column) => printColumns.includes(column.id)).length + : table.getVisibleFlatColumns().length; const total = table.getCoreRowModel().rows.length; const count = (n: number) => n.toLocaleString('en-US'); @@ -97,11 +106,11 @@ export function DataTable({ this filter are not printed.

)} - {printing && rows.length > PRINT_ROW_LIMIT && ( + {printing && rows.length > printRowLimit && (

- Showing {count(PRINT_ROW_LIMIT)} of {count(rows.length)} rows in the current order.{' '} - {count(rows.length - PRINT_ROW_LIMIT)} rows are not printed. Download CSV from this table - on the dashboard for the complete selection. + Showing {count(printRowLimit)} of {count(rows.length)} rows in the current order.{' '} + {count(rows.length - printRowLimit)} rows are not printed. Download CSV from this table on + the dashboard for the complete selection.

)}
@@ -148,6 +157,8 @@ export function DataTable({ {table.getHeaderGroups().map((headerGroup) => ( {headerGroup.headers.map((header, index) => { + if (printing && printColumns && !printColumns.includes(header.column.id)) + return null; const sorted = header.column.getIsSorted() as string; const numeric = header.column.columnDef.meta?.numeric; const SortIcon = @@ -241,11 +252,12 @@ export function DataTable({