diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 30a9e06a..e93bb4b0 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -287,9 +287,9 @@ Note that office hours are not recorded. | 2026-02-27 | 15.30 - 16.30 | ✅ Complete | | 2026-03-27 | 10.30 - 11.30 | ✅ Complete | | 2026-04-24 | 15.30 - 16.30 | ✅ Complete | -| 2026-05-29 | 10.30 - 11.30 | Planned | -| 2026-06-26 | 15.30 - 16.30 | Planned | -| 2026-07-31 | 10.30 - 11.30 | Planned | +| 2026-05-29 | 10.30 - 11.30 | ✅ Complete | +| 2026-06-26 | 15.30 - 16.30 | ✅ Complete | +| 2026-07-31 | 10.30 - 11.30 | ✅ Complete | | 2026-08-28 | 15.30 - 16.30 | Planned | | 2026-09-25 | 10.30 - 11.30 | Planned | | 2026-10-30 | 15.30 - 16.30 | Planned | diff --git a/docs/program/ecosystem.md b/docs/program/ecosystem.md new file mode 100644 index 00000000..52c9eda9 --- /dev/null +++ b/docs/program/ecosystem.md @@ -0,0 +1,49 @@ +# ProtVista adoption & ecosystem + +_A curated record of who uses, re-implements, deploys, or cites ProtVista. This single +hand/LLM-maintained table is the source of truth; its **git history** is the point-in-time record._ + +- **Type:** `package consumer` | `ProtVista-type viewer` | `fork` | `commercial/private` | `unclear`. +- **Evidence:** the strongest public pointer (repo / paper / live site), or + `private — under agreement` where there is no public artifact. +- **Since / Until:** when the project's ProtVista relationship began and, if it has, ended. The + *kind* of date is written in each cell because it means different things per type: + - **package consumer:** `(dep added)` the dependency entered package.json, or `(repo created)`; + Until `(dep removed)` / `(archived)`. + - **fork / ProtVista-type viewer:** `(repo created)` or `(paper)` when their viewer first appeared; + Until is usually `—` (no dependency to remove). + - **commercial/private:** `(interest)` when engagement began; Until when it ended. + - **unclear:** `(paper)` / `(UI mention)`. + - `—` in *Since* means the date is not yet filled (run `protvista_ecosystem.py --backfill-dates` to + fill repo-created dates); `—` in *Until* means the relationship is ongoing / not applicable. +- **Status:** current liveness (active / dormant / archived); **Last activity:** last repo push. +- Package consumers are refreshed by `scripts/protvista_ecosystem.py`; everything else is human-curated. +- **Out of scope:** generic use of individual `@nightingale-elements/*` track components for unrelated + purposes (but a ProtVista-style feature viewer built on nightingale counts as a `ProtVista-type viewer`). +- **Consent / PII:** name a commercial/private partner only with recorded consent — otherwise + `Commercial adopter ()`. Never add an unconsented name to this file. + +## Ecosystem entities + +| Project / repo | Type | Evidence | Since | Until | Status | Last activity | Note | +| --- | --- | --- | --- | --- | --- | --- | --- | +| UniProt website (`ebi-uniprot/uniprot-website`) | package consumer | https://github.com/ebi-uniprot/uniprot-website | 2020-04-29 (dep added) | — | active | 2026-04-30 | ProtVista's largest embedder (UniProtKB) | +| GIFTS curation tool (`ebi-uniprot/gifts-curation-tool`) | package consumer | https://github.com/ebi-uniprot/gifts-curation-tool | 2021-10-11 (dep added) | — | active | 2025-12-09 | EBI | +| JCVI Human Salivary Proteome Wiki (`JCVenterInstitute/HSPW-V3`) | package consumer | https://github.com/JCVenterInstitute/HSPW-V3 | 2023-05-10 (dep added) | — | active | 2026-02-04 | — | +| `KSerditov/ProteinSearch` | package consumer | https://github.com/KSerditov/ProteinSearch | 2023-07-22 (dep added) | — | dormant | 2024-02-10 | independent developer | +| `ekondrashkov/proteins` | package consumer | https://github.com/ekondrashkov/proteins | 2024-12-09 (dep added) | — | dormant | 2024-12-10 | independent developer | +| OTPSS (`maniexcelra/OTPSS`) | package consumer | https://github.com/maniexcelra/OTPSS | 2022-06-10 (dep added) | — | dormant | 2024-09-16 | personal Open Targets-derived repo (not official OT); two package.json manifests | +| Open Targets Platform (`opentargets-archive/platform-app`) | package consumer | https://github.com/opentargets-archive/platform-app | 2019-10-08 (dep added) | — (archived) | dormant | 2022-06-22 | historical — Open Targets no longer uses the package | +| Open Targets Genetics (`opentargets-archive/genetics-app`) | package consumer | https://github.com/opentargets-archive/genetics-app | 2025-01-31 (when repo was archived) | — (archived) | dormant | 2025-01-31 | historical — Open Targets no longer uses the package; adoption/archive dates uncertain (archived repo — likely a late commit, verify) | +| GlyGen (`glygener/glygen-frontend`) | unclear | https://www.glygen.org | 2021 (paper) | — | — | — | live `` deployment via CDN per bioRxiv 10.1101/2021.06.17.448729 (not declared in package.json); current use unverified | +| Pharos / TCRD (`ncats/protvista-viewer`) | fork | https://github.com/ncats/protvista-viewer | 2021-02-09 (repo created) | — | dormant | 2024-08-01 | verified: shares git history with upstream; publishes `ncats-protvista-uniprot` on npm | +| PDBe (`PDBeurope/protvista-pdb`) | ProtVista-type viewer | https://doi.org/10.1101/2022.07.22.500790 | 2022 (paper) | — | active | 2025-07-08 | independent re-implementation (no shared git history) | +| RCSB Saguaro 1D Feature Viewer (`rcsb/rcsb-saguaro`) | ProtVista-type viewer | https://github.com/rcsb/rcsb-saguaro | — (repo created) | — | active | — | RCSB PDB's own 1D sequence-feature viewer (TypeScript); independent — does NOT use protvista or nightingale; the RCSB 1D tools paper (2020) cites the ProtVista paper | +| InterMine BlueGenes (`intermine/bluegenesProtVista`) | ProtVista-type viewer | https://github.com/intermine/bluegenesProtVista | 2018-08-17 (repo created) | — | dormant | 2020-07-10 | independent re-implementation | +| ProteomicsDB (`wilhelm-lab/protvista-proteomicsdb`) | ProtVista-type viewer | https://github.com/wilhelm-lab/protvista-proteomicsdb | 2021-09-18 (repo created) | — | dormant | 2023-08-30 | independent re-implementation | +| 3DBIONOTES (`3dbionotes-community/myProtVista`) | ProtVista-type viewer | https://github.com/3dbionotes-community/myProtVista | 2019-02-14 (repo created) | — | dormant | 2020-07-06 | appears ProtVista-derived; unverified | +| MolArt (`davidhoksza/protvista`) | fork | https://github.com/davidhoksza/protvista | 2017-08-30 (repo created) | — | — | — | fork of the original ProtVista (pre-`protvista-uniprot` rename); basis of the MolArt molecular-annotation tool. Run --backfill-dates for Since | +| ProKinO | ProtVista-type viewer | https://pubmed.ncbi.nlm.nih.gov/38077442/ | 2021 (paper) | — | — | — | no public repo; derivative of PDBe's protvista-pdb; npm `protvista-prokino` (last publish 2021) | +| InterPro (EBI) | ProtVista-type viewer | https://www.ebi.ac.uk/interpro/ | — (date unknown) | — | active | — | builds its own ProtVista-style protein feature viewer (nightingale-based) | +| ENACTdb | ProtVista-type viewer | https://www.iscbglab.in/enactdb/ | 2024 (paper) | — | active | — | ProtVista-style viewer (nightingale-based), live at iscbglab.in/enactdb; described in Bioinformatics Advances (vbae157) | +| ProteInfer (Google Research) | unclear | https://google-research.github.io/proteinfer/ | — (UI mention) | — | — | — | UI label mentions ProtVista; does not unambiguously demonstrate use of the library | diff --git a/docs/program/npm_downloads.csv b/docs/program/npm_downloads.csv new file mode 100644 index 00000000..f1e00bc4 --- /dev/null +++ b/docs/program/npm_downloads.csv @@ -0,0 +1,20 @@ +# npm downloads for protvista-uniprot. Raw HTTP fetch events — inflated by CI/mirror/bot traffic and dominated by UniProt's builds; a baseline trend, NOT an adoption count. +month,package,downloads +2025-01,protvista-uniprot,999 +2025-02,protvista-uniprot,780 +2025-03,protvista-uniprot,1149 +2025-04,protvista-uniprot,381 +2025-05,protvista-uniprot,767 +2025-06,protvista-uniprot,1683 +2025-07,protvista-uniprot,1601 +2025-08,protvista-uniprot,732 +2025-09,protvista-uniprot,1106 +2025-10,protvista-uniprot,2127 +2025-11,protvista-uniprot,1038 +2025-12,protvista-uniprot,2336 +2026-01,protvista-uniprot,1080 +2026-02,protvista-uniprot,1043 +2026-03,protvista-uniprot,1079 +2026-04,protvista-uniprot,1423 +2026-05,protvista-uniprot,1298 +2026-06,protvista-uniprot,1101 diff --git a/scripts/README.md b/scripts/README.md new file mode 100644 index 00000000..944f82c5 --- /dev/null +++ b/scripts/README.md @@ -0,0 +1,115 @@ +# ProtVista metrics & adoption tooling + +Small, dependency-light tools that maintain ProtVista's **adoption / usage metrics**. +Three deliberately independent artifacts — a curated *entity table*, an auto +*download series*, and an ad-hoc *citation number* — kept as separate files and read +individually, never merged into one. + +## Requirements + +| Need | For | Notes | +| --- | --- | --- | +| `python3` (≥ 3.8) | all | standard library only | +| `gh` (authenticated) | ecosystem discovery | `gh auth status`; GitHub code search + contents | +| network | npm + citations | npm downloads API, OpenAlex API | + +No `pip install` step. + +## The tools + +| Script | What it does | Run | +| --- | --- | --- | +| `update_metrics.sh` | runs all three of the below in sequence (continues if one fails) | `bash scripts/update_metrics.sh` | +| `protvista_ecosystem.py` | discover new `package.json` dependents → append rows to `ecosystem.md`, flag fell-out | `python3 scripts/protvista_ecosystem.py` | +| `npm_downloads.py` | update the committed monthly download CSV | `python3 scripts/npm_downloads.py` | +| `protvista_citation_count.py` | print the citing-works count (ad-hoc; writes nothing) | `python3 scripts/protvista_citation_count.py` | +| `test_ecosystem_tools.py` | unit tests (offline) | `python3 scripts/test_ecosystem_tools.py` | + +## 1. The ecosystem table — `../docs/program/ecosystem.md` + +A single hand/LLM-curated Markdown table: **this file is the source of truth**, and +its **git history** is the point-in-time record (no timestamped copies). Columns: +**Project / repo · Type · Evidence · Since · Until · Status · Last activity · Note**. + +- **Type** — `package consumer` | `ProtVista-type viewer` | `fork` | `commercial/private` | `unclear`. +- **Evidence** — the strongest public pointer (repo / paper / live site), or + `private — under agreement` where there is no public artifact. The Type says what + it is, the Evidence is a checkable link, the Note carries any caveat + (`verified: shares git history`, `appears derived; unverified`). +- **Since / Until** — when the project's ProtVista relationship began and (if it has) + ended. Each cell names the *kind* of date because it differs by type: `(dep added)` / + `(dep removed)` for package consumers, `(repo created)` or `(paper)` for forks & + viewers, `(interest)` for commercial, `(paper)` / `(UI mention)` for unclear. A `—` + in *Since* means not yet filled; a `—` in *Until* means ongoing / not applicable. + +`protvista_ecosystem.py` automates only the tedious part — finding GitHub package +consumers. It searches public `package.json` files, **verifies the exact dependency +key** (so look-alikes like `protvista-uniprot-entry-adapter` are rejected), and: + +- **appends** a stub row for any verified repo not already in the table (with + *Since* pre-filled from the repo's GitHub `created_at`); +- **prints** (does not edit) any package-consumer row whose repo no longer appears + in the search, so you can fill its *Until*. + +You then **review `git diff docs/program/ecosystem.md`**, refine Project / Type / +dates, and curate. Re-running with no real change makes no diff (dedup is on the +repo). Forks, viewers, deployments, commercial/private and unclear entries are added +by hand. `--backfill-dates` fills any blank (`—`) *Since* cell of a github-repo row +from its `created_at` (handy after seeding rows without dates). + +**Consent / no-PII:** never type an unconsented partner name into `ecosystem.md`. A +commercial row reads `Commercial adopter ()` until consent is recorded; +discovery only ever finds public package consumers, never commercial entries. + +## 2. npm downloads — `../docs/program/npm_downloads.csv` + +`npm_downloads.py` maintains a committed monthly time series for `protvista-uniprot` +since 2025 (`month,package,downloads`). It is an append-only cache: completed months +are written once and kept; each run only re-fetches the current month and the +previous one (to correct a partial→complete month) plus any months missing from the +file. **Caveat:** npm counts are raw fetch events — inflated by CI/mirror/bot traffic +and dominated by UniProt's own builds; report them as a baseline trend, never an +adoption headline (the caveat travels as a comment line in the CSV too). + +## 3. Citations — print-only, not saved + +`protvista_citation_count.py` prints the count of works citing the foundational +ProtVista paper (OpenAlex). It writes **nothing** to the repo — citations to a 2017 +paper move too slowly to track per period, so just read off the current number when +you need it. `--as-of YYYY-MM-DD` for an as-of count, `--list` for the per-work table, +`--mailto you@example.org` for OpenAlex's faster pool. + +## Refreshing the metrics + +**Shortcut — run all three at once:** `bash scripts/update_metrics.sh`, then review the +`ecosystem.md` diff, curate, and commit. Or step by step: + +```bash +# 1. Refresh discovered package consumers, then REVIEW and curate the diff. +python3 scripts/protvista_ecosystem.py # (or --dry-run to preview) +git diff docs/program/ecosystem.md # fill Project/Type/dates; mark removals +# Add any new fork / viewer / deployment / commercial entry by hand. + +# 2. Update the npm download series. +python3 scripts/npm_downloads.py + +# 3. Read off the current citation number (writes nothing). +python3 scripts/protvista_citation_count.py + +# 4. Commit the table + CSV. +git add docs/program/ecosystem.md docs/program/npm_downloads.csv +git commit -m "metrics: refresh" +``` + +Adoption figures move slowly — a stable period is a healthy, maintained baseline, not +a regression. + +## Notes + +- Hardcoded to `protvista-uniprot` (renamed to `protvista` in v5; the series will + eventually split across both names). +- Out of scope: consumers of the underlying `@nightingale-elements/*` track + components rather than ProtVista itself (e.g. InterPro). +- The two GA4 page-view scripts live in `protvista/documents/`, not here: they need + `pandas` + `google-analytics-data` and a separate project, and query UniProt's + private analytics. diff --git a/scripts/npm_downloads.py b/scripts/npm_downloads.py new file mode 100644 index 00000000..bd343098 --- /dev/null +++ b/scripts/npm_downloads.py @@ -0,0 +1,159 @@ +#!/usr/bin/env python3 +"""Maintain a committed monthly npm-downloads time series for protvista-uniprot. + +Writes/updates docs/program/npm_downloads.csv (columns: month,package,downloads), +one row per calendar month since 2025-01. The CSV is an append-only cache: completed +months are written once and kept; each run only re-fetches the current month and the +previous one (so a month first recorded while still in progress is corrected once it +completes) plus any months missing from the file. + +Data source: npm public downloads API (https://api.npmjs.org/downloads). No API key; +standard library only. + +------------------------------------------------------------------------------ +CAVEAT — read before quoting these numbers +------------------------------------------------------------------------------ +npm download counts are raw HTTP fetch events, NOT distinct users and NOT a measure +of adoption. They are inflated by CI/CD, Docker builds, npm mirrors and registry +replication, and are dominated by the primary consumer (UniProt's own builds). CDN +usage (unpkg/jsDelivr) and cached installs are NOT counted. Treat the series as a +BASELINE TREND to watch over time, never as a per-period adoption headline. +The package is being renamed protvista-uniprot -> protvista in v5, so the series +will eventually split across two names. +------------------------------------------------------------------------------ + +Usage: + python3 scripts/npm_downloads.py + python3 scripts/npm_downloads.py --since 2025-01 +""" + +import argparse +import calendar +import csv +import datetime as dt +import json +import os +import urllib.error +import urllib.parse +import urllib.request +from typing import Dict, List, Optional, Tuple + +PACKAGE = "protvista-uniprot" +DEFAULT_SINCE = "2025-01" +API = "https://api.npmjs.org/downloads/range" +_DOCS = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "docs", "program") +CSV_PATH = os.path.join(_DOCS, "npm_downloads.csv") +CAVEAT = ("# npm downloads for protvista-uniprot. Raw HTTP fetch events — inflated by CI/mirror/bot " + "traffic and dominated by UniProt's builds; a baseline trend, NOT an adoption count.") + + +def fetch_range(package: str, start: str, end: str) -> Optional[int]: + """Total downloads for `package` over [start, end] (inclusive), or None on no data.""" + pkg_path = urllib.parse.quote(package, safe="@") + url = f"{API}/{start}:{end}/{pkg_path}" + req = urllib.request.Request(url, headers={"User-Agent": "protvista-metrics"}) + try: + with urllib.request.urlopen(req, timeout=30) as resp: + data = json.load(resp) + except urllib.error.HTTPError as e: + if e.code == 404: + return None + raise + return sum(d["downloads"] for d in (data.get("downloads") or [])) + + +# --------------------------------------------------------------------------- # +# Pure month/CSV helpers (unit-tested, no network) +# --------------------------------------------------------------------------- # +def months(start_ym: str, today: dt.date) -> List[str]: + """'YYYY-MM' from start_ym up to and including today's month.""" + y, m = (int(x) for x in start_ym.split("-")) + out = [] + while (y, m) <= (today.year, today.month): + out.append(f"{y:04d}-{m:02d}") + y, m = (y + 1, 1) if m == 12 else (y, m + 1) + return out + + +def prev_ym(ym: str) -> str: + y, m = (int(x) for x in ym.split("-")) + return f"{y - 1:04d}-12" if m == 1 else f"{y:04d}-{m - 1:02d}" + + +def month_bounds(ym: str, today: dt.date) -> Tuple[str, str]: + """(start, end) ISO dates for month `ym`; end is clamped to today (partial month).""" + y, m = (int(x) for x in ym.split("-")) + last = dt.date(y, m, calendar.monthrange(y, m)[1]) + return f"{y:04d}-{m:02d}-01", min(last, today).isoformat() + + +def read_csv(path: str) -> Dict[str, int]: + out: Dict[str, int] = {} + if not os.path.exists(path): + return out + with open(path, newline="") as f: + for row in csv.reader(line for line in f if not line.startswith("#")): + if len(row) >= 3 and row[0] != "month": + try: + out[row[0]] = int(row[2]) + except ValueError: + pass + return out + + +def merge_months(wanted: List[str], existing: Dict[str, int], refresh: set, + fetch) -> Tuple[Dict[str, int], int]: + """Keep cached completed months; (re)fetch missing months + those in `refresh`. + + `fetch(ym) -> Optional[int]`. Returns (data, n_fetched).""" + data: Dict[str, int] = {} + fetched = 0 + for ym in wanted: + if ym in existing and ym not in refresh: + data[ym] = existing[ym] + continue + got = fetch(ym) + data[ym] = got if got is not None else existing.get(ym, 0) + fetched += 1 + return data, fetched + + +def write_csv(path: str, data: Dict[str, int]) -> None: + os.makedirs(os.path.dirname(path), exist_ok=True) + tmp = path + ".tmp" # write-then-rename = atomic + with open(tmp, "w", newline="") as f: + f.write(CAVEAT + "\n") + w = csv.writer(f) + w.writerow(["month", "package", "downloads"]) + for ym in sorted(data): + w.writerow([ym, PACKAGE, data[ym]]) + os.replace(tmp, path) + + +def main() -> None: + ap = argparse.ArgumentParser(description="Update the monthly npm-downloads CSV for protvista-uniprot.") + ap.add_argument("--since", default=DEFAULT_SINCE, help="first month, YYYY-MM (default 2025-01)") + args = ap.parse_args() + + today = dt.date.today() + cur = today.strftime("%Y-%m") + refresh = {cur, prev_ym(cur)} # correct partial->complete transitions + existing = read_csv(CSV_PATH) + + def fetch(ym: str) -> Optional[int]: + start, end = month_bounds(ym, today) + return fetch_range(PACKAGE, start, end) + + data, fetched = merge_months(months(args.since, today), existing, refresh, fetch) + write_csv(CSV_PATH, data) + + print(f"Saved {CSV_PATH}: {len(data)} months " + f"({fetched} fetched, {len(data) - fetched} cached).") + if data: + last = sorted(data)[-1] + print(f"Latest: {last} = {data[last]:,} (current month is partial).") + print("NB: baseline trend only — inflated by CI/mirror/bot traffic, UniProt-dominated.") + + +if __name__ == "__main__": + main() diff --git a/scripts/protvista_citation_count.py b/scripts/protvista_citation_count.py new file mode 100644 index 00000000..668dc88e --- /dev/null +++ b/scripts/protvista_citation_count.py @@ -0,0 +1,213 @@ +""" +Module to fetch paper citations from the OpenAlex API using standard libraries, +sort them by publication date (descending), and export them as a Markdown table. +""" + +import argparse +import datetime as dt +import json +import time +import urllib.error +import urllib.parse +import urllib.request +from typing import Any, Dict, List, Optional, TypedDict + + +class Citation(TypedDict): + """Type definition for a single citation record.""" + + title: str + authors: List[str] + date: str + journal: str + doi: str + + +OPENALEX_BASE_URL = "https://api.openalex.org/works" +REPO_URL = "https://github.com/ebi-webcomponents/protvista" + + +def _user_agent(mailto: Optional[str] = None) -> str: + """Identify the tool; a `mailto` opts into OpenAlex's faster 'polite pool'.""" + contact = f"; mailto:{mailto}" if mailto else "" + return f"protvista-citation-count (+{REPO_URL}{contact})" + + +def format_author_string(authors: List[str]) -> str: + """Formats a list of author names using academic conventions.""" + if not authors: + return "Unknown Author" + if len(authors) == 1: + return authors[0] + if len(authors) == 2: + return f"{authors[0]} & {authors[1]}" + if len(authors) > 6: + return f"{authors[0]} et al." + + return ", ".join(authors[:-1]) + f", & {authors[-1]}" + + +def get_all_citations( + doi: str, max_results: int = 1000, mailto: Optional[str] = None +) -> List[Citation]: + """ + Fetch citations using OpenAlex with cursor-based pagination. + + Args: + doi: The Digital Object Identifier of the target paper. + max_results: The maximum number of citations to retrieve. + + Returns: + A list of strongly typed dictionaries containing citation metadata. + """ + print(f"Step 1: Resolving DOI ({doi}) to an OpenAlex ID...\n") + ua = _user_agent(mailto) + + try: + resolve_url: str = f"{OPENALEX_BASE_URL}/doi:{doi}" + req = urllib.request.Request(resolve_url, headers={"User-Agent": ua}) + + with urllib.request.urlopen(req, timeout=10) as response: + work_data: Dict[str, Any] = json.loads(response.read().decode("utf-8")) + + work_id: str = work_data.get("id", "") + if not work_id: + print("Error: Could not find OpenAlex ID for this DOI.") + return [] + + short_id: str = work_id.split("/")[-1] + total_citations: Any = work_data.get("cited_by_count", "Unknown") + + print(f"Target paper has ~{total_citations} citations.") + print(f"Step 2: Fetching citations for ID {short_id} (Newest First)...\n") + + papers: List[Citation] = [] + cursor: Optional[str] = "*" + page_num: int = 1 + + while cursor and len(papers) < max_results: + fetch_size: int = min(200, max_results - len(papers)) + + # Updated select parameter to publication_date and added sort parameter + query_params: Dict[str, Any] = { + "filter": f"cites:{short_id}", + "select": "id,title,publication_date,authorships,doi,primary_location", + "sort": "publication_date:desc", + "per-page": fetch_size, + "cursor": cursor, + } + if mailto: + query_params["mailto"] = mailto # OpenAlex "polite pool" + + encoded_params: str = urllib.parse.urlencode(query_params) + cite_url: str = f"{OPENALEX_BASE_URL}?{encoded_params}" + cite_req = urllib.request.Request(cite_url, headers={"User-Agent": ua}) + + with urllib.request.urlopen(cite_req, timeout=10) as cite_response: + cite_data: Dict[str, Any] = json.loads( + cite_response.read().decode("utf-8") + ) + + results: List[Dict[str, Any]] = cite_data.get("results", []) + + if not results: + break + + for paper in results: + title: str = paper.get("title") or "No Title" + date_str: str = paper.get("publication_date") or "Unknown Date" + paper_doi: str = paper.get("doi") or "" + + raw_authors: List[Dict[str, Any]] = paper.get("authorships", []) + author_names: List[str] = [] + for author_data in raw_authors: + if "author" in author_data: + name = author_data["author"].get("display_name") + if name: + author_names.append(name) + + journal_name: str = "Unknown Journal/Venue" + location = paper.get("primary_location") + if location and isinstance(location, dict): + source = location.get("source") + if source and isinstance(source, dict): + journal_name = ( + source.get("display_name") or "Unknown Journal/Venue" + ) + + papers.append( + { + "title": title, + "authors": author_names, + "date": date_str, + "journal": journal_name, + "doi": paper_doi, + } + ) + + print( + f"Fetched page {page_num}: added {len(results)} records. " + f"(Total so far: {len(papers)})" + ) + + meta: Dict[str, Any] = cite_data.get("meta", {}) + cursor = meta.get("next_cursor") + + page_num += 1 + time.sleep(0.2) + + if len(papers) >= max_results: + print(f"[!] Reached the {max_results}-record cap; the count may be truncated.") + print("\n--- Extraction Complete ---\n") + return papers + + except urllib.error.HTTPError as e: + if e.code == 429: + print("HTTP Error 429: rate limited (try --mailto for OpenAlex's polite pool).") + else: + print(f"HTTP Error: {e.code} - {e.reason}") + except urllib.error.URLError as e: + print(f"URL Error: Failed to reach a server. Reason: {e.reason}") + except json.JSONDecodeError: + print("Error: The API did not return valid JSON.") + + return [] + + +if __name__ == "__main__": + _ap = argparse.ArgumentParser( + description="Print the citing-works count for the ProtVista paper (OpenAlex). " + "Ad-hoc tool — prints the count; nothing is written to the repo " + "(citations move too slowly to track per period).") + _ap.add_argument("--as-of", default=None, + help="count only works published on/before this date (YYYY-MM-DD); " + "default: include all to date") + _ap.add_argument("--mailto", default=None, + help="contact email for OpenAlex's faster 'polite pool'") + _ap.add_argument("--list", action="store_true", + help="also print the per-work table to stdout") + _args = _ap.parse_args() + + TARGET_DOI = "10.1093/bioinformatics/btx120" + papers: List[Citation] = get_all_citations(TARGET_DOI, max_results=500, mailto=_args.mailto) + + if _args.as_of: + def _on_or_before(date_str: str, cutoff: str) -> bool: + try: + dt.date.fromisoformat(date_str) + except ValueError: + return False # drop "Unknown Date" / malformed dates + return date_str <= cutoff + papers = [p for p in papers if _on_or_before(p["date"], _args.as_of)] + print(f"\n{len(papers)} citing works published on or before {_args.as_of}.") + else: + print(f"\n{len(papers)} citing works to date.") + + if _args.list: + print("\n| Date | Title | Authors | Journal/Venue | DOI |") + print("| :--- | :--- | :--- | :--- | :--- |") + for p in papers: + title = p["title"].replace("|", "|") + authors = format_author_string(p["authors"]).replace("|", "|") + journal = p["journal"].replace("|", "|") + print(f"| {p['date']} | {title} | {authors} | {journal} | {p['doi']} |") diff --git a/scripts/protvista_ecosystem.py b/scripts/protvista_ecosystem.py new file mode 100644 index 00000000..b71caaf7 --- /dev/null +++ b/scripts/protvista_ecosystem.py @@ -0,0 +1,323 @@ +#!/usr/bin/env python3 +""" +protvista_ecosystem.py — keep the curated ProtVista ecosystem table fresh. + +The source of truth is a single hand/LLM-curated Markdown table, +docs/program/ecosystem.md. This script does the one thing that benefits from +automation: periodically search public GitHub for repositories that declare +`protvista-uniprot` in a package.json, and: + + * APPEND a stub row for any repo that genuinely declares the dependency and is + not already in the table (verified by fetching the raw package.json — so + code-search look-alikes like `protvista-uniprot-entry-adapter` are rejected), + * PRINT (does not edit) any package-consumer row already in the table whose repo + no longer appears in the search, so you can mark it removed. + +You then review `git diff docs/program/ecosystem.md`, fill in Project / Type / +dates for the new rows, and curate. Re-running with no real-world change makes no +diff (dedup is on the repo). Everything else in the table — forks, ProtVista-type +viewers, deployments, commercial/private, citations — stays human-curated. + +Requirements: `gh` (authenticated; `gh auth status`) and Python 3 stdlib only. + +Usage: + python3 scripts/protvista_ecosystem.py # search, append new, flag fell-out + python3 scripts/protvista_ecosystem.py --dry-run # show what would change, write nothing +""" + +import argparse +import base64 +import binascii +import datetime as dt +import json +import os +import re +import subprocess +from typing import Dict, List, Optional, Set, Tuple + +PACKAGE = "protvista-uniprot" # what we search for +EXCLUDE = {"piwvh/dependabot-emse"} # known false positives (scraped datasets) +SEARCH_LIMIT = 100 # gh code-search caps at 100 +ACTIVE_DAYS = 365 # pushed within this many days -> "active" + +REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +ECOSYSTEM_MD = os.path.join(REPO_ROOT, "docs", "program", "ecosystem.md") +TABLE_HEADER = "| Project / repo |" + + +# --------------------------------------------------------------------------- # +# Shell + GitHub helpers +# --------------------------------------------------------------------------- # +def sh(cmd: List[str], timeout: int = 60) -> Tuple[int, str, str]: + try: + p = subprocess.run(cmd, capture_output=True, text=True, + errors="replace", timeout=timeout) + return p.returncode, p.stdout, p.stderr + except (OSError, subprocess.SubprocessError) as e: + return 1, "", str(e) + + +def search_candidates(pkg: str, limit: int) -> Tuple[List[Tuple[str, str]], int]: + """(sorted [(repo, path)], raw_hit_count) from GitHub code search (broad).""" + code, out, err = sh(["gh", "search", "code", f'"{pkg}"', "--filename", "package.json", + "--limit", str(limit), "--json", "repository,path"]) + if code != 0: + raise SystemExit(f"gh search code failed (is gh authenticated?):\n{err.strip()}") + items = json.loads(out or "[]") + pairs = sorted({(it["repository"]["nameWithOwner"], it["path"]) for it in items}) + return pairs, len(items) + + +def verify_key(repo: str, path: str, pkg: str) -> bool: + """True iff the repo's package.json at `path` declares `pkg` as an exact JSON key.""" + code, out, _ = sh(["gh", "api", f"repos/{repo}/contents/{path}", "--jq", ".content"]) + if code != 0 or not out.strip(): + return False + try: + content = base64.b64decode(out.replace("\n", "")).decode("utf-8", "replace") + except (binascii.Error, ValueError): + return False + return package_declared(content, pkg) + + +def repo_meta(repo: str) -> Dict[str, object]: + code, out, _ = sh(["gh", "api", f"repos/{repo}", "--jq", + "{archived:.archived, pushed_at:.pushed_at, created_at:.created_at}"]) + if code != 0: + return {} + try: + return json.loads(out) + except json.JSONDecodeError: + return {} + + +def repo_created(repo: str) -> Optional[str]: + """The repo's GitHub creation date (YYYY-MM-DD), or None.""" + code, out, _ = sh(["gh", "api", f"repos/{repo}", "--jq", ".created_at"]) + if code != 0: + return None + out = out.strip().strip('"') + return out[:10] if out and out != "null" else None + + +# --------------------------------------------------------------------------- # +# Pure helpers (unit-tested, no network) +# --------------------------------------------------------------------------- # +def package_declared(package_json_text: str, pkg: str) -> bool: + """Exact dependency-key match — excludes look-alikes (…-entry-adapter, …-x).""" + return re.search('"' + re.escape(pkg) + r'"\s*:', package_json_text) is not None + + +def github_repo(text: str) -> Optional[str]: + """owner/repo from a github.com URL in `text`, else None.""" + m = re.search(r"github\.com/([A-Za-z0-9._-]+/[A-Za-z0-9._-]+)", text) + if not m: + return None + repo = m.group(1).rstrip(".") + return repo[:-4] if repo.endswith(".git") else repo + + +def any_repo(text: str) -> Optional[str]: + """owner/repo from a github URL or a backtick `owner/repo` token, else None.""" + gh = github_repo(text) + if gh: + return gh + m = re.search(r"`([A-Za-z0-9._-]+/[A-Za-z0-9._-]+)`", text) + return m.group(1) if m else None + + +def parse_table(md: str) -> List[Dict[str, str]]: + """Rows of the ecosystem table as {project, type, evidence} (lowercased type).""" + rows, in_table = [], False + for line in md.splitlines(): + s = line.strip() + if s.startswith(TABLE_HEADER): + in_table = True + continue + if in_table: + if not s.startswith("|"): + break + if set(s) <= set("|-: "): # separator row + continue + cells = [c.strip() for c in s.strip("|").split("|")] + if len(cells) >= 3: + rows.append({"project": cells[0], "type": cells[1].lower(), + "evidence": cells[2]}) + return rows + + +def known_repos(rows: List[Dict[str, str]]) -> Set[str]: + """Every owner/repo mentioned anywhere in the table (lowercased) — for dedup.""" + out: Set[str] = set() + for r in rows: + for cell in (r["project"], r["evidence"]): + repo = any_repo(cell) + if repo: + out.add(repo.lower()) + return out + + +def tracked_dependents(rows: List[Dict[str, str]]) -> Set[str]: + """Repos we'd expect the GitHub search to re-find: package-consumer rows whose + Evidence is a github.com URL (i.e. previously discovered). Excludes CDN/live + deployments like GlyGen whose evidence is a site/paper.""" + out: Set[str] = set() + for r in rows: + if r["type"] == "package consumer": + repo = github_repo(r["evidence"]) + if repo: + out.add(repo.lower()) + return out + + +def _age_days(iso: Optional[str]) -> Optional[int]: + if not iso: + return None + try: + d = dt.datetime.fromisoformat(iso.replace("Z", "+00:00")) + except (ValueError, AttributeError): + return None + return (dt.datetime.now(dt.timezone.utc) - d).days + + +def build_row(repo: str, path: str, meta: Dict[str, object], today: str, + active_days: int = ACTIVE_DAYS) -> str: + """A ready-to-review Markdown table row for a newly discovered dependent. + + Since = the repo's GitHub creation date (run --backfill-dates later if unknown); + Until = — (ongoing); the human can refine to the exact dep-added date if wanted.""" + created = (str(meta.get("created_at") or ""))[:10] + since = f"{created} (repo created)" if created else "— (repo created)" + pushed = (str(meta.get("pushed_at") or ""))[:10] or "—" + if meta.get("archived"): + status = "archived" + else: + age = _age_days(str(meta.get("pushed_at") or "") or None) + status = "—" if age is None else ("active" if age <= active_days else "dormant") + return (f"| `{repo}` | package consumer | https://github.com/{repo} | {since} | — | " + f"{status} | {pushed} | auto-discovered {today} (`{path}`); verify |") + + +def insert_rows(md: str, new_rows: List[str]) -> str: + """Insert rows after the last existing table row (before any trailing prose).""" + if not new_rows: + return md + lines = md.splitlines() + header = next((i for i, l in enumerate(lines) if l.strip().startswith(TABLE_HEADER)), None) + if header is None: + raise SystemExit(f"Could not find the ecosystem table header in {ECOSYSTEM_MD}") + last = header + i = header + 1 + while i < len(lines) and lines[i].strip().startswith("|"): + last = i + i += 1 + return "\n".join(lines[:last + 1] + new_rows + lines[last + 1:]) + "\n" + + +def backfill_since(md: str, fetch_created) -> Tuple[str, List[str]]: + """Fill 'Since' cells that start with '—' for rows with a github repo. + + `fetch_created(repo) -> 'YYYY-MM-DD' | None`. Only fills (never overwrites a real + date), and only rows whose Since (4th column) begins with '—'. Returns + (new_md, [repos_filled]).""" + lines = md.splitlines() + header = next((i for i, l in enumerate(lines) if l.strip().startswith(TABLE_HEADER)), None) + if header is None: + return md, [] + filled: List[str] = [] + i = header + 1 + while i < len(lines) and lines[i].strip().startswith("|"): + s = lines[i].strip() + if not (set(s) <= set("|-: ")): # skip the separator row + cells = [c.strip() for c in s.strip("|").split("|")] + if len(cells) >= 5 and cells[3].startswith("—"): + repo = github_repo(cells[0]) or github_repo(cells[2]) + if repo: + created = fetch_created(repo) + if created: + cells[3] = f"{created} (repo created)" + lines[i] = "| " + " | ".join(cells) + " |" + filled.append(repo) + i += 1 + return "\n".join(lines) + "\n", filled + + +# --------------------------------------------------------------------------- # +# Main +# --------------------------------------------------------------------------- # +def main() -> None: + ap = argparse.ArgumentParser( + description="Discover new protvista-uniprot package.json dependents and append them " + "to docs/program/ecosystem.md (review via git diff).") + ap.add_argument("--limit", type=int, default=SEARCH_LIMIT, help="gh code-search limit (max 100)") + ap.add_argument("--dry-run", action="store_true", help="print changes; write nothing") + ap.add_argument("--backfill-dates", action="store_true", + help="fill empty 'Since' cells (rows starting with —, with a github repo) " + "from each repo's GitHub created_at, then exit") + args = ap.parse_args() + + with open(ECOSYSTEM_MD, encoding="utf-8") as f: + md = f.read() + + if args.backfill_dates: + new_md, filled = backfill_since(md, repo_created) + if filled and not args.dry_run: + tmp = ECOSYSTEM_MD + ".tmp" + with open(tmp, "w", encoding="utf-8") as f: + f.write(new_md) + os.replace(tmp, ECOSYSTEM_MD) + print(f"Filled 'Since' for {len(filled)} repo(s): {', '.join(filled)}.") + print("Review: git diff docs/program/ecosystem.md") + elif filled: + print(f"[dry-run] would fill 'Since' for {len(filled)}: {', '.join(filled)}") + else: + print("No rows needed backfilling.") + return + rows = parse_table(md) + known = known_repos(rows) + tracked = tracked_dependents(rows) + + print(f'[*] Searching GitHub for "{PACKAGE}" in package.json …') + hits, n_hits = search_candidates(PACKAGE, args.limit) + if n_hits >= args.limit: + print(f"[!] Hit the {args.limit}-result code-search cap; results (and fell-out detection) " + "may be incomplete.") + by_repo: Dict[str, List[str]] = {} + for repo, path in hits: + by_repo.setdefault(repo, []).append(path) + hit_repos = {r.lower() for r in by_repo} + + today = dt.date.today().isoformat() + new_rows: List[str] = [] + for repo in sorted(by_repo): + if repo in EXCLUDE or repo.lower() in known: + continue + path = next((p for p in by_repo[repo] if verify_key(repo, p, PACKAGE)), None) + if not path: + print(f" skip (no exact key / look-alike): {repo}") + continue + new_rows.append(build_row(repo, path, repo_meta(repo), today)) + print(f" + new dependent: {repo} ({path})") + + fell_out = sorted(r for r in tracked if r not in hit_repos) + if fell_out: + print("\n[!] In the table but no longer found by search (verify whether the dependency " + "was removed, then set Status):") + for r in fell_out: + print(f" - {r}") + + if new_rows and not args.dry_run: + tmp = ECOSYSTEM_MD + ".tmp" # write-then-rename = atomic + with open(tmp, "w", encoding="utf-8") as f: + f.write(insert_rows(md, new_rows)) + os.replace(tmp, ECOSYSTEM_MD) + print(f"\nAppended {len(new_rows)} row(s) to {ECOSYSTEM_MD}.") + print("Review and curate: git diff docs/program/ecosystem.md") + elif new_rows: + print(f"\n[dry-run] would append {len(new_rows)} row(s); nothing written.") + else: + print("\nNo new dependents to add.") + + +if __name__ == "__main__": + main() diff --git a/scripts/test_ecosystem_tools.py b/scripts/test_ecosystem_tools.py new file mode 100644 index 00000000..229a8afd --- /dev/null +++ b/scripts/test_ecosystem_tools.py @@ -0,0 +1,190 @@ +#!/usr/bin/env python3 +"""Unit tests for the simplified ProtVista ecosystem tooling. + +Run: python3 scripts/test_ecosystem_tools.py (or: python3 -m unittest, from this dir) + +No network required. +""" +import datetime as dt +import os +import sys +import tempfile +import unittest + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import protvista_ecosystem as eco # noqa: E402 +import npm_downloads as npm # noqa: E402 +import protvista_citation_count as cite # noqa: E402 + + +SAMPLE_MD = """# ProtVista adoption & ecosystem + +intro paragraph + +## Ecosystem entities + +| Project / repo | Type | Evidence | Since | Until | Status | Last activity | Note | +| --- | --- | --- | --- | --- | --- | --- | --- | +| Foo (`acme/foo`) | package consumer | https://github.com/acme/foo | 2024-01-01 (dep added) | — | active | 2026-01-01 | a note | +| GlyGen (`glygener/glygen-frontend`) | package consumer | https://www.glygen.org | 2021 (paper) | — | active | — | CDN deployment | +| Pharos (`ncats/protvista-viewer`) | fork | https://github.com/ncats/protvista-viewer | — (repo created) | — | dormant | 2024-08-01 | verified | +| ProKinO | ProtVista-type viewer | https://pubmed.ncbi.nlm.nih.gov/38077442/ | — (UI mention) | — | — | — | no public repo | + +_trailing prose, out of the table._ +""" + + +class TestTableParsing(unittest.TestCase): + def setUp(self): + self.rows = eco.parse_table(SAMPLE_MD) + + def test_row_count_and_cells(self): + self.assertEqual(len(self.rows), 4) + self.assertEqual(self.rows[0]["type"], "package consumer") + self.assertEqual(self.rows[2]["type"], "fork") + + def test_known_repos(self): + known = eco.known_repos(self.rows) + self.assertEqual(known, + {"acme/foo", "glygener/glygen-frontend", "ncats/protvista-viewer"}) + self.assertNotIn("prokino", known) # no repo for ProKinO + + def test_tracked_dependents_excludes_cdn_and_nonconsumers(self): + tracked = eco.tracked_dependents(self.rows) + self.assertEqual(tracked, {"acme/foo"}) # GlyGen (CDN evidence) + Pharos (fork) excluded + + +class TestRepoExtraction(unittest.TestCase): + def test_github_repo(self): + self.assertEqual(eco.github_repo("see https://github.com/a/b for more"), "a/b") + self.assertEqual(eco.github_repo("https://github.com/a/b.git"), "a/b") + self.assertIsNone(eco.github_repo("https://example.org/a/b")) + + def test_any_repo_backtick_fallback(self): + self.assertEqual(eco.any_repo("Foo (`acme/foo`)"), "acme/foo") + self.assertIsNone(eco.any_repo("ProKinO")) + + +class TestKeyVerification(unittest.TestCase): + def test_exact_key_accepted(self): + self.assertTrue(eco.package_declared( + '{"dependencies": {"protvista-uniprot": "^2.0.0"}}', "protvista-uniprot")) + + def test_lookalike_rejected(self): + self.assertFalse(eco.package_declared( + '{"dependencies": {"protvista-uniprot-entry-adapter": "^1.0.0"}}', "protvista-uniprot")) + self.assertFalse(eco.package_declared('{"name": "protvista-uniprot-x"}', "protvista-uniprot")) + + +class TestAppendAndDedup(unittest.TestCase): + def test_build_row_archived(self): + row = eco.build_row("new/repo", "package.json", + {"archived": True, "created_at": "2018-03-04T00:00:00Z", + "pushed_at": "2025-01-01T00:00:00Z"}, "2026-06-22") + self.assertTrue(row.startswith( + "| `new/repo` | package consumer | https://github.com/new/repo | " + "2018-03-04 (repo created) | — | archived |")) + self.assertIn("auto-discovered 2026-06-22 (`package.json`); verify", row) + + def test_build_row_unknown_created(self): + row = eco.build_row("new/repo", "package.json", {}, "2026-06-22") + self.assertIn("| — (repo created) | — |", row) # Since placeholder, Until ongoing + + def test_insert_before_trailing_prose_and_dedups(self): + row = eco.build_row("new/repo", "package.json", {}, "2026-06-22") + md2 = eco.insert_rows(SAMPLE_MD, [row]) + self.assertIn("new/repo", md2) + self.assertLess(md2.index("new/repo"), md2.index("trailing prose")) # inside the table + self.assertGreater(md2.index("new/repo"), md2.index("acme/foo")) # after existing rows + # a repo already present is now found by dedup on re-parse + self.assertIn("new/repo", eco.known_repos(eco.parse_table(md2))) + + +class TestFellOut(unittest.TestCase): + """Fell-out = tracked github dependents no longer in the search hit set.""" + def setUp(self): + self.tracked = eco.tracked_dependents(eco.parse_table(SAMPLE_MD)) # {acme/foo} + + def test_fell_out_when_absent(self): + hit_repos = set() # search returned nothing + self.assertEqual(sorted(r for r in self.tracked if r not in hit_repos), ["acme/foo"]) + + def test_not_fell_out_when_present(self): + hit_repos = {"acme/foo"} + self.assertEqual([r for r in self.tracked if r not in hit_repos], []) + + +class TestBackfill(unittest.TestCase): + """--backfill-dates fills only '—' Since cells of rows with a github repo.""" + def test_fills_dash_github_rows_only(self): + created = {"ncats/protvista-viewer": "2019-05-02"} + md2, filled = eco.backfill_since(SAMPLE_MD, lambda r: created.get(r)) + self.assertEqual(filled, ["ncats/protvista-viewer"]) + self.assertIn("2019-05-02 (repo created)", md2) # the '—' github row filled + self.assertIn("2024-01-01 (dep added)", md2) # a dated row is untouched + self.assertIn("— (UI mention)", md2) # a no-repo '—' row is skipped + + def test_no_created_date_leaves_row_unchanged(self): + md2, filled = eco.backfill_since(SAMPLE_MD, lambda r: None) + self.assertEqual(filled, []) + self.assertIn("— (repo created)", md2) + + +class TestNpmMonths(unittest.TestCase): + today = dt.date(2026, 6, 22) + + def test_months_range(self): + ms = npm.months("2025-01", self.today) + self.assertEqual(ms[0], "2025-01") + self.assertEqual(ms[-1], "2026-06") + self.assertEqual(len(ms), 18) + + def test_prev_ym_year_wrap(self): + self.assertEqual(npm.prev_ym("2026-01"), "2025-12") + self.assertEqual(npm.prev_ym("2026-06"), "2026-05") + + def test_month_bounds_clamped_to_today(self): + self.assertEqual(npm.month_bounds("2026-02", self.today), ("2026-02-01", "2026-02-28")) + self.assertEqual(npm.month_bounds("2026-06", self.today), ("2026-06-01", "2026-06-22")) + + +class TestNpmMerge(unittest.TestCase): + def test_caches_completed_refetches_current_prev_and_missing(self): + today = dt.date(2026, 6, 22) + wanted = npm.months("2025-01", today) + existing = {m: 100 for m in wanted if m not in ("2026-06", "2026-05", "2025-03")} + refresh = {"2026-06", npm.prev_ym("2026-06")} # current + previous + calls = [] + + def fetch(ym): + calls.append(ym) + return 999 + + data, fetched = npm.merge_months(wanted, existing, refresh, fetch) + self.assertEqual(sorted(calls), ["2025-03", "2026-05", "2026-06"]) # missing + refresh + self.assertEqual(fetched, 3) + self.assertEqual(data["2025-01"], 100) # completed month kept + self.assertEqual(data["2026-06"], 999) # current re-fetched + self.assertEqual(data["2025-03"], 999) # missing fetched + + def test_csv_roundtrip_skips_comment_and_header(self): + with tempfile.TemporaryDirectory() as d: + tmp = os.path.join(d, "npm_downloads.csv") + npm.write_csv(tmp, {"2025-01": 100, "2025-02": 150}) + self.assertEqual(npm.read_csv(tmp), {"2025-01": 100, "2025-02": 150}) + with open(tmp) as f: + first = f.readline() + self.assertTrue(first.startswith("#")) # caveat travels with the file + + +class TestCitationFormat(unittest.TestCase): + def test_author_formatting(self): + self.assertEqual(cite.format_author_string([]), "Unknown Author") + self.assertEqual(cite.format_author_string(["A"]), "A") + self.assertEqual(cite.format_author_string(["A", "B"]), "A & B") + self.assertEqual(cite.format_author_string(["A", "B", "C"]), "A, B, & C") + self.assertEqual(cite.format_author_string(list("ABCDEFG")), "A et al.") + + +if __name__ == "__main__": + unittest.main(verbosity=2) diff --git a/scripts/update_metrics.sh b/scripts/update_metrics.sh new file mode 100755 index 00000000..9f351759 --- /dev/null +++ b/scripts/update_metrics.sh @@ -0,0 +1,47 @@ +#!/usr/bin/env bash +# +# update_metrics.sh — run all the ProtVista metrics tools in one go. +# +# 1. protvista_ecosystem.py discover new package.json dependents -> ecosystem.md +# 2. npm_downloads.py refresh the monthly npm download CSV +# 3. protvista_citation_count.py print the current citation count (writes nothing) +# +# Run on a host with `gh` authenticated (`gh auth status`) and network access. +# Steps are independent: if one fails (e.g. gh not logged in), the rest still run. +# +# Usage: +# bash scripts/update_metrics.sh +# PYTHON=python3.12 bash scripts/update_metrics.sh # pick a specific interpreter +# +set -u + +DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +PY="${PYTHON:-python3}" +fail=0 + +run() { + echo + echo "===================================================================" + echo "==> $1" + echo "===================================================================" + if ! "$PY" "$DIR/$1"; then + echo "!! $1 failed (continuing with the rest)" + fail=1 + fi +} + +echo "ProtVista metrics refresh — $("$PY" --version 2>&1)" +run protvista_ecosystem.py +run npm_downloads.py +run protvista_citation_count.py + +echo +echo "-------------------------------------------------------------------" +echo "Done. Next:" +echo " • review & curate any new ecosystem rows:" +echo " git diff docs/program/ecosystem.md" +echo " • commit the table + CSV:" +echo " git add docs/program/ecosystem.md docs/program/npm_downloads.csv && git commit" +echo " • note the citation count printed above (it is not written to the repo)." +[ "$fail" -eq 0 ] || echo " (some steps failed — see the output above)" +exit "$fail" diff --git a/src/protvista-uniprot-datatable.ts b/src/protvista-uniprot-datatable.ts index 028c569a..c792ea04 100644 --- a/src/protvista-uniprot-datatable.ts +++ b/src/protvista-uniprot-datatable.ts @@ -530,3 +530,5 @@ declare global { >; } } + +export default ProtvistaUniprotDatatable; diff --git a/src/protvista-uniprot-structure.ts b/src/protvista-uniprot-structure.ts index 27b33580..31d63b75 100644 --- a/src/protvista-uniprot-structure.ts +++ b/src/protvista-uniprot-structure.ts @@ -1,11 +1,20 @@ -import { LitElement, html, svg, type TemplateResult, css, nothing } from 'lit'; +import { + LitElement, + html, + svg, + type TemplateResult, + type PropertyValues, + css, + nothing, +} from 'lit'; import { customElement, state } from 'lit/decorators.js'; import { unsafeHTML } from 'lit/directives/unsafe-html.js'; import NightingaleStructure, { type AlphaFoldPayload, } from '@nightingale-elements/nightingale-structure'; -import type { ColumnConfig } from './protvista-uniprot-datatable.js'; -import './protvista-uniprot-datatable.js'; +import ProtvistaUniprotDatatable, { + type ColumnConfig, +} from './protvista-uniprot-datatable.js'; import { fetchAll, loadComponent } from './utils/index.js'; import downloadIcon from './icons/download.svg'; import externalLinkIcon from './icons/external-link.svg'; @@ -15,29 +24,10 @@ import { inlineSvg } from './icons/inline.js'; import loaderStyles from './styles/loader-styles.js'; import { injectStyleOnce, installTokenDefaults } from './styles/inject.js'; -const PDBLinks = [ - { name: 'PDBe', link: 'https://www.ebi.ac.uk/pdbe-srv/view/entry/' }, - { name: 'RCSB-PDB', link: 'https://www.rcsb.org/structure/' }, - { name: 'PDBj', link: 'https://pdbj.org/mine/summary/' }, - { name: 'PDBsum', link: 'https://www.ebi.ac.uk/pdbsum/' }, -]; -const alphaFoldUrl = 'https://alphafold.ebi.ac.uk/entry/'; +const alphaFoldLinkUrl = 'https://alphafold.ebi.ac.uk/search/text/'; const foldseekUrl = `https://search.foldseek.com/search`; const uniprotKBUrl = 'https://www.uniprot.org/uniprotkb/'; -// Excluded source from 3d-beacons is PDBe as we fetch them separately from UniProt -const providersFrom3DBeacons = [ - 'AlphaFold DB', - 'SWISS-MODEL', - 'ModelArchive', - 'PED', - 'SASBDB', - 'isoform.io', - 'AlphaFill', - 'HEGELAB', - 'levylab', -]; - const sourceMethods = new Map([ ['AlphaFold DB', 'Predicted'], ['SWISS-MODEL', 'Modeling'], @@ -50,6 +40,18 @@ const sourceMethods = new Map([ ['levylab', 'Modeling'], ]); +// Excluded source from 3d-beacons is PDBe as we fetch them separately from UniProt +const providersFrom3DBeacons = [ + 'SWISS-MODEL', + 'ModelArchive', + 'PED', + 'SASBDB', + 'isoform.io', + 'AlphaFill', + 'HEGELAB', + 'levylab', +]; + type UniProtKBData = { uniProtKBCrossReferences: UniProtKBCrossReference[]; sequence: Sequence; @@ -119,7 +121,7 @@ type BeaconsData = { }[]; }; -type ProcessedStructureData = { +export type ProcessedStructureData = { id: string; source: string; method?: string; @@ -130,16 +132,15 @@ type ProcessedStructureData = { sourceDBLink?: string; protvistaFeatureId: string; amAnnotationsUrl?: string; - isoform?: TemplateResult; - afPrediction?: boolean; // Flag to differentiate the structure source as AlphaFold prediction API vs 3DBeacons AlphaFold + isoformId?: string; + isoformIsCanonical?: boolean; + oligomericState?: string; }; -type IsoformIdSequence = [ - { - isoformId: string; - sequence: string; - }, -]; +type IsoformIdSequence = Array<{ + isoformId: string; + sequence: string; +}>; const getIsoformNum = (s: string) => { const match = s.match(/-(\d+)-F1$/); @@ -183,7 +184,7 @@ const processPDBData = (data: UniProtKBData): ProcessedStructureData[] => method, resolution: !resolution || resolution === '-' ? undefined : resolution, - downloadUrl: `https://www.ebi.ac.uk/pdbe/entry-files/download/pdb${id.toLowerCase()}.ent`, + downloadUrl: `https://www.ebi.ac.uk/pdbe/entry-files/download/${id.toLowerCase()}_updated.cif`, chain, positions, protvistaFeatureId: id, @@ -195,38 +196,56 @@ const processPDBData = (data: UniProtKBData): ProcessedStructureData[] => const processAFData = ( data: AlphaFoldPayload, - accession?: string, isoforms?: IsoformIdSequence, canonicalSequence?: string -): ProcessedStructureData[] => - data +): ProcessedStructureData[] => { + const uniqueData = [ + ...new Map(data.map((d) => [d.modelEntityId, d])).values(), + ]; + + return uniqueData .map((d) => { const isoformMatch = isoforms?.find( ({ sequence }) => d.sequence === sequence ); - const isoformElement = isoformMatch - ? html` - ${isoformMatch.isoformId} - ${isoformMatch.sequence === canonicalSequence ? '(Canonical)' : ''} - ` - : null; - + let chain = d.chainId; + const oligomericState = d.isComplex + ? `${d.assemblyType}${d.oligomericState}` + : 'Monomer'; + + if (d.isComplex && oligomericState === 'Homodimer') { + chain = data + .filter(({ modelEntityId }) => modelEntityId === d.modelEntityId) + .flatMap(({ chainId }) => chainId) + .sort() + .join(', '); + } return { id: d.modelEntityId, source: 'AlphaFold DB', - method: 'Predicted', positions: `${d.sequenceStart}-${d.sequenceEnd}`, protvistaFeatureId: d.modelEntityId, - downloadUrl: d.pdbUrl, + downloadUrl: d.cifUrl, amAnnotationsUrl: d.amAnnotationsUrl, - isoform: isoformElement, + isoformId: !d.isComplex ? isoformMatch?.isoformId : undefined, + isoformIsCanonical: + !d.isComplex && isoformMatch + ? isoformMatch.sequence === canonicalSequence + : undefined, afPrediction: true, + oligomericState, + chain, + method: sourceMethods.get('AlphaFold DB') || undefined, }; }) - .sort((a, b) => getIsoformNum(a.id) - getIsoformNum(b.id)); + .sort((a, b) => getIsoformNum(a.id) - getIsoformNum(b.id)) + .sort((a, b) => { + const aMonomer = a.oligomericState === 'Monomer' ? 0 : 1; + const bMonomer = b.oligomericState === 'Monomer' ? 0 : 1; + return aMonomer - bMonomer; + }); +}; const process3DBeaconsData = ( data: BeaconsData, @@ -264,7 +283,6 @@ const process3DBeaconsData = ( structures?.map(({ summary }) => ({ id: summary.model_identifier, source: summary.provider, - method: sourceMethods.get(summary.provider), positions: summary.uniprot_start && summary.uniprot_end ? `${summary.uniprot_start}-${summary.uniprot_end}` @@ -276,8 +294,13 @@ const process3DBeaconsData = ( ? 'https://www.isoform.io/home' : summary.model_page_url, chain: - summary.entities?.flatMap((entity) => entity.chain_ids).join(', ') || - undefined, + summary.entities + ?.flatMap((entity) => + entity.identifier_category === 'UNIPROT' ? entity.chain_ids : [] + ) + .join(', ') || undefined, + oligomericState: summary.oligomeric_state || undefined, + method: sourceMethods.get(summary.provider) || undefined, })) || [] ); }; @@ -389,18 +412,21 @@ class ProtvistaUniprotStructure extends LitElement { private modelUrl = ''; private columns: ColumnConfig[] = []; - private selectedRowId?: string; + selectedId?: string; + noTable?: boolean; constructor() { super(); loadComponent('nightingale-structure', NightingaleStructure); + // Registered unconditionally (even when no-table is set) so the import is + // preserved through tree-shaking; the element only renders when noTable is + // false, so an unused registration is cheap. + loadComponent('protvista-uniprot-datatable', ProtvistaUniprotDatatable); this.loading = true; this.addStyles(); this.colorTheme = 'alphafold'; this.alphamissenseAvailable = false; - - this.columns = this.getColumns(); } static get properties() { @@ -414,6 +440,8 @@ class ProtvistaUniprotStructure extends LitElement { colorTheme: { type: String }, alphamissenseAvailable: { type: Boolean }, isoforms: { type: Object, attribute: false }, + selectedId: { type: String, attribute: 'selected-id' }, + noTable: { type: Boolean, attribute: 'no-table' }, }; } @@ -431,8 +459,15 @@ class ProtvistaUniprotStructure extends LitElement { if (this.isoforms) { cols.push({ label: 'Isoform', - key: 'isoform', - render: (row) => row.isoform ?? nothing, + key: 'isoformId', + render: (row) => + row.isoformId + ? html` + ${row.isoformId}${row.isoformIsCanonical ? ' (Canonical)' : ''} + ` + : nothing, }); } @@ -466,15 +501,10 @@ class ProtvistaUniprotStructure extends LitElement { return html` ${source === 'PDB' - ? html` - ${PDBLinks.map( - (pdbLink) => - html`${pdbLink.name}` - ).reduce((prev, curr) => html`${prev} · ${curr}`)} - ` + ? html` PDBe ` : nothing} ${source === 'AlphaFold DB' && this.accession - ? html`AlphaFold` + ? html`AlphaFold` : nothing} ${sourceDBLink ? html`${source}` : nothing} `; @@ -506,7 +536,7 @@ class ProtvistaUniprotStructure extends LitElement { // AlphaMissense predictions are only available in AF predictions endpoint const alphaFoldUrl = this.accession && !this.checksum - ? `https://alphafold.ebi.ac.uk/api/prediction/${this.accession}` + ? `https://alphafold.ebi.ac.uk/api/prediction/${this.accession}?include_complexes=true` : ''; // exclude_provider accepts only value hence 'pdbe' as majority of the models are from there if querying by accession const beaconsUrl = @@ -523,13 +553,12 @@ class ProtvistaUniprotStructure extends LitElement { if (this.isoforms && rawData[alphaFoldUrl]?.length) { // Include isoforms that are provided in the UniProt isoforms mapping and ignore the rest from AF payload that are out of sync with UniProt const alphaFoldSequenceMatches = rawData[alphaFoldUrl]?.filter( - ({ sequence: afSequence }) => + ({ sequence: afSequence }: { sequence: string }) => this.isoforms?.some(({ sequence }) => afSequence === sequence) ); afData = processAFData( alphaFoldSequenceMatches, - this.accession, this.isoforms, rawData[pdbUrl]?.sequence?.value ); @@ -538,14 +567,14 @@ class ProtvistaUniprotStructure extends LitElement { } else { // Check if AF sequence matches UniProt sequence const alphaFoldSequenceMatch = rawData[alphaFoldUrl]?.filter( - ({ sequence: afSequence }) => + ({ sequence: afSequence }: { sequence: string }) => rawData[pdbUrl]?.sequence?.value === afSequence || this.sequence === afSequence ); if (alphaFoldSequenceMatch?.length) { afData = processAFData(alphaFoldSequenceMatch); this.alphamissenseAvailable = alphaFoldSequenceMatch.some( - (data) => data.amAnnotationsUrl + (data: { amAnnotationsUrl?: string }) => data.amAnnotationsUrl ); } } @@ -559,30 +588,26 @@ class ProtvistaUniprotStructure extends LitElement { // TODO: return if no data at all // if (!payload) return; - const beaconsAFData = beaconsData.filter( - ({ source }) => source === 'AlphaFold DB' - ); - const beaconsNonAFData = beaconsData.filter( - ({ source }) => source !== 'AlphaFold DB' - ); - - const uniqueAFData = [ - ...new Map( - // The order of the spread is important as we want to prioritise AF data from AF predictions API over 3DBeacons - [...beaconsAFData, ...afData].map((obj) => [obj.id, obj]) - ).values(), - ]; - - const data = [...pdbData, ...uniqueAFData, ...beaconsNonAFData]; - - if (!data || !data.length) return; + const data = [...pdbData, ...afData, ...beaconsData]; this.data = data; this.columns = this.getColumns(); - // Select first row by default - this.selectedRowId = data[0].id; - this.onRowSelected(data[0]); + // Default to the first row only if the consumer hasn't pre-set a selection. + if (data.length > 0 && !this.selectedId) { + this.selectedId = data[0].id; + } + + this.dispatchEvent( + new CustomEvent>( + 'structures-loaded', + { + detail: data, + bubbles: true, + composed: true, + } + ) + ); } addStyles() { @@ -597,21 +622,51 @@ class ProtvistaUniprotStructure extends LitElement { injectStyleOnce(styleId, ProtvistaUniprotStructure.cssStyle.toString()); } - private onRowSelected(row: ProcessedStructureData) { - const { id, source, downloadUrl, amAnnotationsUrl, afPrediction } = row; - this.selectedRowId = id; + protected override willUpdate(changed: PropertyValues) { + // Apply when either selection or data changes — covers consumer + // pre-setting selected-id before the async fetch resolves. + if (!changed.has('selectedId') && !changed.has('data')) return; + // Wait for the first data assignment before deciding what to show; a + // consumer-set selectedId arriving before fetch should not clear anything. + if (!this.data) return; + + const row = this.selectedId + ? this.data.find((r) => r.id === this.selectedId) + : undefined; + if (row) { + this.applySelection(row); + } else { + // No match (empty data, stale selectedId, or selection cleared) — drop + // the viewer state so a previous structure does not linger on screen. + this.clearViewer(); + } + } + + private clearViewer() { + // No-op when nothing was ever applied — avoids handing Mol* an undefined + // ref on the initial-load empty-data path. + if (!this.structureId && !this.modelUrl) return; + this.structureId = undefined; + this.modelUrl = ''; + this.metaInfo = undefined; + } + + private applySelection(row: ProcessedStructureData) { + const { id, source, downloadUrl, amAnnotationsUrl, oligomericState } = row; if ( this.checksum || - (providersFrom3DBeacons.includes(source) && !afPrediction) + providersFrom3DBeacons.includes(source) || + (source === 'AlphaFold DB' && oligomericState !== 'Monomer') ) { - this.modelUrl = downloadUrl; + this.modelUrl = downloadUrl ?? ''; // Reset the rest this.structureId = undefined; this.metaInfo = undefined; this.colorTheme = 'alphafold'; if (source === 'AlphaFold DB') { this.metaInfo = AFMetaInfo; + this.alphamissenseAvailable = !!amAnnotationsUrl; } } else { this.structureId = id; @@ -626,7 +681,7 @@ class ProtvistaUniprotStructure extends LitElement { } private onDatatableRowClick = (e: CustomEvent) => { - this.onRowSelected(e.detail); + this.selectedId = e.detail.id; }; // Built once at class definition rather than per instance: addStyles() @@ -782,30 +837,32 @@ class ProtvistaUniprotStructure extends LitElement { : nothing} -
- ${this.data && this.data.length - ? html` - - ` - : nothing} - ${this.loading - ? html`
- ${svg`${unsafeHTML(inlineSvg(loaderIcon))}`} -
` - : nothing} - ${!this.data && !this.loading - ? html`
- No structure information available - ${this.accession ? `for ${this.accession}` : ''} -
` - : nothing} -
+ ${this.noTable + ? nothing + : html`
+ ${this.data && this.data.length + ? html` + + ` + : nothing} + ${this.loading + ? html`
+ ${svg`${unsafeHTML(inlineSvg(loaderIcon))}`} +
` + : nothing} + ${(!this.data || this.data.length === 0) && !this.loading + ? html`
+ No structure information available + ${this.accession ? `for ${this.accession}` : ''} +
` + : nothing} +
`} `; }