diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 636318eec..32014d8cd 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -2,7 +2,7 @@ "name": "doiget", "metadata": { "description": "The doiget marketplace — one plugin: an OA-first paper fetcher for DOIs and arXiv IDs.", - "version": "0.8.11" + "version": "0.8.12" }, "owner": { "name": "QAtlasHub", diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index a4de95f81..4f90c228d 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "doiget", "description": "Turn DOIs and arXiv IDs into local PDFs and structured metadata via official Open-Access APIs. Never bypasses paywalls.", - "version": "0.8.11", + "version": "0.8.12", "author": { "name": "Sota Shimozono", "url": "https://github.com/QAtlasHub/doiget" diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 52453579f..edcefd2e9 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -89,6 +89,29 @@ jobs: # so test and clippy exercise an identical feature surface. - run: cargo test --workspace --all-targets --no-default-features --features oa-only + # The five jobs above skip `#[ignore]`d tests, which is where the one + # 10-minute test lives. It is a dispatch-loop property that varies with + # neither OS nor Cargo feature, so running it in all five spent ~50 minutes + # per CI run learning the same thing five times. Once is enough. + # + # NOT a sixth leg of the `test` matrix: it has to run in PARALLEL with the + # others rather than lengthen one of them, and it needs no OS spread. + test-slow: + name: test (slow) + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 + - uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 + with: + toolchain: stable + - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 + with: + cache-bin: false + # `--ignored` runs ONLY the ignored tests, so this job is exactly the + # slow set and the five broad jobs are exactly everything else. Nothing + # is dropped by the split and nothing is run twice. + - run: cargo test --workspace --all-targets --no-default-features --features oa-only -- --ignored + test-citation: name: test (citation feature) runs-on: ubuntu-latest diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 519af8b48..854909ba2 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -42,14 +42,14 @@ jobs: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Initialize CodeQL - uses: github/codeql-action/init@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8 + uses: github/codeql-action/init@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 with: languages: ${{ matrix.language }} - name: Autobuild - uses: github/codeql-action/autobuild@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8 + uses: github/codeql-action/autobuild@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 - name: Perform CodeQL analysis - uses: github/codeql-action/analyze@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4.37.8 + uses: github/codeql-action/analyze@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9 with: category: "/language:${{ matrix.language }}" diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index cc978af8a..d4ced98f7 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -35,7 +35,7 @@ jobs: with: components: llvm-tools-preview - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 - - uses: taiki-e/install-action@ba47c86ac325773530516bb756137ac718732518 # v2.86.5 + - uses: taiki-e/install-action@37f7c5781271959fb65b6b35224e28652ff2b63d # v2.87.0 with: tool: cargo-llvm-cov - name: Generate workspace coverage (lcov) diff --git a/.github/workflows/posture-lint.yml b/.github/workflows/posture-lint.yml index 92f197750..a22ef6e14 100644 --- a/.github/workflows/posture-lint.yml +++ b/.github/workflows/posture-lint.yml @@ -119,6 +119,45 @@ jobs: fi echo "LICENSE OK: verbatim SPDX MIT, no trailing prose, ends at SOFTWARE., LF." + - name: MCP_TOOLS.md lists exactly the tools the router exposes (#553) + shell: bash + run: | + set -euo pipefail + # `MCP_TOOLS.md` is what an integrator reads to decide what the server + # can do, and `docs/INTEGRATION/*.md` points at it. Nothing checked it + # against the router, and it had drifted in BOTH directions at once: + # four tools shipped with no entry (#553), and one had an entry while + # not shipping at all (#552). Neither is visible from inside the other + # file. + # + # rmcp derives a tool's name from its `#[tool]` method name, so the + # method list is the router's own truth. Compared against the table + # rows, which are the document's. + fns="$(grep -oE '^[[:space:]]+(pub )?async fn doiget_[a-z_0-9]+' \ + crates/doiget-mcp/src/lib.rs \ + | grep -oE 'doiget_[a-z_0-9]+' | sort -u)" + rows="$(grep -oE '^\| .doiget_[a-z_0-9]+.' docs/MCP_TOOLS.md \ + | grep -oE 'doiget_[a-z_0-9]+' | sort -u)" + if [ -z "$fns" ] || [ -z "$rows" ]; then + echo "::error::posture-lint mcp-tools: read no tools from the router or no rows from MCP_TOOLS.md — one of the two patterns has stopped matching (#553)" + exit 1 + fi + missing="$(comm -23 <(printf '%s\n' "$fns") <(printf '%s\n' "$rows"))" + extra="$(comm -13 <(printf '%s\n' "$fns") <(printf '%s\n' "$rows"))" + fail=0 + if [ -n "$missing" ]; then + echo "::error::posture-lint mcp-tools: these tools ship but have no row in docs/MCP_TOOLS.md (#553)" + printf ' %s\n' $missing + fail=1 + fi + if [ -n "$extra" ]; then + echo "::error::posture-lint mcp-tools: docs/MCP_TOOLS.md documents these, but the router has no such tool (#552)" + printf ' %s\n' $extra + fail=1 + fi + [ "$fail" -eq 0 ] + echo "MCP_TOOLS.md and the router agree on $(printf '%s\n' "$fns" | wc -l | tr -d ' ') tools" + - name: doiget-cli forwards every doiget-mcp feature (#373 / #516) shell: bash run: | @@ -244,6 +283,82 @@ jobs: shell: bash run: bash npm/doiget-cli/test/release-checksums.test.sh + - name: Homebrew formula is generator output (#501) + shell: bash + # `Formula/doiget.rb` is generated from a release's published `.sha256` + # assets. #247 was closed as completed while four fifths of it had not + # shipped, so the formula is asserted rather than trusted. + # + # What this fails on: a hand-edit, a MALFORMED checksum, a `v`-prefixed + # version. What it CANNOT fail on, because it re-reads the version and + # the checksums from the very file under test: a well-formed checksum + # carrying the wrong value, or a version that is simply stale. It said + # it caught "a bad checksum" and it does not -- nothing here reaches the + # published `.sha256` assets, offline being the point. The stale-version + # half is covered by the release-sync check below; a wrong-value + # checksum is caught by `brew install` and by nothing before it. + run: bash scripts/update-homebrew-formula.test.sh + + - name: release-tracking files name one version (#501, #511) + shell: bash + # Four files record "the last published stable", and a stable release + # bumps all four in one maintainer step. Nothing used to hold them + # together, which is how `.claude-plugin/*` sat at 0.8.11 while 0.8.12 + # was the shipped release. + # + # What this CANNOT catch, stated so the check does not repeat the + # overclaim it exists to fix: forgetting the step ENTIRELY leaves all + # four at the previous release, mutually consistent, and green. + # Nothing in the repository knows which version is currently published. + # What it does catch is bumping some and not others, and pinning a + # version that was never released -- CHANGELOG.md is the only + # independent source of truth available offline. + run: | + set -euo pipefail + # `|| true` on each is load-bearing. Under `set -euo pipefail` a bare + # assignment takes the exit status of its command substitution, so a + # `grep` that matches nothing -- somebody deletes the `@version` pin from + # .mcp.json, or reformats a manifest -- kills the step HERE, before the + # `::error::` written to explain it. It failed closed, which is right, and + # said nothing, which is the defect this whole job exists to catch. Found + # by review of this change; the same shape was fixed one file over in + # npm/doiget-cli/test/stage-npm.test.sh. + formula=$(grep -oE '^ version "[^"]+"' Formula/doiget.rb | grep -oE '[0-9][^"]*' || true) + plugin=$(grep -oE '"version": *"[^"]+"' .claude-plugin/plugin.json | grep -oE '[0-9][^"]*' || true) + market=$(grep -oE '"version": *"[^"]+"' .claude-plugin/marketplace.json | grep -oE '[0-9][^"]*' || true) + pinned=$(grep -oE '"doiget-cli@[^"]+"' .mcp.json | grep -oE '[0-9][^"]*' || true) + for pair in "Formula/doiget.rb:$formula" ".claude-plugin/plugin.json:$plugin" ".claude-plugin/marketplace.json:$market" ".mcp.json:$pinned"; do + if [ -z "${pair#*:}" ]; then + echo "::error::posture-lint release-sync: no version found in ${pair%%:*} -- the file was reformatted, or .mcp.json lost its doiget-cli@ pin" + exit 1 + fi + done + echo "formula=$formula plugin=$plugin marketplace=$market .mcp.json=$pinned" + for v in "$plugin" "$market" "$pinned"; do + if [ "$v" != "$formula" ]; then + echo "::error::posture-lint release-sync: release-tracking files disagree. Bump Formula/doiget.rb, .claude-plugin/plugin.json, .claude-plugin/marketplace.json and .mcp.json together (CONTRIBUTING.md, 'one step after a stable release')" + exit 1 + fi + done + # An unpublished pin would send every plugin user to a 404. Any `-` + # suffix is a prerelease per SemVer; the first version of this listed + # spellings (`*-beta.*|*-rc.*`) and let `0.9.0-alpha.2` straight + # through, which is the enumerate-the-known-cases mistake this + # release is otherwise about. + case "$pinned" in + *-*) + echo "::error::posture-lint release-sync: .mcp.json pins a prerelease ($pinned); the plugin must point at a published stable" + exit 1 + ;; + esac + # The one independent check available without a network: a version + # these files claim to track must have a released CHANGELOG section. + if ! grep -qE "^## \[$formula\]" CHANGELOG.md; then + echo "::error::posture-lint release-sync: the tracking files name $formula, which has no '## [$formula]' section in CHANGELOG.md -- they point at a version that was never released" + exit 1 + fi + echo "release-tracking files agree on $formula, and CHANGELOG.md has its section" + - name: npm packaging tests (#511) shell: bash # The name-list greps above cannot catch a wrong binary name, a @@ -266,6 +381,57 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Every hand-built error object carries a disposition (#506) + shell: bash + run: | + set -euo pipefail + # `error_object` builds most failure envelopes, but five are assembled + # field-by-field with `insert("code", ...)` because they also attach a + # `denial_context`. The first pass at #506 converted the former and + # missed the latter -- including the two most common failures an agent + # sees -- so `disposition` shipped present on some failures and absent + # on others, which is worse than absent everywhere. + # + # One `insert("disposition"` per `insert("code"`. Not a proof, but it + # fails on exactly the mistake that was made. + f=crates/doiget-mcp/src/lib.rs + # `|| true`: `grep -c` exits 1 on a zero count, and under + # `set -euo pipefail` that kills the step at the assignment, before + # the `::error::` written below to explain it. Same guard as the + # release-sync step and `capture()`. + codes="$(grep -c 'insert("code"' "$f" || true)" + disps="$(grep -c 'insert("disposition"' "$f" || true)" + if [ "$codes" != "$disps" ]; then + echo "::error::posture-lint: $codes hand-built error objects but $disps dispositions in $f - a failure envelope is missing error.disposition (#506)" + grep -n 'insert("code"' "$f" + exit 1 + fi + + - name: Host adjudication uses `permits`, never `matches` (#533) + shell: bash + run: | + set -euo pipefail + # `SourceAllowlist::matches` answers "is this host on the list". + # `SourceAllowlist::permits` answers "may we go here", which is what + # every gate is actually asking -- and it is the one that treats a + # DOI resolver as addressing rather than as a content host. + # + # #533 was a gate that asked the first question and acted on the + # answer: a gold-OA cc-by paper was refused at `doi.org`, one hop + # before the publisher whose host was already allowlisted. There + # were FIVE such gates and only one of them was ever walked end to + # end, which is #462's point exactly ("every 'unreachable source' + # bug passed its unit tests"). + # + # The discriminator: adjudication passes a host VARIABLE + # (`.matches(&host)`); list-membership assertions in tests pass a + # LITERAL (`.matches("doaj.org")`). So a `.matches(&` anywhere is a + # gate that should be a `permits`. + if grep -rn --include='*.rs' '\.matches(&' crates/ ; then + echo "::error::posture-lint: host adjudication must call permits(), not matches() - see http::is_transparent_resolver (#533)" + exit 1 + fi + - name: No outbound-network APIs in unit / integration tests shell: bash run: | diff --git a/.github/workflows/release-plz.yml b/.github/workflows/release-plz.yml index 3175d28b3..8ed2d0b86 100644 --- a/.github/workflows/release-plz.yml +++ b/.github/workflows/release-plz.yml @@ -793,7 +793,45 @@ jobs: # `EALLOWGIT: Refusing to fetch "github:npm-stage/doiget-darwin-arm64"` # and the job died before reaching the registry at all. A path spec # has to look like a path. - for p in ./npm-stage/doiget-*; do - npm publish "$p" --provenance --access public --tag "$DIST_TAG" + # The platform list comes from `stage-npm.sh`'s MAP, not from a + # `doiget-*` glob. The glob used to be safe because the wrapper was + # named `doiget`; renaming it to `doiget-cli` made the glob match it + # too, so v0.8.12 published the wrapper inside the loop AND again on + # the line below -- `npm error You cannot publish over the previously + # published versions: 0.8.12`, after everything had in fact shipped. + # It also sorted first, so the wrapper went out ahead of the packages + # it pins: exactly the window the paragraph above warns about. + # + # Reading the MAP cannot drift the same way. It lists platform + # packages only, so the wrapper is excluded by construction rather + # than by a name test somebody has to remember to update. + # + # The empty case is guarded, and that is not defensive noise. The + # pipeline ends in `sort -u`, which exits 0 on empty input, so + # `set -euo pipefail` gives NOTHING here: a grep that matches + # nothing (a reformat of the MAP block, a delimiter change) yields + # an empty word list, the loop runs zero times, and the step falls + # through to publish the wrapper alone -- green, with every platform + # binary silently missing. `npm install doiget-cli` would then + # succeed while its optionalDependencies fail to resolve. + # + # The `doiget-*` glob this replaced failed LOUDLY on no-match (the + # unexpanded pattern reached `npm publish` as a path that does not + # exist). Trading that for silence in the step that performs the + # irreversible publish is the wrong direction. posture-lint's + # `capture()` already guards the identical grep for the identical + # reason; it was applied to the check and not to the publish. + pkgs="$(grep -oE '^doiget-[a-z0-9-]+:' scripts/stage-npm.sh | tr -d ':' | sort -u || true)" + if [ -z "$pkgs" ]; then + echo "::error::release: no platform packages found in scripts/stage-npm.sh -- refusing to publish the wrapper alone" + exit 1 + fi + count=$(echo "$pkgs" | wc -l | tr -d " ") + if [ "$count" -ne 4 ]; then + echo "::error::release: expected 4 platform packages, found $count: $(echo $pkgs)" + exit 1 + fi + for pkg in $pkgs; do + npm publish "./npm-stage/$pkg" --provenance --access public --tag "$DIST_TAG" done npm publish ./npm-stage/doiget-cli --provenance --access public --tag "$DIST_TAG" diff --git a/.github/workflows/typos.yml b/.github/workflows/typos.yml index d7a84222e..9f710dc7f 100644 --- a/.github/workflows/typos.yml +++ b/.github/workflows/typos.yml @@ -20,4 +20,4 @@ jobs: timeout-minutes: 5 steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - uses: crate-ci/typos@8a48f81b6c64dcfea44b3633223084c4be58ac5f # v1.49.0 + - uses: crate-ci/typos@4d9c206a77c041268485162b8e2579ad7a5cb9a3 # v1.50.0 diff --git a/.mcp.json b/.mcp.json index bee414dcc..7fe049159 100644 --- a/.mcp.json +++ b/.mcp.json @@ -1,8 +1,10 @@ { "mcpServers": { "doiget": { - "command": "doiget", + "command": "npx", "args": [ + "-y", + "doiget-cli@0.8.12", "serve" ] } diff --git a/CHANGELOG.md b/CHANGELOG.md index d0a9fb428..ece17a414 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,7 +8,853 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 `doiget-core` is the only crate with strict semver guarantees during the 0.x line; CLI flag changes and `doiget-mcp` tool spec changes will be called out explicitly here. -## [Unreleased] +## [0.8.13] - 2026-09-01 + +### Fixed + +- **[mcp] (breaking)** `doiget_batch_fetch` and `doiget_batch_from_bibliography` + reported a **refused paper as `INTERNAL_ERROR` and a 404 as `NETWORK_ERROR`** + (#538, review of #580). + + `fetch_paper_fetch_error_envelope` shed a hand-rolled copy of + `From<&FetchError> for ErrorCode` this cycle -- the one ending in + `_ => InternalError`. Two verbatim copies of that match stayed behind in + `build_bibliography_envelope` and `batch_fetch_success_envelope`, one screen + below the fix. So the headline of this release, "an access refusal is not an + internal error", held on `doiget_fetch_paper` and on no other fetch surface: + `NotFound`, `Ambiguous` and the brand-new `NotRetrievable` all fell through + the wildcard, and every `Http` status collapsed to `NETWORK_ERROR`. + + The last one is the damaging one. `disposition` is derived from the code, so + a 404 on a batch entry arrived as `retry_after` -- an agent told to keep + retrying a DOI that will never resolve, which is the exact failure + [ADR-0055](docs/DECISIONS/0055-error-disposition.md) exists to prevent. Both + sites call `ErrorCode::from(err)` now, and a test drives the real envelope + builders rather than a re-implementation of them. + +- **[cli] (breaking)** `doiget batch` reported a PMID-only bibliography entry as + `INVALID_REF` (#500). + + The reporting half of #500 landed on `doiget verify` and the MCP bibliography + tool and not here. The verdict the parser reached was flattened into a + placeholder string (``) and handed back to `Ref::parse`, + which has exactly one thing to say. So a PubMed-exported `.bib` record with a + valid `pmid` field was still reported as malformed -- the one claim #500 + exists to stop doiget making -- and the message described the placeholder + rather than the entry. The verdict travels as a value now + (`BatchEntry::Rejected`), so it cannot be flattened, and the comment claiming + the two arms made "different claims" is no longer false. + +- **[cli/mcp] (breaking)** `GraphError::Source` mapped to `NETWORK_ERROR` on the + MCP surface and to the wrapped `FetchError` code on the CLI. + + An OpenAlex 404 during citation-graph expansion was `NOT_FOUND` (terminal) on + `doiget graph` and `NETWORK_ERROR` (`retry_after`) on + `doiget_expand_citation_graph`. The CLI doc comment said the two "cannot say + different things about the same error". `From<&GraphError> for ErrorCode` now + lives beside the enum in `doiget-core`, and all three call sites -- the third + was an inline copy that did not even call the helper directly above it -- go + through it. + +- **[core]** A 404/410/451 carrying a `Retry-After` header was classified + `NETWORK_ERROR` (follow-up to #506). + + Adding `retry_after_ms` to `HttpError::HttpStatus` also added + `retry_after_ms: None` to the arm that maps authoritative absences, narrowing + it without comment or test. The header says how long to wait *if* you retry; + it does not make a 404 provisional. + +- **[cli/mcp] (breaking)** The arXiv-only text tools answered a DOI with + `NO_OA_AVAILABLE`, whose disposition is `needs_config`. + + There is no config knob: they are arXiv-only and DOI-to-arXiv linking is not + built. `NOT_IMPLEMENTED` is terminal and is what `verify` and + `batch_from_bibliography` already give the same situation for PMIDs -- valid + input, absent support. + + All five surfaces, not two: `doiget_paper_text` and `doiget_paper_tex_source` + over MCP, and `doiget text` / `doiget source` / `doiget tex-source` on the + CLI. The first pass changed the MCP pair only, which would have left one + binary answering the same question two ways -- and left the tools' own + advertised `description` strings, their input-schema field docs and + `docs/MCP_TOOLS.md` naming a code they no longer return. + +- **[mcp] (breaking)** `doiget_resolve_citation`, `doiget_batch_resolve_citations`, + `doiget_tag` and `doiget_annotate` put a **bare string** in `error`, so a + caller had no `code` to branch on and no `disposition` to decide a retry from. + All thirteen sites build `error_object` now. + + Section 3 of `docs/ERRORS.md` named `every_bare_string_error_site_is_a_known_one` + as the guard that kept that set from growing. **The guard did not exist.** The + document was accurate about the four tools and wrong about the mechanism + keeping the list honest -- a claim resting on code nobody had written, which + is the defect class that document defines. It exists now, the set it pins is + empty, and a second test fails if the document and the list ever disagree. + +- **[core]** `doiget config doctor --network` still adjudicated hosts with + `matches` rather than `permits` (#533). + + The `permits` doc comment said "the four adjudication sites ... all route + through here so they cannot disagree". This was a fifth, and the one whose job + is explaining allowlist refusals: the moment a resolver host entered its probe + list, the command that diagnoses #533 would have reproduced it. + +- **[core] (breaking)** CSL-JSON entries identified only by a PMID or PMCID + reported "entry has no DOI / arXiv id" (#500), and came out as `INVALID_REF`. + The BibTeX parser learned the distinction; this format did not, and it is the + one a Zotero user is most likely to export. Marked breaking for the same + reason as the `doiget batch` bullet above: the wire code changes to + `NOT_IMPLEMENTED` on `verify`, `batch` and the MCP bibliography tool for this + input shape. + +- **[cli]** `doiget search --mode json` emitted `{"ok": true, "total_results": 0}` + with no hint for an over-long query (#534). The fix landed on the MCP tool + only, so this surface went on emitting the exact envelope the issue was filed + about. `zero_result_hint` moved to `doiget-core`; both surfaces call it, and + human mode gets a `= note:` on stderr. + +- **[mcp]** `doiget_fetch_paper`, `doiget_batch_fetch` and + `doiget_batch_from_bibliography` logged `SessionEnd` as `result: "ok"` for a + blocked PDF leg (#507) -- on the single-ref tool, beside a non-null + `error_code`, so the row contradicted itself. + + The clean-success rule of the CLI was ported to the MCP surface in the + `error_code` half only, and then only to the single-ref tool: the two batch + tools went on calling `.all(|r| r.outcome.is_ok())` a success for a batch + where every entry was refused. That is the outcome the repeat suppression of + #507 has to be able to see. `FetchPaperOutcome::is_clean_success` is the + single boundary for all four call sites now, including the identity-line + check in the CLI that had hand-rolled it a fourth time. + +### Changed + +- **[test]** The Tier-3 TDM-fetched route is asserted end to end and no longer + `#[ignore]`d (#462). + + It was shipped as a failing reproduction, diagnosed as + "`tier_3_allowlists()` is `#[cfg]`-gated and the MCP server does extend its + allowlists with it, so the two disagree somewhere between construction and + use" -- the shape of #454, reachable again -- and raised as something to + decide before cutting this release. + + That diagnosis was wrong, in the subject matter of this very release: + accurate about the code, false about the world. Both client builders have two + branches. The production branch does extend with `tier_3_allowlists()` and was + correct throughout. The test-override branch -- taken whenever any + `DOIGET_*_BASE` is set, which every wiremock test does -- built its allowlist + from a fixed table of Tier-1/2 keys with no Tier-3 entry, so no e2e on either + surface could reach the route. The defect was in the harness. Registering the + Tier-3 keys there is what the test now proves by passing, and + `route_coverage_e2e` has no remaining `Gap`. + + What survives from that diagnosis, and is still true: `fetch_content` is + implemented by APS alone, so three of the four Tier-3 sources cannot reach the + route the tier exists for. Read it as APS coverage, not Tier-3 coverage. + +- **[docs]** `Formula/doiget.rb` and `scripts/update-homebrew-formula.sh` said + the release workflow regenerates the formula "so this file cannot describe a + release that does not exist". No workflow calls the generator, and + `CONTRIBUTING.md` -- added in the same release -- says outright that it is not + automated. The headers describe the real process now. + +- **[ci]** The Homebrew posture check said it "fails on a hand-edit, a bad + checksum, or a tag/version confusion". It re-reads the version and the + checksums from the file under test, so a well-formed checksum carrying the + wrong value, and a version that is simply stale, both pass. The step now says + what it checks and what it does not. + +- **[ci]** New `release-tracking files name one version` posture check. + `Formula/doiget.rb`, `.claude-plugin/plugin.json`, + `.claude-plugin/marketplace.json` and `.mcp.json` all record the last + published stable, nothing held them together, and that is how `.claude-plugin` + sat at 0.8.11 while 0.8.12 was shipping. Forgetting one of the four is red; + pinning a prerelease is red. + +- **[plugin]** `.mcp.json` pins `doiget-cli@0.8.12` instead of resolving the npm + `latest` tag on every server start. The unpinned form also meant that opening + this repository in Claude Code ran the *published release* rather than the + working tree -- editing `crates/doiget-mcp` and testing the build from last + month. `just mcp-dev` registers this checkout as a local-scoped server + instead, and `CONTRIBUTING.md` says so rather than leaving `.mcp.json` to be + edited. + +- **[cli]** `verify` and `batch` wrote wire codes as string literals + (`"INVALID_REF"`, `"NOT_IMPLEMENTED"`) next to a doc comment rejecting exactly + that. They come from `ErrorCode::as_wire` now. + +- **[test]** The `#[ignore]` detector in `route_coverage_e2e` matched any text + containing `#[ignore` in the block above a test, so a doc comment *discussing* + an ignored test made it report that test as skipped. It reads attribute lines. + +- **[docs]** In `crates/doiget-mcp/src/lib.rs`, the doc comment for + `store_root_env_is_usable` had drifted onto `zero_result_hint`. + +- **[scripts]** `scripts/*.sh` are executable, and `CONTRIBUTING.md` no longer + documents a command that fails on a fresh Linux or macOS clone. + +- **[test]** In `npm/doiget-cli/test/stage-npm.test.sh`, a `grep` matching + nothing killed the script under `set -e` *before* the `check no` written to + report that case, so the diagnostic was unreachable in the one situation it + exists for. + +### Fixed (accumulated on `next` since 0.8.12) + +- **[store]** A default `metadata_only` re-write **downgraded a known + `oa_status` and `license` to their not-determined markers**, and said nothing + (#583). + + `docs/STORE.md` §6 does let a re-fetch rewrite the `[doiget]` table, but the + permission is conditional: *"This is intentional, not silent: ... the operator + always learns the entry was downgraded and why."* Since #539, `metadata_only` + only runs the OA lookup when `include_oa_location` is set, so the ordinary call + shape produces `oa_status: None` and `license: "unknown"` — and those won, + with no `note:`, no `pdf.status` and no log row. The condition the permission + rests on was not met. + + Both values are markers, not readings: `oa_status` is "omitted when not + determined" and `license` falls back to `"unknown"`. A paper that genuinely + stops being open access reports `Some("closed")`; a license that changes + reports the new string. So a call that carries the marker did not look, and + preferring the stored value is not a guess about which is newer. + `merge_metadata` now keeps the stored value in exactly that case, and + `"unknown"` has a name (`LICENSE_UNDETERMINED`) so the check does not hang off + a bare literal. [ADR-0056](docs/DECISIONS/0056-not-determined-is-not-an-answer.md), + and `docs/STORE.md` §6's note is amended to scope its claim to determinations. + + The issue as filed said `oa_url` was overwritten with null. It is not — + `merge_opt!(url)` already protected it, which a probe confirmed before any of + this was written. Only the two `[doiget]` fields were affected, and the fix is + a merge rule rather than the store partition the issue proposed. + +### Changed + +- **[ci]** One test was taking ten minutes and being run five times. The five + `test` jobs each run the whole workspace suite, and 95% of that suite's time was + a single test: `batch_above_window_size_fetches_every_ref`, 627 s out of 663 s, + with the next-slowest at 36 s. + + It is not waste. The test fetches `MCP_BATCH_MAX_SIZE + 2` arXiv refs; + `SOURCE_RATE_OVERRIDES` puts 3 s between arXiv requests because arXiv's Terms of + Use do, and one attempt issues two requests — the Atom feed then the PDF, both + paced since #493. So 102 x 2 x 3 s = 612 s, against 609 s measured. The rate + limit is a legal safeguard (`RateLimits`' fields are `pub(crate)` precisely so + tests cannot weaken it), and it stays exactly as it is. + + What was wasteful was paying it five times, on three operating systems and two + feature sets, for a dispatch-loop property that varies with neither. The test is + now `#[ignore]`d and a `test (slow)` job runs it once via `--ignored` — exactly + one ignored test exists workspace-wide, so the split drops nothing and duplicates + nothing. The broad suite went from 627 s to 18 s locally; the slow job runs in + parallel rather than lengthening any of the others. + + The test's own comment claimed "~20 s". That was true when it was written + (2026-06-16, 200 ms x 102); `2cc32ab` on 2026-08-25 gave arXiv its published 3 s + interval and made it 15x slower, and the only symptom was CI minutes. Corrected + in place with the arithmetic. + + Not attempted: `tokio::time::pause()`. `http.rs` already records why — wiremock + serves over real localhost IO, and paused time auto-advances past reqwest's + timeout. + +- **[dist]** The Claude Code plugin runs `npx -y doiget-cli serve` instead of a bare + `doiget serve`, so installing it **needs nothing installed beforehand** — npm + fetches the wrapper and the one matching platform binary on first run. Until now + the plugin only worked for people who had already installed doiget some other + way, which is why it was kept as a self-hosted marketplace rather than submitted + to the Anthropic plugin directory: a listing whose first run fails for everyone + without the binary on PATH is worse than no listing. That objection is gone, so + the directory submission is now viable (#513). + + Verified by launching it: `npx -y doiget-cli serve` from the real registry + answers `initialize` as `doiget 0.8.12` and lists 22 tools, with pure JSON-RPC on + stdout and zero bytes on stderr. + +- **[deps]** dependency bumps merged from Dependabot: `flate2` 1.1.9 -> 1.1.10, + `uuid` 1.25.0 -> 1.26.0 (#578), `quick-xml` 0.41.0 -> 0.42.0 (#579), and the + `github/codeql-action`, `taiki-e/install-action` and `crate-ci/typos` CI + actions (#581). `cargo-vet` `safe-to-deploy` exemptions updated to match. + + `quick-xml` 0.42 is not a drop-in bump: the reader is UTF-8 throughout, so + `QName::as_ref()` and `Attribute::key` yield `&str` rather than `&[u8]` and + `BytesText::decode()` is gone. Both XML parsers were migrated -- `local_name` + takes and returns `&str`, XML-name comparisons lost their `b` prefixes, and + decode-then-unescape collapsed to `quick_xml::escape::unescape`. Behaviour is + unchanged; the existing parser tests cover the touched paths. + + `flate2` 1.1.10 changes its own dependency set: with `rust_backend` it now + pulls `zlib-rs` 0.6.7, a dependency this tree did not have before. It is + exempted at `safe-to-deploy` like every other unaudited crate here, which is a + statement about process, not about anyone having read it. + +### Added + +- **[mcp]** `error.retry_after_ms` - the server's own `Retry-After`, on the + failure envelope (#506). + + I had deferred this on the grounds that no honest number survives to the + boundary: `Retry-After` is parsed by `parse_retry_after`, and the retries are + exhausted by the time an error surfaces. That was too pessimistic and worth + correcting. The header was read **only on the retry path**; the terminal + `return` discarded it. The response that *ends* the attempt carries its own + `Retry-After`, and that is exactly the one a caller should wait. + + `HttpError::HttpStatus` carries it now, and the envelope surfaces it - but + **only when the server sent one**. It is never backfilled from doiget's + internal `backoff_delay`: that is a guess about the server, and a guess + wearing the name of a server-supplied value is the defect `disposition` and + this field exist to remove. Absent means absent. + + Pairs with the disposition: `retry_after` says retry, and this says how long + the server asked you to wait first. +- **[test]** A registry of which PDF route each e2e test asserts, because + measurement showed almost none of them did (#462). + + Four "unreachable source" bugs shipped with green unit tests - #413, #442, + #454, #458 - and they share one shape: a source was implemented, gated, + allowlisted and unit-tested, and was never reached. The unit test drove the + `Source` impl directly and the production entry point was never in the + picture. #454's own guard carried a doc comment describing the failure it + could not catch. + + Measured before writing anything: of the five `PdfLegStatus` routes, exactly + **one** (`blocked`) was asserted anywhere in the e2e suites. `tdm_fetched` had + none - which is how #458, "the Tier-3 chain is skipped whenever Crossref + answers", could ship. + + `route_coverage_e2e.rs` records, per route, either the test that asserts it or + a stated reason it does not. Three checks keep that honest: + + - a claim of coverage must name a test that **exists and contains the route + string**, so it cannot outlive the assertion it names. This caught a wrong + test name on its first run; + - a gap must carry a reason, because an unexplained gap is indistinguishable + from an oversight; + - the registry is compared against the `PdfLegStatus` variants in + `doiget-core`, so a new route cannot be added without deciding how it is + covered - the step all four bugs skipped. + + Coverage went from **1 of 5** to **4 of 5**. `fetched`, `no_oa_url` and + `preprint_fallback` gained assertions; the known-gap count is asserted, so + closing one is a visible edit rather than a silent improvement. + + `preprint_fallback` is worth naming: the existing blocked-leg test asserted + `suggested_arxiv_id`, which is the SUGGESTION. The #325 fallback actually + running is a different route one field away in the same envelope, and the + only thing separating them in the harness was an unset `DOIGET_ARXIV_BASE`. + +- **[test]** Writing the fifth route's test found a defect, and it is left as a + failing reproduction rather than muted (#462, #454). + + `tdm_fetched` is reachable only through `tdm-aps` - it is the sole Tier-3 + source that implements `fetch_content`; Elsevier, Springer and IEEE inherit + the default `Ok(None)` and are metadata-only. Driven over MCP with the grant + set, the chain **does** fire, and then: + + ``` + "detail": "network error: no allowlist registered for source tdm-aps" + ``` + + That is `HttpError::UnknownSource` - the source key is not in the client's map + at all - even though `tier_3_allowlists()` is purely `#[cfg]`-gated and the + MCP server extends its allowlists with it. #454's shape ("Tier-3 allowlists + never registered in the client"), reachable again. + + The test is `#[ignore]`d with the error in its doc comment, so it is a + reproduction someone can run rather than a gap someone has to rediscover. + +- **[cli]** The found-nothing path now orders the sources it did not consult - + and says which part of that order is a finding and which is not (#505 part 3). + + The issue is emphatic about the risk, and it governs the whole design: *"a + ranking that is wrong is worse than no ranking, because it makes people stop + early."* + + ``` + = note: of the sources not consulted: + 1. openalex lists every location a work has -- a lookup, not a guess + then, in NO particular order: doaj europe-pmc hal openaire + last: core the broadest index outside Unpaywall, so never the first try + ``` + + Two positions have a real signal. `openalex` is categorically different from + everything else in the list: it *lists* a work's locations, so with it enabled + the answer is "this repository has it", not "might". `core` is last on its own + module's documented grounds - broadest means least discriminating. + + **The middle is returned unordered, deliberately.** The issue proposes ranking + it on venue, author affiliation and funder; none of those reach this point - + `FetchPaperOutcome` carries title, authors and year, and the DOI-prefix map is + Tier-3-only (ADR-0041) and absent from an `oa-only` build entirely. Rendering + an invented order would put a guess in the shape of a finding, which is the + one thing this must not do, so the output says so instead. + + It is an ordering of the **full** list, never a shortlist: a source dropped + from the list is a source the reader will not try, and there is a test that + every unconsulted source appears somewhere. + +- **[dist]** Homebrew. `#247` promised a tap in 2026-06 and was closed as + completed; measured against 0.8.12 it was one of the rows that did not exist + (#501). + + ```sh + brew tap QAtlasHub/doiget https://github.com/QAtlasHub/doiget + brew install doiget + ``` + + The tap lives in this repository rather than a separate `homebrew-doiget`, + which is why the tap line carries an explicit URL; a dedicated tap repo would + shorten it and the formula would move across unchanged. + + `Formula/doiget.rb` installs the same signed release binary the GitHub Release + publishes, pinned by the `sha256` from that release's own `.sha256` asset - + the file the shell installer verifies against, so the two channels cannot + disagree about what they installed. + + The formula is **generated**, never hand-edited, by + `scripts/update-homebrew-formula.sh`, and CI fails if the committed file is + not what the generator produces. #247 was closed while four fifths of it had + not shipped, so this channel is asserted rather than trusted: the tests refuse + a bad or truncated checksum, catch a `v`-prefixed version that would install + but compare wrong, and catch a hand-edit. + + Updating it after a stable release is a documented maintainer step rather than + an automated one, because the checksums do not exist until the release has + published and committing them back needs a token with write access to a + protected branch - which is what #426 says is broken. The step cannot be done + wrong, only forgotten. + +### Changed + +- **[docs]** The README channel table stopped guessing. Nix said "whether it + exposes an installable package rather than a dev shell is unverified"; it does + (`flake.nix` has `packages.default` and `packages.doiget`, not only a dev + shell), and the row now says that the outputs exist while `nix profile + install` has not been exercised - which is what is actually known. + + **Docker is recorded as not planned**, with the reason. #501 ranked it second + on the grounds that a container is "the only architecture the Tier-3 features + can legally be used in". That premise does not survive ADR-0002, which decides + the default published binary contains no TDM source code at all - an image + built from it would ship `oa-only,citation` like every other prebuilt channel. + What remains is "a shape enterprises can pin and scan", and doiget is a single + statically-linked binary, so a container solves no dependency problem that the + existing checksum and cosign bundle do not already address. + +- **[mcp]** Citation candidates now carry `confidence` and `matched`, so an + agent can tell an identity from a coincidence. A 0.5 near-miss and a 1.0 exact + match arrived in the same shape, and nothing in the envelope said which was + which (#536). + + The reported case: a citation for a paper in *Psychiatria Danubina* came back + as a different 2010 paper, in a different journal, by a different author, at + `score: 0.5` - `quality`, `life`, `bipolar`, `2010` were enough to clear the + bar. Same structure as the `score: 1.0` identity returned minutes earlier. + + A bare float is not judgement material. To use it as a gate the consumer has + to already know that the scorer is token overlap rather than semantic + similarity, and that **0.5 is the floor** - so the worst candidate the tool + can ever emit still looks like a positive number. + + | `confidence` | meaning | + |---|---| + | `exact` | every query token was found | + | `probable` | at least four query tokens in five | + | `weak` | cleared the floor and no more - a near-miss, not a match | + + `matched` lists which of the query's tokens were found. In the reported case + that is the whole story: not the author, not the journal. + + This is the half of #372 that was never specified. #372's remedy - return the + top candidate with its score rather than an empty list, so the calling agent + can judge - is in place and correct; what the agent judges *with* was missing. + Both citation tool descriptions now say to branch on `confidence`, not + `score`. + + **`doiget-core` API:** `ResolvedCandidate` gains two fields and becomes + `#[non_exhaustive]`, matching `MetadataOnlyOutcome` and `AttemptOutcome`, so + the next field is not another break. +- **[mcp]** Every failure envelope now carries `error.disposition`, so the retry + decision stops being a markdown table the agent never reads. `docs/ERRORS.md` + §2 has had good per-code guidance since Phase 0; none of it reached the wire + (`grep -riE 'retryable|transient|retry' crates/doiget-mcp` returned nothing in + the tool descriptions), so an agent's only signal was the **name** of the code + - and several names point the wrong way (#506). + + Three states, because two cannot express the one that matters: + + | value | meaning | + |---|---| + | `terminal` | the answer will not change. Do not retry. | + | `retry_after` | it may change on its own. Retry with backoff. | + | `needs_config` | it will not change by itself, but a named change makes it. | + + `NO_OA_AVAILABLE` is `needs_config`. It is the most common failure there is, + and both its name and its old row ("Try later, or enable opt-in source") read + to a machine as *wait* when it is nearly always *configure* - an invitation to + loop forever over something that will not change. + + Derived in one place (`ErrorCode::disposition`, an exhaustive match with no + wildcard) and built in one place (`error_object`), because a field present on + some failures and absent on others teaches the reader to go back to guessing. + `docs/ERRORS.md` §2 gains a Disposition column, and a test parses the shipped + document and asserts every row against the function - so the doc and the wire + cannot drift. The test also asserts it parsed exactly 15 rows, since a parser + that silently matches nothing passes every time. + + The contract is stated in the MCP server `instructions`, which every client + receives on `initialize`, so an agent that has never opened `ERRORS.md` still + meets it. See ADR-0055. + + Three parts of #506 are **not** here and it stays open for them: + `error.retry_after_ms` (the `Retry-After` is consumed inside the retry loop + and no honest number survives to the boundary - a plausible default would be + indistinguishable from a measured one), `remediation` on `ok:false`, and + `rate_limit_budget` on live responses. + +- **[cli]** The found-nothing fetch now reports what it consulted. Previously + `fetched ... (metadata-only: no OA PDF available)` was byte-identical whether + the optional sources were on and had nothing or off and never asked - and it + exits 0, so unlike the blocked path there was no `error[...]` block for the + #413 trace to hang on. It is the one outcome that reads as a *result*, which is + where the silence misleads most: with the default profile that sentence means + only "three of eleven sources had nothing" (#505). + + It now prints the same attempt trace the blocked and NOT_FOUND paths already + print, plus the command that widens the search - the line to paste, not prose + about it: + + ``` + = suggest: to widen the search: + DOIGET_ENABLE_HAL=1 DOIGET_ENABLE_CORE=1 doiget fetch 10.1137/0117004 + ``` + + The variables come from the `Disabled` rows themselves rather than a second + registry, so the advice cannot claim a source was skipped when it was + consulted, or name a switch the chain does not read. + + A switch that is **already set** is reported as a build problem instead: + `resolve_metadata_flag` returns false when the variable is set but the Cargo + feature was not compiled in, so the source still reports `Disabled` naming a + variable the user set an hour ago. Telling them to set it again would be the + same species of unhelpful as the bare `no OA PDF available`. + + Part 3 of #505 - ranking which source is most likely to hold the paper - is + deliberately not here. The issue makes the case that a wrong ranking is worse + than none because it makes people stop early, and that deserves its own change. + #505 stays open for it. + +### Fixed + +- **[review]** A six-agent review of the 0.8.13 promotion found eleven defects, + and most of them were the same class the release is about: **a statement that + was accurate about the mechanism and wrong about the world.** Fixed here. + + - `oa_url` was documented as "always null unless `include_oa_location` is + set". **False.** The pre-existing Crossref-failure fallback fills it from + Unpaywall regardless of the flag. `source` distinguishes the two. + - `AttemptOutcome::NotOpenAccess` could assert the **opposite** of what + OpenAlex reported: a location with `is_oa: true` and no `pdf_url` was + labelled "not open access" on the machine-readable token, with the + contradicting evidence buried in prose. Now only when nothing was flagged + open. + - The "line to paste" rendered every widening variable as `VAR=1`, producing + `DOIGET_KEY_APS=1` - an API key that can never be valid. + - `docs/ERRORS.md` claimed `disposition` is **ALWAYS** present. Four tools + (`resolve_citation`, `batch_resolve_citations`, `tag`, `annotate`) put a + bare string in `error`. The claim now names its real scope. + - #507's bookend fix landed on the CLI and not on `doiget_fetch_paper`, so + the MCP path logged a blocked PDF leg as a clean, error-free success. + - `fetch_paper_fetch_error_envelope` hand-rolled a copy of + `From<&FetchError> for ErrorCode` ending in `_ => InternalError`, so a + mistyped DOI reported `INTERNAL_ERROR` instead of `NOT_FOUND`. It + delegates now. + - The #462 route registry's own self-check could be satisfied without the + claim being true, and had **no concept of `#[ignore]`** - a covering test + CI never runs would have counted. It now reads the named function's body + and refuses a skipped test. The guard had the bug it was built to catch. + - `Confidence::from_score` accepted any `f64` and answered confidently; + `europepmc`'s `NotRetrievable` carried `source_key: "europepmc"` against a + `Source::name()` of `"europe-pmc"`; `blocked_trace_lines` lost its doc + comment to a function inserted above it; ADR-0055 listed as "not in scope" + something the same cycle shipped; user-facing strings carried runs of + joined-line whitespace. + + Also: `disposition` was inserted at six envelope sites and asserted at two - + deleting one insert failed no test. The batch site now asserts it. + +- **[bib]** A PubMed-exported bibliography entry was reported as having **no + identifier**. It has a PMID (#500). + + `entry has no DOI / arXiv id` is accurate about what the parser did and wrong + about the entry. A user reading it goes and edits a `.bib` that was fine; the + missing piece is on doiget's side. An existing test pinned exactly that claim, + with the comment "the entry has no resolvable identifier" - about an entry + carrying `eprinttype = {pubmed}`. + + Both PubMed shapes are now named: the `pmid = {...}` field that PubMed's own + BibTeX export writes, and the BibLaTeX `eprint` + `eprinttype = {pubmed}` pair + that `arxiv_eligible` already refused, correctly and until now silently. + `pmcid` too. + + It surfaces as **`NOT_IMPLEMENTED`, not `INVALID_REF`** - the input is valid + and the support is absent, and the two carry opposite advice ("wait for a + release" versus "correct your input"). Its disposition is `terminal` + accordingly (ADR-0055). + + An entry with genuinely no identifier still reports `NoIdentifier`, and a DOI + alongside a PMID still wins: the check runs only after every supported + identifier has been tried, so it cannot divert an entry doiget could have + resolved. + + This is the reporting half of #500. `Ref::Pmid` itself is blocked on something + the issue did not anticipate - see below. + + The sentence the user reads lives in one place now + (`refs::unsupported_identifier_claim`). It had been hand-copied to the CLI + `verify` row and the MCP `batch_from_bibliography` envelope, and both copies + had been wrapped across source lines and re-joined with the indentation still + in them - shipping `which doiget cannot resolve yet` to a + reader. Nothing asserted the text, so nothing failed. `#[error]` already + carried the same claim; the copies existed only to drop the `entry_key` + prefix, which each site puts in a field of its own. + + Three more of the same, pre-existing and found while looking: two `= note:` + lines in `fetch fetch`'s widening advice and the `config.toml could not be + read` warning in `commands/mod.rs`. A workspace-wide lint for the pattern was + written and abandoned - it flags `config doctor`'s aligned two-column output + and test assertion messages at every space threshold from four to ten, and a + lint that cries wolf is a lint that gets deleted. Removing the duplication is + the durable half. + + +- **[mcp]** `error.disposition` was **missing from five failure envelopes**, + including the two most common failures an agent sees. The change that + introduced it converted the envelopes built by `error_object` and missed the + ones assembled field-by-field, which are exactly the ones that also attach a + `denial_context` (#506). + + A field present on some failures and absent on others is worse than no field: + it teaches the reader to fall back to guessing from the code's name, which is + the habit the disposition exists to replace. That is stated in ADR-0055 and + was then not upheld. Now pinned by a posture-lint step that counts one + `insert("disposition"` per `insert("code"` - not a proof, but it fails on + exactly the mistake that was made. + +- **[mcp]** `remediation` is carried on failure envelopes. `docs/ERRORS.md` §3 + said outright that it "belongs to the `{ ok: true, ... }` envelope", so the one + field naming the fix was present when a call succeeded with a blocked leg and + absent when the call actually failed. It comes from the same + `remediation::for_denial` the blocked leg and the CLI `= help:` block use, and + is omitted when there is no named fix rather than emitted empty. +- **[provenance]** A `session_end` row recorded **that** a call failed and not + **what** it failed with: every one carried `error_code: null`, including the + rows that carry a `ref`. So the log could not answer "what did this session + tell the caller about this ref?" - only "something went wrong" (#507). + + That is a gap in the audit trail on its own terms. It is also what blocks the + repeat suppression #507 asks for: the rule is "do not re-fetch a prior + `terminal` or `needs_config` answer", and a disposition (ADR-0055) cannot be + recovered from a row with no code. Measured before building anything - the + only per-`ref` `fetch`/`err` rows carry `NETWORK_ERROR`, whose disposition is + correctly `retry_after`, so a suppression guard written against today's log + fires on nothing at all. + + The bookend now records the code the caller was given. For `doiget fetch` + that is deliberately **not** always the `Result`'s: a blocked PDF leg is `Ok` + with a failed leg and an unclean session, and it is the outcome an agent is + most likely to retry, so the leg's closed-set code is recorded rather than + `null`. `doiget graph` gains an exhaustive `GraphError` -> `ErrorCode` + mapping placed next to the `map_err` that renders it, so the row and the + message cannot disagree. A batch bookend spans many refs and still records + `null`, because there is no single code; the per-ref rows carry those. + + `docs/PROVENANCE_LOG.md` §3 gains the `error_code` row it never had, and + states the layering explicitly: a failed `fetch` leg records the **transport** + mechanism (`NETWORK_ERROR` for a policy block, per `ERRORS.md` §6.1) while + `session_end` records the code the caller saw, after reclassification. The two + rows for one blocked fetch legitimately differ, and each is true about its own + layer. + +- **[fetch]** When OpenAlex named a repository copy that could not be followed, the + run said `no OA PDF available` - a different claim from what OpenAlex actually + reported (#547). + + `10.1109/tsp.2023.3269664` has two OpenAlex locations, and the second **is** the + Strathprints institutional deposit. It has no `pdf_url`, `is_oa` is `false`, and + its `landing_page_url` is an author-listing page (`/view/author/70486.html`) + rather than an item, so `open_access_pdf_url` correctly returns nothing. The + fact that a repository had been NAMED then reached only a `tracing::debug!`. + + The attempt row now carries what the source said: how many locations, how many + flagged OA, how many had a PDF URL, and the repository's name. "openalex named + 2 location(s) (IEEE Transactions on Signal Processing; Strathprints: The + University of Strathclyde): 0 flagged open access, 0 with a PDF URL" points the + reader at the repository. "no OA PDF available" points them at giving up. + + Diagnostics only - no new request, no change to what is fetched. The report's + second suggestion, following a repository platform's own search when its URL is + not an item, is a real capability change and belongs with #474; #547 stays open + for it. + +- **[core]** An access refusal was classified by reading an error message back. + A source saying "I found it and cannot give it to you" returned + `FetchError::SourceSchema` with an explanatory hint, and the orchestrator + decided what the trace row said by substring-matching that hint for + `not open access` / `openAccess` / `no retrievable PDF`. Match and the row + read *"found, not open access"*; miss and it read *"failed"* - which tells an + operator the source broke rather than that the paper is not free there (#538). + + It had already fired. #503 reworded Europe PMC's refusal for good reasons, the + hint fell out of the predicate, and every Europe PMC refusal silently became + `Failed`. Nothing in the source said the wording was load-bearing, and `hal` + matched on `openAccess` - a JSON **field name**, not prose anyone chose. + + Sources now return `FetchError::NotRetrievable { source_key, detail }` and the + classifier matches the variant. `is_access_refusal` is gone. The compiler + found a second exhaustive match the substring approach had no way to flag. + + It collapses to the **existing** `NO_OA_AVAILABLE` rather than adding a wire + code: "found it, no free copy" is what that already means, so the closed set + in `docs/ERRORS.md` §3 does not widen. Its §2 description does - it said + "Tier 1 sources reported no OA URL" and now also covers an optional source + holding the record with nothing retrievable. See ADR-0054. + + Side effect worth naming: `SourceSchema` collapses to `INTERNAL_ERROR`, so + until now a paper simply not being free at one repository could be reported as + a bug in doiget. It no longer is. + +- **[fetch]** A gold-OA, cc-by paper was refused at `doi.org`, one hop before the + publisher whose host was already on the allowlist. Unpaywall reports + `best_oa_location.url` for `10.1002/pcn5.205` as literally + `https://doi.org/10.1002/pcn5.205`, with no `url_for_pdf` - the normal shape for + publisher-hosted gold OA - so the first host doiget touched on the fetch leg was + the DOI resolver, and it was adjudicated as though it were where the bytes come + from (#533). + + The remediation the denial emitted was worse than the refusal: it advised adding + `doi.org` to `[[network.additional_hosts]]`. That does not widen the trusted + surface toward one publisher. It removes the bound entirely, because every DOI in + existence resolves through it, and an agent following the advice would get its + PDF while silently losing the invariant the allowlist exists to hold (ADR-0027). + + A closed set of resolver hosts - `doi.org`, `dx.doi.org`, `hdl.handle.net`, each + measured issuing a single 302 straight to the publisher - is now **transparent**: + followed, but never allowlisted, never named as remediation, never counted as the + source of the content. Matching is exact, not the usual suffix-glob, because + `*.doi.org` would sweep in `www.doi.org` (the DOI Foundation's website, not a + resolver) and `evil-doi.org` is what an attacker registers. The host that actually + serves the response is adjudicated exactly as before. See ADR-0053. + + This does **not** manufacture access. `10.1002/pcn5.205` now reaches Wiley and + meets Wiley's own cookie wall; it fails honestly at the publisher instead of + dishonestly at the addressing layer. + + There were **five** gates asking this question and only one had ever been walked + end to end - #462's point exactly. They now share one predicate, + `SourceAllowlist::permits`, and a posture-lint step fails any gate that calls + `matches` on a host variable, so a sixth cannot be added without the fifth's + lesson. +- **[mcp]** `oa_url` was documented as an OA URL "for the caller to act on + separately" and was, in practice, `null` for every DOI. The DOI path is + Crossref-first on the rationale that Crossref's `message.link[]` supplied an + OA URL without a second request - but #517 measured twelve live `link[]` + entries and eight captured fixtures and found **not one** general-purpose + entry: every one was scoped to a licensed programme (Similarity Check, TDM, + syndication). The extractor refuses all of them, correctly, which left the + field permanently empty while its documentation told agents to act on it. An + agent reading `oa_url: null` had no way to tell it from "this work has no free + copy" (#539). + + A caller that wants a real OA location now passes `include_oa_location: true` + to `doiget_metadata_only` or `doiget_resolve_paper`, which consults Unpaywall. + It is off by default: the default path stays one round-trip, and nobody pays + for a location they will not use. No guarantee is weakened - Unpaywall is a + metadata source, and the URL is still reported and never followed. + + With the flag set, `oa_status` says which answer you got: `"closed"` means the + lookup completed and there is no OA location; `null` means the lookup did not + complete. A failed Unpaywall call therefore leaves **both** fields null rather + than inventing a status, because asserting `"closed"` on the strength of a 500 + would recreate the exact ambiguity being fixed. The Crossref metadata is still + returned; an optional extra failing does not sink the resolve. This mirrors + the `oa_status` + `pdf.status` pairing `doiget_fetch_paper` already uses. + + The resolver cache keys on the options as well as the ref, so a default entry + is never served to a caller that asked for the location. The key is a + subdirectory rather than a `.oa` filename suffix: `Ref::safekey` keeps `.`, so + a suffix would make the DOI `10.1234/foo.oa` and the opt-in entry for + `10.1234/foo` collide on one file. That collision was written, then caught by + the test that now guards it. + + `doiget-core` gains `MetadataOnlyOptions` and `*_with_options` variants of + `metadata_only`, `resolve_only` and `metadata_only_to_store`; the existing + three delegate with defaults, so nothing downstream breaks. + +- **[mcp]** `doiget_paper_search` returning nothing now says whether that is about + the query or about the literature. OpenAlex free-text matching degrades sharply + past roughly eight terms and returns **nothing** rather than a partial match; a + human reading `0 results` shortens the query and retries, but an agent reading + `ok: true` with an empty array reads it as a fact about the world and stops. In + the session behind #534 that happened eleven times in a row, and a known 1992 + *Am J Psychiatry* paper was written off as unavailable — until an unrelated + three-term query surfaced it immediately. + + A zero-result search is a **success** envelope, so #506's error-disposition work + does not reach it and #505's is scoped to the CLI. The envelope now carries a + `hint` naming the submitted term count and what to retry with, at the exact point + an agent would otherwise conclude absence. Short queries get no hint: a two-term + search really may mean the work is not indexed, and hinting on every empty result + would train readers to skip it. The tool description says the same thing, so the + advice survives an agent that reads schemas but not envelopes (#534). +- **[docs]** Four tools shipped in every release with no entry in `MCP_TOOLS.md`: + `doiget_batch_from_bibliography`, `doiget_paper_tex_source`, `doiget_tag` and + `doiget_annotate`. That document is what an integrator reads to decide what the + server can do, and `docs/INTEGRATION/*.md` points at it, so a tool absent from it + is discoverable only by calling `tools/list` and reading JSON Schemas — the work + the document exists to save. Two of the four arrived with the tags work (#294) and + one with the TeX source work: the tool landed, the reference page did not (#553). +- **[ci]** `posture-lint` now compares the tool table against the router in **both** + directions. Nothing did, and it had drifted both ways at once: four tools with no + row (#553) and one row with no tool (#552, the citation graph absent from every + release binary). Neither is visible from inside the other file. rmcp derives a + tool's name from its `#[tool]` method name, so the method list is the router's own + truth; the table rows are the document's. Mutation-checked in both directions — + deleting a row and inventing one each fail the check with the right message. + +- **[dist]** `.claude-plugin/plugin.json` and `marketplace.json` said **0.8.11** + while the repository was two releases ahead. Nothing stamps them — no script, no + workflow references either file — so they drift silently, and `/plugin + marketplace add` reads the default branch, meaning users saw the stale number. + Set to 0.8.12, the last released version (#513). +- **[docs]** Two more references the `doiget` → `doiget-cli` rename left behind: + `README.md`'s plugin paragraph promised `npx -y doiget serve`, and the 0.8.11 + release notes lead with two npm commands that were never valid — the wrapper is + not called `doiget`, and that release's npm publish failed outright, so nothing + was on the registry to install. The 0.8.11 entry is annotated rather than + rewritten (#511). + +- **[ci]** The npm publish job published the wrapper **twice** and failed on the + second attempt: `npm error You cannot publish over the previously published + versions: 0.8.12`. The loop globbed `./npm-stage/doiget-*`, which was safe while + the wrapper was named `doiget` and stopped being safe the moment it was renamed + to `doiget-cli` — that name starts with `doiget-` too. So the wrapper went out + inside the loop and again on the explicit line after it. It also sorts before + `doiget-darwin-*`, so it published ahead of the packages its + `optionalDependencies` pin: the exact window the comment above that loop warns + about. + + Everything had already shipped by then, which is the worst shape a failure can + take — a red job on a complete release. The 0.8.12 npm packages are correct and + live; only the job's exit status was wrong. + + The loop now reads the platform list from `stage-npm.sh`'s MAP, which lists + platform packages only, so the wrapper is excluded by construction rather than by + a name test somebody has to remember. `stage-npm.test.sh` bans the glob outright + and asserts no package is published twice; mutation-checked in both directions. + + Worth recording: the identical trap was spotted in `posture-lint`'s + `find -name 'doiget-*'` and excluded there, in the very PR that did the rename + (#549). One of the two `doiget-*` patterns got the fix (#511). ## [0.8.12] - 2026-08-27 @@ -222,6 +1068,12 @@ flag changes and `doiget-mcp` tool spec changes will be called out explicitly he supply-chain shape a reviewer is trained to reject. Published by the release workflow over npm Trusted Publishing (OIDC, no long-lived token), after verifying each binary against the release's own `.sha256` (#511). + + **Neither command in this entry was ever valid.** The npm publish failed on this + release (see the `[ci]` entry below), so nothing was on the registry; and the + wrapper is `doiget-cli`, not `doiget` — npm refuses the bare name as too similar + to the unrelated `giget`. The channel first went live at 0.8.12. Left in place + rather than rewritten, with this note, because the failure is the point. - **[dist]** **Claude Code plugin.** `/plugin marketplace add QAtlasHub/doiget` then `/plugin install doiget@doiget`. Self-hosted, so it needs approval from nobody; both manifests pass `claude plugin validate` (#513). diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 512dc1bec..1920928bb 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -216,6 +216,66 @@ is the binding spec and full runbook. Summary: to the fixed commit (the pipeline runs the workflow/scripts *as of the tagged tree*). Do not reintroduce a perpetual "release PR". +### One step after a stable release + +Four files record *the last published stable*, and none of them can be written +before the release exists: `Formula/doiget.rb` pins the binaries by `sha256`, +and `.mcp.json` pins the npm version the plugin runs. So they lag the tag by +one commit, on purpose, and closing that gap is a single maintainer step: + +```sh +V=0.8.13 # the version that was just published + +bash scripts/update-homebrew-formula.sh "$V" # reads that release's .sha256 assets +sed -i "s/\"version\": \"[^\"]*\"/\"version\": \"$V\"/" .claude-plugin/plugin.json .claude-plugin/marketplace.json +sed -i "s/doiget-cli@[^\"]*/doiget-cli@$V/" .mcp.json + +bash scripts/update-homebrew-formula.test.sh # the same check CI runs +git commit -s -m "chore(release): tracking files for $V" -- Formula/doiget.rb .claude-plugin .mcp.json +``` + +All four edits are commands on purpose. The first version of this block gave +three of them as `#` comments between two real commands, so copy-pasting it +bumped the formula, silently skipped the other three, and produced a commit +whose message said all four had been done. + +Only for **stable** releases. Beta tags are not published to the tap, and the +plugin must never pin a prerelease -- posture-lint fails on both. + +It is not automated, and that is a decision rather than an omission: committing +these back from the release workflow needs a token with write access to a +protected branch, and that token is exactly what +[#426](https://github.com/QAtlasHub/doiget/issues/426) says is broken. Building +the tap's correctness on a known-broken token would be worse than a documented +step. + +What CI does enforce, so the step cannot be done *wrong* or half-done: + +- `update-homebrew-formula.test.sh` fails on a hand-edit, a malformed + checksum, or a `v`-prefixed version. It does **not** reach the published + `.sha256` assets (it is deliberately offline), so it cannot see a + well-formed checksum carrying the wrong value. +- The `release-tracking files name one version` posture check fails when the + four files disagree. Bumping three and forgetting the fourth is red; + forgetting all four is not, which is what `brew info doiget` is for. + +### Running the MCP server from your working tree + +`.mcp.json` is the plugin's server declaration, so it pins the last published +release. Opening this repo in Claude Code therefore runs *the release*, not +your checkout -- you can edit `crates/doiget-mcp` and be testing the version +you shipped last month. Do not edit `.mcp.json` to work around that; a +local-scoped server of the **same name** shadows the project-scoped one: + +```sh +just mcp-dev # registers this checkout's build, as `doiget` +just mcp-dev-off # removes it; `.mcp.json`'s pinned entry reappears +``` + +The name matters. `claude mcp add doiget-dev --scope local` leaves BOTH servers +connected, so the release-pinned tools stay callable and nothing is shadowed -- +which is how the first version of this section was written. + ### npm: the one-time bootstrap The `publish to npm` job authenticates over OIDC and holds no `NPM_TOKEN`. diff --git a/Cargo.lock b/Cargo.lock index f84b7ed8f..cb76a31cb 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -524,7 +524,7 @@ dependencies = [ [[package]] name = "doiget-cli" -version = "0.8.12" +version = "0.8.13" dependencies = [ "anyhow", "assert_cmd", @@ -554,7 +554,7 @@ dependencies = [ [[package]] name = "doiget-core" -version = "0.8.12" +version = "0.8.13" dependencies = [ "async-trait", "biblatex", @@ -591,7 +591,7 @@ dependencies = [ [[package]] name = "doiget-mcp" -version = "0.8.12" +version = "0.8.13" dependencies = [ "anyhow", "assert_cmd", @@ -658,12 +658,13 @@ checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" [[package]] name = "flate2" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" dependencies = [ "crc32fast", "miniz_oxide", + "zlib-rs", ] [[package]] @@ -1298,9 +1299,9 @@ checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" [[package]] name = "miniz_oxide" -version = "0.8.9" +version = "0.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fa76a2c86f704bdb222d66965fb3d63269ce38518b83cb0575fca855ebb6316" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" dependencies = [ "adler2", "simd-adler32", @@ -1516,9 +1517,9 @@ checksum = "a1d01941d82fa2ab50be1e79e6714289dd7cde78eba4c074bc5a4374f650dfe0" [[package]] name = "quick-xml" -version = "0.41.0" +version = "0.42.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" +checksum = "41b1177fdf999d2321d3fb46ff47159d9c1fb9ad66a4879f8c50a0b504615e9b" dependencies = [ "memchr", ] @@ -2624,9 +2625,9 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" [[package]] name = "uuid" -version = "1.25.0" +version = "1.26.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f053576934f05a761a402421fbbe3d425d9366f75f978806a037b3ca481abecc" +checksum = "b5772d71c9be8a8a6ac2117d949c5b224c1b72241bb611d9a3012edcf8af7812" dependencies = [ "getrandom 0.4.2", "js-sys", @@ -3240,6 +3241,12 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "zlib-rs" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" + [[package]] name = "zmij" version = "1.0.21" diff --git a/Cargo.toml b/Cargo.toml index ad99dcf26..489652f8d 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -15,7 +15,7 @@ exclude = [ ] [workspace.package] -version = "0.8.12" +version = "0.8.13" edition = "2021" rust-version = "1.86" license = "MIT" @@ -41,9 +41,9 @@ features-doc-link = "docs/SOURCES.md" [workspace.dependencies] # Core -doiget-core = { path = "crates/doiget-core", version = "0.8.12" } -doiget-cli = { path = "crates/doiget-cli", version = "0.8.12" } -doiget-mcp = { path = "crates/doiget-mcp", version = "0.8.12" } +doiget-core = { path = "crates/doiget-core", version = "0.8.13" } +doiget-cli = { path = "crates/doiget-cli", version = "0.8.13" } +doiget-mcp = { path = "crates/doiget-mcp", version = "0.8.13" } # Async runtime — features are deliberately limited (avoid `full`). tokio = { version = "1", default-features = false, features = [ @@ -156,7 +156,7 @@ url = "2" # response). Default features (no `serde`, no `encoding`) are sufficient # — we hand-roll a small event walker in `sources::arxiv`. Not on # `deny.toml`'s banned list. -quick-xml = { version = "0.41", default-features = false } +quick-xml = { version = "0.42", default-features = false } # BibTeX / BibLaTeX bibliography input parser (ADR-0030 D2). Enabled in # the default feature set — `.bib` is the dominant export format from diff --git a/Formula/doiget.rb b/Formula/doiget.rb new file mode 100644 index 000000000..5492c1e86 --- /dev/null +++ b/Formula/doiget.rb @@ -0,0 +1,44 @@ +# GENERATED by scripts/update-homebrew-formula.sh -- do not edit by hand. +# +# Regenerate with `scripts/update-homebrew-formula.sh ` after a stable +# release has published its assets. That is a MAINTAINER step, not a workflow +# step -- committing the formula back from CI needs a token with write access +# to a protected branch, and #426 says that token is broken. So the formula +# lags the tag until someone runs the script, and CONTRIBUTING.md says so +# (#501). +class Doiget < Formula + desc "Open Access paper fetcher with an MCP server" + homepage "https://github.com/QAtlasHub/doiget" + version "0.8.12" + license "MIT" + + # The published binaries are the release assets themselves, not archives, so + # each URL is fetched verbatim. They are statically linked (musl on Linux), + # which is why there are no dependencies to declare. + on_macos do + on_arm do + url "https://github.com/QAtlasHub/doiget/releases/download/v0.8.12/doiget-macos-aarch64" + sha256 "8aa163dba6c34527ae4ce819835aa138779fdad5041236830ae8833ad18b51cf" + end + on_intel do + url "https://github.com/QAtlasHub/doiget/releases/download/v0.8.12/doiget-macos-x86_64" + sha256 "caabe258798c9b10f53d3f4238e233e8daa3060596b05bed740e591224f10c5f" + end + end + + on_linux do + on_intel do + url "https://github.com/QAtlasHub/doiget/releases/download/v0.8.12/doiget-linux-x86_64" + sha256 "e22c603447b1a2bc9b7c8cf9616085e789334541c6611e368e69aa2333fa61a5" + end + end + + def install + # One asset per platform, whichever was downloaded. + bin.install Dir["doiget-*"].first => "doiget" + end + + test do + assert_match "doiget #{version}", shell_output("#{bin}/doiget --version") + end +end diff --git a/README.md b/README.md index 0d16640ee..0e4b7f49d 100644 --- a/README.md +++ b/README.md @@ -130,6 +130,27 @@ download**, so this works under `--ignore-scripts` and through a corporate registry mirror. `npm view doiget version` tells you what is published; the packages ship from tagged releases, so a very new commit may be ahead of them. +### Homebrew + +```sh +brew tap QAtlasHub/doiget https://github.com/QAtlasHub/doiget +brew install doiget +``` + +The tap lives in this repository rather than a separate `homebrew-doiget`, which +is why the tap line carries an explicit URL. A dedicated tap repo would shorten +it to `brew tap QAtlasHub/doiget`; the formula would move across unchanged. + +`Formula/doiget.rb` installs the **same signed release binary** the GitHub +Release publishes, pinned by the `sha256` from that release's own `.sha256` +asset — the file the shell installer verifies against, so the two channels +cannot disagree about what they installed. It is generated by +`scripts/update-homebrew-formula.sh`, never hand-edited, and CI fails if the +committed formula is not what the generator produces. + +The formula tracks the latest **stable** release, so it can trail a very new +tag by one commit; `brew info doiget` shows which version it pins. + ### Claude Code plugin ``` @@ -138,12 +159,12 @@ packages ship from tagged releases, so a very new commit may be ahead of them. ``` Reads `.claude-plugin/` from this repository's default branch. The plugin's -`.mcp.json` runs `doiget serve`, so install the binary by one of the routes -above first — the plugin installs nothing itself. That is also why this is a -self-hosted marketplace rather than a submission to the Anthropic plugin -directory: a listing whose first run fails for everyone without the binary -already on PATH is worse than no listing. Once npm is live the plugin can call -`npx -y doiget serve` and need nothing installed. +`.mcp.json` runs `npx -y doiget-cli serve`, so it needs **nothing installed +beforehand** — npm fetches the wrapper and the one matching platform binary on +first run. Until this it ran a bare `doiget`, which meant the plugin worked +only for people who had already installed doiget some other way; that is why it +was self-hosted rather than submitted to the Anthropic plugin directory, and +why it is now submittable. ### Channel status @@ -156,8 +177,22 @@ already on PATH is worse than no listing. Once npm is live the plugin can call | MCP Registry | listed | | npm / npx | `doiget-cli` (installs the `doiget` command); see below for what is published | | Claude Code plugin | self-hosted marketplace, as above | -| Homebrew tap, `.deb`, Docker | **not built** | -| Nix | `flake.nix` exists; whether it exposes an installable package rather than a dev shell is unverified | +| Homebrew | `Formula/doiget.rb` in this repo; see above for the tap line | +| Nix | `flake.nix` exposes `packages.default` / `packages.doiget`, not only a dev shell. The outputs exist; `nix profile install` has not been exercised | +| `.deb` | **not built** — low value; most Linux users take the binary or Nix | +| Docker | **not planned** — see below | + +Docker was ranked second in #501 on the grounds that a container is "the only +architecture the Tier-3 features can legally be used in". That premise does not +hold: ADR-0002 decides that the default published binary contains **no TDM +source code at all**, and an image built from the published binary would ship +`oa-only,citation` like every other prebuilt channel. The architecture the +sentence describes is real, and it is served by building from source with +`--features tdm-`, which is not a distribution channel. What remains +is "a shape enterprises can pin and scan" — worth something, but doiget is a +single statically-linked binary, so a container solves no dependency problem +here and the existing `.sha256` plus cosign bundle already give a pinnable, +verifiable artefact. npm was the one channel whose pipeline was written and whose packages did not exist. npm Trusted Publishing cannot perform a package's *first* publish — the diff --git a/crates/doiget-cli/src/commands/batch.rs b/crates/doiget-cli/src/commands/batch.rs index 3c711b7a0..fa9f3f3ab 100644 --- a/crates/doiget-cli/src/commands/batch.rs +++ b/crates/doiget-cli/src/commands/batch.rs @@ -51,6 +51,32 @@ use super::fetch::{ }; use super::resolve_store_root; +/// One entry of the input file, on its way into the dispatch window. +/// +/// `Rejected` carries the verdict the bibliography parser already reached. +/// Before #500's reporting half this was a plain `Vec`: every entry +/// the parser could not turn into a `Ref` became a synthetic placeholder +/// (``, ``) that was handed back to +/// `Ref::parse`, whose only possible answer is `INVALID_REF`. So a PubMed +/// `.bib` record with a valid `pmid` field was reported to the operator as a +/// malformed reference -- the one claim #500 exists to stop doiget making -- +/// and the accompanying message described the placeholder rather than the +/// entry. The verdict travels as a value now, so it cannot be flattened. +enum BatchEntry { + /// A ref string still to be parsed. + Ref(String), + /// Unfetchable, with the reason already decided. + Rejected { + /// What to show in the JSONL `ref` field and the failure digest. + display: String, + /// The closed-set code the operator should see. + code: ErrorCode, + /// The parser's own message, about the entry rather than a + /// placeholder synthesised from it. + message: String, + }, +} + /// Run the `doiget batch ` subcommand. /// /// When `dry_run` is `true` (per ADR-0022 §1 + §3): read the input @@ -116,23 +142,63 @@ pub async fn run_with_options( // error rather than a silently-empty batch. let path_utf8 = Utf8Path::new(&path); let parsed = refs::parse_input(&raw, Format::Auto, Some(path_utf8)); - let mut inputs: Vec = Vec::with_capacity(parsed.len()); + let mut inputs: Vec = Vec::with_capacity(parsed.len()); for entry in parsed { match entry { - Ok(p) => inputs.push(p.ref_.as_input_str().to_string()), - Err(ParseError::InvalidRef { raw, .. }) => inputs.push(raw), - Err(ParseError::NoIdentifier { entry_key }) => { - // Synthesise a recognisable placeholder so Step 7's - // `Ref::parse` rejects this entry as `INVALID_REF` - // with the operator's citation key visible in the - // JSONL `ref` field. A future slice will plumb the - // structured `entry_key` into a dedicated error - // object field (ADR-0030 §6). - let placeholder = match entry_key { + Ok(p) => inputs.push(BatchEntry::Ref(p.ref_.as_input_str().to_string())), + // The raw identifier verbatim, so the downstream `Ref::parse` + // reports `INVALID_REF` against the string the user wrote. That + // IS the right claim here. + Err(ParseError::InvalidRef { raw, .. }) => inputs.push(BatchEntry::Ref(raw)), + // #500. The verdict travels as a value. It used to travel as a + // placeholder string handed back to `Ref::parse`, and `Ref::parse` + // has exactly one thing to say -- `INVALID_REF` -- so this arm and + // the `NoIdentifier` arm below reached the user as the SAME claim + // ("your bibliography is malformed") despite the comment here + // saying they differed, and the message described the synthetic + // placeholder rather than the user's entry. That claim is what + // #500 exists to stop doiget making about an entry that is fine. + Err(err @ ParseError::UnsupportedIdentifier { .. }) => { + let message = err.to_string(); + let ParseError::UnsupportedIdentifier { + kind, + value, + entry_key, + } = err + else { + unreachable!("guarded by the pattern above") + }; + let display = match entry_key { + Some(k) => format!(""), + None => format!(""), + }; + inputs.push(BatchEntry::Rejected { + display, + // The input is valid and the support is absent. The same + // code the MCP `batch_from_bibliography` and CLI `verify` + // surfaces already gave this entry. + code: ErrorCode::NotImplemented, + message, + }); + } + Err(err @ ParseError::NoIdentifier { .. }) => { + let message = err.to_string(); + let ParseError::NoIdentifier { entry_key } = err else { + unreachable!("guarded by the pattern above") + }; + // The operator's citation key stays visible in the JSONL + // `ref` field. A future slice will plumb the structured + // `entry_key` into a dedicated error-object field + // (ADR-0030 §6). + let display = match entry_key { Some(k) => format!(""), None => "".to_string(), }; - inputs.push(placeholder); + inputs.push(BatchEntry::Rejected { + display, + code: ErrorCode::InvalidRef, + message, + }); } Err(ParseError::Decode { format, message }) => { return Err(anyhow!("input did not deserialise as {format}: {message}")); @@ -157,7 +223,11 @@ pub async fn run_with_options( "encountered unknown ParseError variant; batch continues with placeholder \ INVALID_REF — this should never happen on a current doiget-core build" ); - inputs.push(format!("")); + inputs.push(BatchEntry::Rejected { + display: format!(""), + code: ErrorCode::InvalidRef, + message: other.to_string(), + }); } } } @@ -181,7 +251,24 @@ pub async fn run_with_options( if dry_run { let store_root = resolve_store_root()?; let mut parse_errors: usize = 0; - for input in &inputs { + for entry in &inputs { + let input = match entry { + BatchEntry::Ref(input) => input, + BatchEntry::Rejected { + display: shown, + code, + message, + } => { + parse_errors += 1; + tracing::warn!( + input = %shown, + error = %message, + code = code.as_wire(), + "skipping unfetchable batch entry in dry-run mode", + ); + continue; + } + }; match Ref::parse(input) { Ok(ref_) => { let plan = build_fetch_plan(&ref_, &store_root); @@ -244,24 +331,40 @@ pub async fn run_with_options( // Skip delay before the very first fetch; apply between all subsequent ones. let mut is_first_spawn = true; loop { - let window: Vec = remaining.by_ref().take(MCP_BATCH_MAX_SIZE).collect(); + let window: Vec = remaining.by_ref().take(MCP_BATCH_MAX_SIZE).collect(); if window.is_empty() { break; } let mut joins: tokio::task::JoinSet = tokio::task::JoinSet::new(); - for input in window { - let ref_ = match Ref::parse(&input) { - Ok(r) => r, - Err(e) => { + for entry in window { + // Either the parser already reached a verdict about this entry, or + // `Ref::parse` reaches one now. Both land in the SAME reporting + // block below -- which is the point: a verdict that is not + // `INVALID_REF` now survives to the caller instead of being + // laundered through a placeholder string. + let parsed = match entry { + BatchEntry::Ref(input) => match Ref::parse(&input) { + Ok(r) => Ok((input, r)), + Err(e) => Err((input, ErrorCode::InvalidRef, e.to_string())), + }, + BatchEntry::Rejected { + display, + code, + message, + } => Err((display, code, message)), + }; + let (input, ref_) = match parsed { + Ok(pair) => pair, + Err((input, code, message)) => { parse_errors += 1; - failures.push((input.clone(), "INVALID_REF")); + failures.push((input.clone(), code.as_wire())); if json_mode { - // #205: parse failures get an INVALID_REF JSONL line - // with the human message in `error.message`. Per - // ERRORS.md §3.1 there is no `denial_context` on - // INVALID_REF (the input never reached a guard). - emit_jsonl_failure(Some(&input), "INVALID_REF", &e.to_string()); + // #205: an unfetchable entry gets a JSONL line with the + // human message in `error.message`. Per ERRORS.md §3.1 + // there is no `denial_context` on these -- the input + // never reached a guard. + emit_jsonl_failure(Some(&input), code.as_wire(), &message); } // Best-effort `Resolve` row capturing the parse failure; // we do NOT abort the batch on a single bad line. @@ -271,7 +374,7 @@ pub async fn run_with_options( capability: Capability::Oa, ref_: Some(&input), source: None, - error_code: Some("INVALID_REF"), + error_code: Some(code.as_wire()), size_bytes: None, license: None, store_path: None, @@ -282,8 +385,9 @@ pub async fn run_with_options( }); tracing::warn!( %input, - error = %e, - "skipping malformed batch entry", + error = %message, + code = code.as_wire(), + "skipping unfetchable batch entry", ); continue; } @@ -383,7 +487,9 @@ pub async fn run_with_options( // Step 9: SessionEnd, always. Failure to append is best-effort; the // caller already has whatever per-ref errors were observed. - harness.log_session_end(all_ok, None); + // A batch bookend spans many refs, so there is no single terminal + // code to record; the per-ref rows carry those. + harness.log_session_end(all_ok, None, None); // Step 10: stderr summary. ADR-0001: success / progress lines go to // stderr; the workspace `print_stderr` lint is `warn`, promoted to deny diff --git a/crates/doiget-cli/src/commands/config.rs b/crates/doiget-cli/src/commands/config.rs index 10b357e32..76a9e84a8 100644 --- a/crates/doiget-cli/src/commands/config.rs +++ b/crates/doiget-cli/src/commands/config.rs @@ -861,7 +861,12 @@ async fn network_report(cfg: &ResolvedConfig) { ("doaj.org", "https://doaj.org/robots.txt"), ]; for (host, url) in PROBES { - let verdict = if !allow.matches(host) { + // `permits`, not `matches` (#533). This is the fifth adjudication + // site, and the one whose whole job is telling a user why a host was + // refused: with `matches` it would report a DOI resolver as + // `NotAllowlisted` while the real fetch path follows it, which is the + // exact wrong answer #533 was about. + let verdict = if !allow.permits(host) { ProbeVerdict::NotAllowlisted } else { match url::Url::parse(url) { diff --git a/crates/doiget-cli/src/commands/fetch.rs b/crates/doiget-cli/src/commands/fetch.rs index e65c176dd..8aa0d39b3 100644 --- a/crates/doiget-cli/src/commands/fetch.rs +++ b/crates/doiget-cli/src/commands/fetch.rs @@ -233,6 +233,14 @@ pub(crate) fn build_http_client(user_agent: Option<&str>) -> Result // mirroring `DOIGET_ARXIV_BASE`. let ar5iv_base = std::env::var("DOIGET_AR5IV_BASE").ok(); + #[cfg(feature = "tdm-aps")] + let tdm_aps = std::env::var("DOIGET_APS_BASE").ok(); + #[cfg(feature = "tdm-elsevier")] + let tdm_elsevier = std::env::var("DOIGET_ELSEVIER_BASE").ok(); + #[cfg(feature = "tdm-springer")] + let tdm_springer = std::env::var("DOIGET_SPRINGER_BASE").ok(); + #[cfg(feature = "tdm-ieee")] + let tdm_ieee = std::env::var("DOIGET_IEEE_BASE").ok(); if arxiv.is_none() && crossref.is_none() && unpaywall.is_none() @@ -345,8 +353,25 @@ pub(crate) fn build_http_client(user_agent: Option<&str>) -> Result // Test-base mode: build a relaxed client per overridden source. let mut owned: Vec<(String, String)> = Vec::new(); + // Tier-3 test bases, mirroring the MCP builder. Without these a wiremock + // e2e cannot reach the TDM-fetched route on this surface either: the + // override branch's table held only Tier-1/2 keys, so `tdm-aps` was absent + // from the client's map and the attempt died as `no allowlist registered + // for source tdm-aps` -- a harness gap that read like #454 coming back. + // + // Deliberately NOT part of the production-branch test above: setting only + // `DOIGET_APS_BASE` to replay a recorded fixture must not silently switch + // the process to the allow-http test client. for (source, base) in [ ("arxiv", arxiv.as_deref()), + #[cfg(feature = "tdm-aps")] + ("tdm-aps", tdm_aps.as_deref()), + #[cfg(feature = "tdm-elsevier")] + ("tdm-elsevier", tdm_elsevier.as_deref()), + #[cfg(feature = "tdm-springer")] + ("tdm-springer", tdm_springer.as_deref()), + #[cfg(feature = "tdm-ieee")] + ("tdm-ieee", tdm_ieee.as_deref()), ("crossref", crossref.as_deref()), ("unpaywall", unpaywall.as_deref()), ("oa-publisher", oa_publisher.as_deref()), @@ -520,7 +545,15 @@ impl FetchHarness { /// argument; pass `None` for batch sessions. The result is best-effort — /// if this append fails, the caller already has the underlying fetch /// error (if any) and we don't override it. - pub(crate) fn log_session_end(&self, ok: bool, ref_input: Option<&str>) { + /// `error_code` is the terminal code the caller was given, and it is what + /// makes the row answer "what did this session tell the user about this + /// ref?" rather than only "something went wrong" (#507). + pub(crate) fn log_session_end( + &self, + ok: bool, + ref_input: Option<&str>, + error_code: Option<&str>, + ) { let result = if ok { LogResult::Ok } else { LogResult::Err }; let _ = self.log.append(RowInput { event: LogEvent::SessionEnd, @@ -528,7 +561,7 @@ impl FetchHarness { capability: Capability::Oa, ref_: ref_input, source: None, - error_code: None, + error_code, size_bytes: None, license: None, store_path: None, @@ -571,7 +604,9 @@ impl FetchHarness { /// Pulled out so both `run_with_options` and `commands::batch` agree on /// the failure boundary. pub(crate) fn outcome_is_clean_success(outcome: &FetchPaperOutcome) -> bool { - !matches!(outcome.pdf_leg, PdfLegStatus::Blocked { .. }) + // The rule lives in `doiget-core` now, because the MCP surface needs the + // same boundary and had only half of it. + outcome.is_clean_success() } /// CLI-only one-line success message on stderr (ADR-0001 stdio @@ -596,6 +631,15 @@ fn emit_success_line(ref_: &Ref, outcome: &FetchPaperOutcome) { "fetched {} (metadata-only: no OA PDF available) -> {}", label, outcome.path )); + // #505: this is the ONLY outcome that reads as a result rather + // than an error, which is why it had no trace -- there was no + // `error[...]` block to hang one on. It is also the one where the + // absence misleads most: the line above is byte-identical whether + // the optional sources were on and had nothing, or off and never + // asked. + for line in not_found_trace_lines(ref_, &outcome.attempts) { + print_err(format_args!("{line}")); + } } // Issue #325: publisher PDF was blocked, arXiv preprint auto-fetched. PdfLegStatus::PreprintFallback { arxiv_id, .. } => { @@ -663,7 +707,7 @@ fn emit_success_line(ref_: &Ref, outcome: &FetchPaperOutcome) { // paper landed without a second `doiget info` call. Skipped for the // Blocked fail-closed arm (it rendered an `error[CODE]:` line above, not // a success). - if !matches!(outcome.pdf_leg, PdfLegStatus::Blocked { .. }) { + if outcome.is_clean_success() { emit_identity_line(outcome); } } @@ -767,7 +811,19 @@ pub async fn run_with_options( Ok(o) => outcome_is_clean_success(o), Err(_) => false, }; - harness.log_session_end(session_ok, Some(ref_.as_input_str())); + // #507: the code the USER was given, which for this command is not + // always the `Result`'s. A blocked PDF leg is `Ok` with a failed leg and + // an unclean session, and the leg carries the closed-set code -- recording + // `None` there would log the one outcome an agent is most likely to retry + // as having no reason at all. + let session_err = match &result { + Err(e) => Some(doiget_core::ErrorCode::from(e).as_wire()), + Ok(o) => match &o.pdf_leg { + PdfLegStatus::Blocked { code, .. } => Some(code.as_wire()), + _ => None, + }, + }; + harness.log_session_end(session_ok, Some(ref_.as_input_str()), session_err); // Step 6: render the user-facing surface and map to `CliExit`. // The Blocked-PDF reclassification logic that used to live inside @@ -1250,6 +1306,155 @@ fn render_blocked_error( } } +/// The diagnostics for a found-nothing fetch (#505). +/// +/// `no OA PDF available` means only "the sources that ran had nothing". With +/// the default profile that is three of eleven, and the sentence does not say +/// so -- #413 built the trace for exactly this distinction ("we asked and it +/// had nothing" versus "we never asked") and this was the path it never +/// reached. +/// +/// Three blocks: what ran, what did not, and the line to paste. +/// Order the sources that were NOT consulted, for the found-nothing path +/// (#505 part 3). +/// +/// The issue is explicit about the risk, and it governs this whole function: +/// +/// > a ranking that is wrong is worse than no ranking, because it makes people +/// > stop early. So it must be an *ordering* of the full list, never a +/// > shortlist, and it must name the signal it ranked on. +/// +/// Two positions have a real signal and the middle does not, so only two are +/// ranked: +/// +/// * **`openalex` first.** It is categorically different from the rest: it +/// *lists* every location a work has, so with it enabled the answer is "this +/// repository has it", not "this repository might". Item 1 is a lookup; the +/// others are guesses, and the issue is emphatic that presenting both in one +/// list without saying which is which is the failure mode it is about. +/// * **`core` last.** Not a guess either -- its own module doc calls it "the +/// broadest single OA index outside Unpaywall and therefore the LAST fallback +/// in the chain". Broadest means least discriminating, so it is never the +/// first thing to try and never absent from the list. +/// +/// **Everything between them is returned unordered, deliberately.** The issue +/// proposes ranking the middle on venue, author affiliation and funder, and +/// none of those reach this point: `FetchPaperOutcome` carries `title`, +/// `authors` and `year`, and the DOI prefix map is Tier-3-only (ADR-0041, +/// publisher TDM scoping) and absent from an `oa-only` build entirely. Putting +/// them in an order anyway would render a guess in the shape of a finding, +/// which is the one thing this must not do. +fn rank_unconsulted( + attempts: &[SourceAttempt], +) -> (Vec<&'static str>, Vec<&'static str>, Vec<&'static str>) { + let mut first = Vec::new(); + let mut middle = Vec::new(); + let mut last = Vec::new(); + for a in attempts { + if a.outcome.required_env().is_none() { + continue; + } + match a.source { + "openalex" => first.push(a.source), + "core" => last.push(a.source), + other => middle.push(other), + } + } + middle.sort_unstable(); + (first, middle, last) +} + +fn not_found_trace_lines(ref_: &Ref, attempts: &[SourceAttempt]) -> Vec { + let mut out = Vec::new(); + if attempts.is_empty() { + return out; + } + + out.push(" = note: no OA copy found. sources this run:".to_string()); + out.extend( + doiget_core::orchestrator::render_attempts(attempts) + .lines() + .map(|l| format!(" {l}")), + ); + + // Split the widening advice by whether it can actually be acted on. + // + // `resolve_metadata_flag` returns false when the variable IS set but the + // Cargo feature was not compiled in -- it warns through `tracing` and + // moves on, so the source reports `Disabled` naming a variable the user + // has already set. Printing "set DOIGET_ENABLE_X" at someone who set it + // an hour ago is the same species of unhelpful as the bare + // `no OA PDF available` this issue is about, so say which case it is. + let (unset, already_set): (Vec<_>, Vec<_>) = doiget_core::orchestrator::widening_env(attempts) + .into_iter() + .partition(|v| std::env::var_os(v).is_none()); + + if !unset.is_empty() { + // `widening_env` returns Tier-2 switches AND Tier-3 credential pairs. + // Rendering every one as `VAR=1` produced `DOIGET_KEY_APS=1` -- an API + // key that can never be valid, in a line whose whole purpose is to be + // pasted. A flag is a flag; a key is a key. + let assignments = unset + .iter() + .map(|v| { + if v.starts_with("DOIGET_KEY_") { + format!("{v}=") + } else { + format!("{v}=1") + } + }) + .collect::>() + .join(" "); + let target = match ref_ { + Ref::Arxiv(id) => id.as_str().to_string(), + Ref::Doi(doi) => doi.as_str().to_string(), + }; + out.push(" = suggest: to widen the search:".to_string()); + out.push(format!(" {assignments} doiget fetch {target}")); + } + + if !already_set.is_empty() { + out.push(format!( + " = note: {} already set, but the source is still off -- this binary was built without the Cargo feature that provides it. Widening needs a differently-built binary, not another variable.", + already_set.join(", ") + )); + } + + // #505 part 3. Ordered only where there is something to order on; see + // `rank_unconsulted`. + let (first, middle, last) = rank_unconsulted(attempts); + if !first.is_empty() || !middle.is_empty() || !last.is_empty() { + out.push(" = note: of the sources not consulted:".to_string()); + for s in &first { + out.push(format!( + " 1. {s:<12} lists every location a work has -- a lookup, not a guess" + )); + } + if !middle.is_empty() { + out.push(format!( + " then, in NO particular order: {}", + middle.join(" ") + )); + } + for s in &last { + out.push(format!( + " last: {s:<9} the broadest index outside Unpaywall, so never the first try" + )); + } + // Naming the signal is half of what the ranking is for. Saying "the + // middle has none" is the honest form of that, and it stops the list + // reading as an ordering it is not. + if !middle.is_empty() { + out.push( + " = note: the middle is unordered because nothing in this run distinguishes those sources -- venue, affiliation and funder would, and none of them reach here. An invented order would read as information." + .to_string(), + ); + } + } + + out +} + /// The `= note:`/`= suggest:` block appended to a blocked PDF leg (#445). /// /// #413 attached the resolution trace to `NotFound` only. But "found @@ -2225,6 +2430,211 @@ host = "*.uj.edu.pl" /// The half of #445 that the #413 trace already answered for /// `NotFound`: *did anything else have it?* + /// #505: the found-nothing path is the one outcome that reads as a + /// result, so its silence is the most misleading. `no OA PDF available` + /// is byte-identical whether the optional sources were on and had + /// nothing or off and never asked. + #[test] + fn a_found_nothing_fetch_says_what_it_consulted_and_what_it_did_not() { + use doiget_core::orchestrator::{AttemptOutcome, SourceAttempt}; + let ref_ = Ref::parse("10.1137/0117004").expect("valid doi"); + let attempts = vec![ + SourceAttempt::new("unpaywall", AttemptOutcome::NoRecord), + SourceAttempt::new( + "hal", + AttemptOutcome::Disabled { + env: &["DOIGET_ENABLE_HAL"], + }, + ), + ]; + let joined = not_found_trace_lines(&ref_, &attempts).join( + " +", + ); + + assert!( + joined.contains("unpaywall") && joined.contains("no record"), + "what ran, and what it said: +{joined}" + ); + assert!( + joined.contains("DOIGET_ENABLE_HAL"), + "a source never asked must still name its switch: +{joined}" + ); + // The line to paste, not prose about it. + assert!( + joined.contains("DOIGET_ENABLE_HAL=1 doiget fetch 10.1137/0117004"), + "the widening command must be runnable as printed: +{joined}" + ); + } + + /// #505 part 3, and the property the issue cares about most: the ranking + /// is an ORDERING OF THE FULL LIST, never a shortlist. + /// + /// > a ranking that is wrong is worse than no ranking, because it makes + /// > people stop early. + /// + /// A source that is dropped from the list is a source the reader will not + /// try, so every unconsulted source must appear somewhere. + #[test] + fn the_ranking_lists_every_unconsulted_source_and_drops_none() { + use doiget_core::orchestrator::{AttemptOutcome, SourceAttempt}; + let disabled = |name: &'static str, env: &'static [&'static str]| { + SourceAttempt::new(name, AttemptOutcome::Disabled { env }) + }; + let attempts = vec![ + SourceAttempt::new("crossref", AttemptOutcome::NoRecord), + disabled("core", &["DOIGET_ENABLE_CORE"]), + disabled("openalex", &["DOIGET_ENABLE_OPENALEX"]), + disabled("hal", &["DOIGET_ENABLE_HAL"]), + disabled("europe-pmc", &["DOIGET_ENABLE_EUROPE_PMC"]), + ]; + + let (first, middle, last) = rank_unconsulted(&attempts); + let mut all: Vec<&str> = first + .iter() + .chain(middle.iter()) + .chain(last.iter()) + .copied() + .collect(); + all.sort_unstable(); + assert_eq!( + all, + vec!["core", "europe-pmc", "hal", "openalex"], + "every source that was not consulted must appear, and only those" + ); + + // A consulted source contributes nothing: it already answered. + assert!(!all.contains(&"crossref")); + + // The two positions that HAVE a signal. + assert_eq!(first, vec!["openalex"], "the lookup goes first"); + assert_eq!(last, vec!["core"], "the broadest index goes last"); + assert_eq!(middle, vec!["europe-pmc", "hal"]); + } + + /// The rendered form must mark item 1 as categorically different and must + /// say the middle is unordered. Presenting a lookup and a guess in one + /// list without saying which is which is the failure mode #505 is about. + #[test] + fn the_rendered_ranking_says_which_part_is_a_guess() { + use doiget_core::orchestrator::{AttemptOutcome, SourceAttempt}; + let ref_ = Ref::parse("10.1137/0117004").expect("valid doi"); + let attempts = vec![ + SourceAttempt::new("crossref", AttemptOutcome::NoRecord), + SourceAttempt::new( + "openalex", + AttemptOutcome::Disabled { + env: &["DOIGET_ENABLE_OPENALEX"], + }, + ), + SourceAttempt::new( + "hal", + AttemptOutcome::Disabled { + env: &["DOIGET_ENABLE_HAL"], + }, + ), + SourceAttempt::new( + "core", + AttemptOutcome::Disabled { + env: &["DOIGET_ENABLE_CORE"], + }, + ), + ]; + let joined = not_found_trace_lines(&ref_, &attempts).join( + " +", + ); + + assert!( + joined.contains("a lookup, not a guess"), + "item 1 must be marked as categorically different: +{joined}" + ); + assert!( + joined.contains("NO particular order"), + "the middle must not read as an ordering: +{joined}" + ); + assert!( + joined.contains("An invented order would read as information"), + "and it must say WHY there is no order, which is the named signal: +{joined}" + ); + assert!( + joined.contains("never the first try"), + "core's position must carry its own reason: +{joined}" + ); + } + + /// Nothing to rank when nothing was skipped, and the common path gains no + /// noise from a feature about the uncommon one. + #[test] + fn a_run_that_skipped_nothing_gets_no_ranking() { + use doiget_core::orchestrator::{AttemptOutcome, SourceAttempt}; + let ref_ = Ref::parse("10.1137/0117004").expect("valid doi"); + let attempts = vec![SourceAttempt::new("crossref", AttemptOutcome::NoRecord)]; + let joined = not_found_trace_lines(&ref_, &attempts).join( + " +", + ); + assert!( + !joined.contains("not consulted:"), + "no skipped sources means no ranking block: +{joined}" + ); + } + + /// No trace at all when there is nothing to say. An empty attempt list + /// means the chain never recorded anything, and inventing a block for it + /// would be noise on the one path users see most. + #[test] + fn no_attempts_means_no_found_nothing_trace() { + let ref_ = Ref::parse("10.1137/0117004").expect("valid doi"); + assert!(not_found_trace_lines(&ref_, &[]).is_empty()); + } + + /// The advice has to be actionable to be worth printing. + /// + /// `resolve_metadata_flag` returns false when the variable is SET but the + /// Cargo feature was not compiled in, so the source still reports + /// `Disabled` naming a variable the user already set. Telling them to set + /// it again is the same species of unhelpful as the bare + /// `no OA PDF available` this issue is about. + #[test] + #[serial] + fn an_already_set_switch_is_reported_as_a_build_problem_not_a_config_one() { + use doiget_core::orchestrator::{AttemptOutcome, SourceAttempt}; + let _guard = EnvGuard::save("DOIGET_ENABLE_HAL"); + std::env::set_var("DOIGET_ENABLE_HAL", "1"); + + let ref_ = Ref::parse("10.1137/0117004").expect("valid doi"); + let attempts = vec![SourceAttempt::new( + "hal", + AttemptOutcome::Disabled { + env: &["DOIGET_ENABLE_HAL"], + }, + )]; + let joined = not_found_trace_lines(&ref_, &attempts).join( + " +", + ); + + assert!( + !joined.contains("doiget fetch 10.1137/0117004"), + "must NOT tell them to set what they have already set: +{joined}" + ); + assert!( + joined.contains("built without"), + "must name the real blocker, which is the build: +{joined}" + ); + } + #[test] fn a_blocked_leg_reports_which_other_sources_were_consulted() { use doiget_core::orchestrator::{AttemptOutcome, SourceAttempt}; diff --git a/crates/doiget-cli/src/commands/frontier.rs b/crates/doiget-cli/src/commands/frontier.rs index 394c2631c..9cdbea6f5 100644 --- a/crates/doiget-cli/src/commands/frontier.rs +++ b/crates/doiget-cli/src/commands/frontier.rs @@ -81,7 +81,11 @@ pub async fn run( }; let outcome = frontier_view(&query, &base, &contact_email, &ctx).await; - harness.log_session_end(outcome.is_ok(), Some(&doi_str)); + harness.log_session_end( + outcome.is_ok(), + Some(&doi_str), + outcome.as_ref().err().map(|e| ErrorCode::from(e).as_wire()), + ); let mut results = match outcome { Ok(r) => r, diff --git a/crates/doiget-cli/src/commands/graph.rs b/crates/doiget-cli/src/commands/graph.rs index 5219e46f9..b876c8da2 100644 --- a/crates/doiget-cli/src/commands/graph.rs +++ b/crates/doiget-cli/src/commands/graph.rs @@ -36,6 +36,24 @@ use doiget_core::{ErrorCode, Ref}; use super::fetch::{cli_exit_code, CliExit, FetchHarness}; use super::output::print_err; +/// The closed-set code a `GraphError` is reported as (#507). +/// +/// The `map_err` below already decides this per variant, but it does so while +/// building a user-facing message and an exit code, so the value is not +/// available to the provenance bookend that runs first. +/// +/// This used to hold its own exhaustive match, above a claim that "the two +/// cannot say different things about the same error". Two is the wrong count: +/// the MCP surface held a third copy, which flattened `Source(_)` to +/// `NETWORK_ERROR`, so an OpenAlex 404 was terminal here and retriable there. +/// The mapping lives beside the enum in `doiget-core` now. +fn graph_error_code(e: &GraphError) -> ErrorCode { + // The mapping lives in `doiget_core::citation_graph`, next to the enum, so + // this surface and the MCP one cannot drift apart again -- they did, on + // `Source(_)`, which the MCP copy flattened to `NETWORK_ERROR`. + ErrorCode::from(e) +} + /// Run the `graph` subcommand against the live source set. /// /// `input` is the user-supplied ref string (DOI only — arXiv ids are @@ -121,7 +139,13 @@ pub async fn run( let outcome = expand(&doi, caps, &source, &harness.profile, &ctx).await; let session_ok = outcome.is_ok(); - harness.log_session_end(session_ok, Some(&input)); + // `GraphError` has its own mapping to the closed set, applied just + // below; the bookend records the same code the caller is given (#507). + let session_err = outcome + .as_ref() + .err() + .map(|e| graph_error_code(e).as_wire()); + harness.log_session_end(session_ok, Some(&input), session_err); let graph = outcome.map_err(|e| match e { GraphError::CapabilityDenied => { diff --git a/crates/doiget-cli/src/commands/mod.rs b/crates/doiget-cli/src/commands/mod.rs index 6b082d099..9b77cd969 100644 --- a/crates/doiget-cli/src/commands/mod.rs +++ b/crates/doiget-cli/src/commands/mod.rs @@ -231,7 +231,7 @@ fn store_root_from_config() -> Option { tracing::warn!( path = %path, error = %e, - "config.toml could not be read; [store] root ignored and the default store root used instead. Run `doiget config doctor` to see which root is in effect." + "config.toml could not be read; [store] root ignored and the default store root used instead. Run `doiget config doctor` to see which root is in effect." ); return None; } diff --git a/crates/doiget-cli/src/commands/search.rs b/crates/doiget-cli/src/commands/search.rs index 60b19a4d3..1d1aa6240 100644 --- a/crates/doiget-cli/src/commands/search.rs +++ b/crates/doiget-cli/src/commands/search.rs @@ -228,7 +228,11 @@ async fn run_external( let ctx = harness.fetch_context(); let outcome = paper_search(&base, &contact_email, &q, &ctx).await; - harness.log_session_end(outcome.is_ok(), Some(query)); + harness.log_session_end( + outcome.is_ok(), + Some(query), + outcome.as_ref().err().map(|e| ErrorCode::from(e).as_wire()), + ); let results = match outcome { Ok(r) => r, @@ -256,6 +260,14 @@ async fn run_external( // year / OA / DOI / title. Tab-separated, `cut(1)`-compatible. writeln!(out, "cited_by\tyear\toa\tdoi\ttitle") .context("failed to write search header to stdout")?; + // #534, human half: a header with no rows under it says "not indexed" just + // as flatly as the JSON envelope did. The note goes to stderr so the table + // on stdout stays `cut(1)`-clean (ADR-0001). + if results.results.is_empty() { + if let Some(hint) = doiget_core::discovery::zero_result_hint(query) { + print_err(format_args!(" = note: {hint}")); + } + } for hit in &results.results { let year = dash_or(hit.year); let oa = hit.oa_status.as_deref().unwrap_or("-"); @@ -295,14 +307,25 @@ fn local_envelope(query: &str, entries: &[EntryInfo]) -> serde_json::Value { /// `{ ok, scope, query, total_results, count, results }`. Extracted as a pure /// function so the wire shape is unit-testable without capturing stdout. fn external_envelope(query: &str, results: &PaperSearchResults) -> serde_json::Value { - serde_json::json!({ + let mut envelope = serde_json::json!({ "ok": true, "scope": "external", "query": query, "total_results": results.total_results, "count": results.results.len(), "results": results.results, - }) + }); + // #534. `{"ok": true, "total_results": 0}` is a success envelope, so + // nothing in the error machinery reaches it, and a script or agent reading + // it takes zero results as a fact about the literature and stops. The fix + // first landed on the MCP tool only -- this surface went on emitting the + // exact envelope the issue was filed about, byte for byte. + if results.results.is_empty() { + if let Some(hint) = doiget_core::discovery::zero_result_hint(query) { + envelope["hint"] = serde_json::json!(hint); + } + } + envelope } /// Pretty-serialize a JSON value and write it as one line to `out`. Shared @@ -357,6 +380,52 @@ mod tests { assert_eq!(v["results"][0]["abstract"], "abs"); } + /// #534 on THIS surface. The fix landed on the MCP tool first, so + /// `doiget search --mode json` went on emitting the exact envelope the + /// issue was filed about -- `ok: true`, `total_results: 0`, nothing to + /// tell a script that the query, not the literature, was the problem. + /// + /// Driven through the real `external_envelope`, the function `--mode json` + /// prints. + #[test] + fn a_long_query_that_matched_nothing_carries_the_hint() { + let results = PaperSearchResults { + results: vec![], + total_results: Some(0), + }; + let q = "lithium refractoriness after discontinuation kindling sensitization course of illness Post"; + let v = external_envelope(q, &results); + assert_eq!(v["ok"], true, "still a success envelope: {v}"); + assert_eq!(v["count"], 0); + let hint = v["hint"].as_str().unwrap_or_default(); + assert!(hint.contains("10 terms"), "names the count: {hint:?}"); + assert!(hint.contains("3-5"), "says what to do instead: {hint:?}"); + } + + /// A short query matching nothing may genuinely mean nothing is indexed, + /// and a hint on every empty result would train readers to skip it. + #[test] + fn a_short_query_that_matched_nothing_is_left_alone() { + let results = PaperSearchResults { + results: vec![], + total_results: Some(0), + }; + let v = external_envelope("depersonalization derealization", &results); + assert!(v.get("hint").is_none(), "no hint on a short query: {v}"); + } + + /// And a query that DID match carries no hint, however long it is. + #[test] + fn a_long_query_with_results_carries_no_hint() { + let results = PaperSearchResults { + results: vec![hit()], + total_results: Some(1), + }; + let q = "a b c d e f g h i j k l"; + let v = external_envelope(q, &results); + assert!(v.get("hint").is_none(), "results present: {v}"); + } + #[test] fn sort_arg_lowers_to_core() { // Relevance is the only sort (#290); `cited` / `recent` were removed. diff --git a/crates/doiget-cli/src/commands/source.rs b/crates/doiget-cli/src/commands/source.rs index 3f25f4180..d98a0a21e 100644 --- a/crates/doiget-cli/src/commands/source.rs +++ b/crates/doiget-cli/src/commands/source.rs @@ -52,7 +52,14 @@ pub async fn run( let id: ArxivId = match parsed { Ref::Arxiv(a) => a, Ref::Doi(_) => { - let code = ErrorCode::NoOaAvailable; + // `NOT_IMPLEMENTED`, not `NO_OA_AVAILABLE`. The latter carries + // disposition `needs_config` -- "a named change makes it" -- and + // there is no knob: this command is arXiv-only and DOI-to-arXiv + // linking is not built (#281 item 5). Kept identical to the MCP + // sibling; changing one surface and not the other is the defect + // this release is about, and the first pass at this fix did + // exactly that. + let code = ErrorCode::NotImplemented; print_err(format_args!( "error[{}]: no source bundle for a bare DOI — if an arXiv preprint exists, \ pass its id (e.g. `doiget source arxiv:2401.12345 --out ./src`)", diff --git a/crates/doiget-cli/src/commands/tag.rs b/crates/doiget-cli/src/commands/tag.rs index 230f7628c..b9409263c 100644 --- a/crates/doiget-cli/src/commands/tag.rs +++ b/crates/doiget-cli/src/commands/tag.rs @@ -122,7 +122,7 @@ pub fn run( } store - .write(&safekey, &metadata, None) + .write_user_authored(&safekey, &metadata, None) .with_context(|| format!("failed to write updated metadata for {ref_str}"))?; Ok(()) @@ -174,7 +174,7 @@ pub fn run_annotate(ref_str: String, text: Option, clear: bool) -> Resul } store - .write(&safekey, &metadata, None) + .write_user_authored(&safekey, &metadata, None) .with_context(|| format!("failed to write updated metadata for {ref_str}"))?; Ok(()) diff --git a/crates/doiget-cli/src/commands/tex_source.rs b/crates/doiget-cli/src/commands/tex_source.rs index 39e7576d8..4cb2ca89a 100644 --- a/crates/doiget-cli/src/commands/tex_source.rs +++ b/crates/doiget-cli/src/commands/tex_source.rs @@ -42,7 +42,14 @@ pub async fn run( let id: ArxivId = match parsed { Ref::Arxiv(a) => a, Ref::Doi(_) => { - let code = ErrorCode::NoOaAvailable; + // `NOT_IMPLEMENTED`, not `NO_OA_AVAILABLE`. The latter carries + // disposition `needs_config` -- "a named change makes it" -- and + // there is no knob: this command is arXiv-only and DOI-to-arXiv + // linking is not built (#281 item 5). Kept identical to the MCP + // sibling; changing one surface and not the other is the defect + // this release is about, and the first pass at this fix did + // exactly that. + let code = ErrorCode::NotImplemented; print_err(format_args!( "error[{}]: no TeX source for a bare DOI — if an arXiv preprint exists, \ pass its id (e.g. `doiget tex-source arxiv:2401.12345`)", diff --git a/crates/doiget-cli/src/commands/text.rs b/crates/doiget-cli/src/commands/text.rs index c7c81a9f6..911bdae1a 100644 --- a/crates/doiget-cli/src/commands/text.rs +++ b/crates/doiget-cli/src/commands/text.rs @@ -52,7 +52,14 @@ pub async fn run( // No full-text HTML source for a bare DOI in PR4; DOI→arXiv // linking is #281 item 5 (ADR-0032 D5). Report honestly rather // than silently failing. - let code = ErrorCode::NoOaAvailable; + // `NOT_IMPLEMENTED`, not `NO_OA_AVAILABLE`. The latter carries + // disposition `needs_config` -- "a named change makes it" -- and + // there is no knob: this command is arXiv-only and DOI-to-arXiv + // linking is not built (#281 item 5). Kept identical to the MCP + // sibling; changing one surface and not the other is the defect + // this release is about, and the first pass at this fix did + // exactly that. + let code = ErrorCode::NotImplemented; print_err(format_args!( "error[{}]: no full-text source for a DOI — if an arXiv preprint exists, \ pass its id (e.g. `doiget text arxiv:2401.12345`)", diff --git a/crates/doiget-cli/src/commands/verify.rs b/crates/doiget-cli/src/commands/verify.rs index cd6e7d8ca..2b1fc727f 100644 --- a/crates/doiget-cli/src/commands/verify.rs +++ b/crates/doiget-cli/src/commands/verify.rs @@ -40,6 +40,7 @@ use doiget_core::orchestrator::resolve_only; use doiget_core::refs::{parse_input, Format, ParseError}; use doiget_core::verify_config::{self, OnMissingId}; use doiget_core::CapabilityProfile; +use doiget_core::ErrorCode; use super::fetch::CliExit; use super::output::OutputMode; @@ -264,7 +265,28 @@ pub async fn run(path: String, format: String, cli_strict: bool, mode: OutputMod "ref": raw, "status": VerifyStatus::Illegal.as_wire(), "entry_key": entry_key, - "error": { "code": "INVALID_REF", "message": source.to_string() }, + "error": { "code": ErrorCode::InvalidRef.as_wire(), "message": source.to_string() }, + }), + ), + // #500: still unverifiable, but for a reason the reader can act on + // -- and the action is not "fix the bibliography". + Err(ParseError::UnsupportedIdentifier { + kind, + value, + entry_key, + }) => ( + VerifyStatus::Unverifiable, + serde_json::json!({ + "ok": false, + "ref": serde_json::Value::Null, + "status": VerifyStatus::Unverifiable.as_wire(), + "entry_key": entry_key, + "error": { + "code": ErrorCode::NotImplemented.as_wire(), + "message": doiget_core::refs::unsupported_identifier_claim( + kind, &value, + ), + }, }), ), Err(ParseError::NoIdentifier { entry_key }) => ( @@ -274,7 +296,7 @@ pub async fn run(path: String, format: String, cli_strict: bool, mode: OutputMod "ref": serde_json::Value::Null, "status": VerifyStatus::Unverifiable.as_wire(), "entry_key": entry_key, - "error": { "code": "INVALID_REF", "message": "entry has no DOI / arXiv id" }, + "error": { "code": ErrorCode::InvalidRef.as_wire(), "message": "entry has no DOI / arXiv id" }, }), ), Err(ParseError::Decode { format, message }) => ( @@ -283,7 +305,7 @@ pub async fn run(path: String, format: String, cli_strict: bool, mode: OutputMod "ok": false, "status": VerifyStatus::Illegal.as_wire(), "error": { - "code": "INVALID_REF", + "code": ErrorCode::InvalidRef.as_wire(), "message": format!("input did not parse as {format}: {message}"), }, }), diff --git a/crates/doiget-cli/tests/batch_e2e.rs b/crates/doiget-cli/tests/batch_e2e.rs index a0ef9ee0f..6b52e8d1e 100644 --- a/crates/doiget-cli/tests/batch_e2e.rs +++ b/crates/doiget-cli/tests/batch_e2e.rs @@ -315,6 +315,7 @@ async fn batch_with_malformed_ref_continues_and_returns_err() { #[tokio::test] #[serial] +#[ignore = "612 s at arXiv's published 3 s/request rate; run by the `test (slow)` CI job via --ignored"] async fn batch_above_window_size_fetches_every_ref() { // Issue #304: a refs file LARGER than `MCP_BATCH_MAX_SIZE` must no longer // abort fetching nothing — every ref is processed across multiple @@ -323,9 +324,23 @@ async fn batch_above_window_size_fetches_every_ref() { // just the windowing arithmetic), so it confirms counters accumulate // across windows and the bookends are emitted exactly once. // - // Slow by construction: `MCP_BATCH_MAX_SIZE + 2` real fetches through the - // hard-coded 5-per-second rate cap (~20 s). Kept `#[serial]` so it never - // contends with the other batch fixtures. + // SLOW BY CONSTRUCTION, and far slower than this comment used to claim. + // It said "~20 s", counting only the global 5-per-second cap. Two things + // it missed: + // + // * `SOURCE_RATE_OVERRIDES` puts 3 s between arXiv requests, because + // arXiv's Terms of Use do (2cc32ab, 2026-08-25 -- after this test was + // written, which is why the estimate was right when it was made); + // * one arXiv attempt issues TWO requests, the Atom feed then the PDF, + // and #493 paced the second one too. + // + // So the real cost is 102 * 2 * 3 s = 612 s, and CI measured 609 s. That + // is not waste -- it is arXiv's published rate, faithfully obeyed -- but + // it is a property of the dispatch loop, which varies with neither OS nor + // Cargo feature. `#[ignore]` keeps it off the five broad `test` jobs; the + // `test (slow)` job runs it once with `--ignored`. + // + // Kept `#[serial]` so it never contends with the other batch fixtures. let server = MockServer::start().await; let body = b"%PDF-1.7\n%batch-window-fixture\n".to_vec(); // One regex mock serves every generated arXiv id rather than mounting one diff --git a/crates/doiget-cli/tests/batch_jsonl_e2e.rs b/crates/doiget-cli/tests/batch_jsonl_e2e.rs index 87fede5b6..ed140117e 100644 --- a/crates/doiget-cli/tests/batch_jsonl_e2e.rs +++ b/crates/doiget-cli/tests/batch_jsonl_e2e.rs @@ -366,3 +366,61 @@ fn batch_failure_digest_includes_parse_errors() { "digest must list the parse failure: {stderr}" ); } + +/// #500 on the CLI `batch` surface. +/// +/// A PubMed-exported `.bib` record carries `pmid = {9659853}` and no DOI. The +/// entry is fine; doiget cannot resolve that identifier class yet. Reporting +/// `INVALID_REF` sends the user to edit a bibliography that is correct, which +/// is the one claim #500 exists to stop doiget making. +/// +/// It reported `INVALID_REF` anyway, on this surface only: the parser's +/// verdict was flattened into a placeholder string and handed back to +/// `Ref::parse`, whose only possible answer is `INVALID_REF`. The MCP tool and +/// `doiget verify` had said `NOT_IMPLEMENTED` all along. +#[test] +fn batch_json_pmid_only_entry_is_not_implemented_not_invalid_ref() { + let dir = TempDir::new().expect("tempdir"); + let bib = dir.path().join("refs.bib"); + { + let mut f = std::fs::File::create(&bib).expect("create bib file"); + f.write_all( + b"@article{Smith2020, + title = {A PubMed-only record}, + pmid = {9659853}, +} +", + ) + .expect("write bib"); + } + + let out = doiget(&dir) + .args(["batch", bib.to_str().expect("utf-8"), "--mode", "json"]) + .output() + .expect("run batch"); + + let stdout = String::from_utf8(out.stdout).expect("utf-8 stdout"); + let line = stdout + .lines() + .find(|l| l.contains("\"ok\"")) + .unwrap_or_else(|| panic!("no JSONL record in: {stdout}")); + let v: Value = serde_json::from_str(line).expect("JSONL record parses"); + + assert_eq!(v["ok"], serde_json::json!(false), "record: {v}"); + assert_eq!( + v["error"]["code"], + serde_json::json!("NOT_IMPLEMENTED"), + "a PMID entry is unsupported, not malformed: {v}" + ); + // The message must describe the user's entry, not the synthetic + // placeholder the old code round-tripped through `Ref::parse`. + let message = v["error"]["message"].as_str().unwrap_or_default(); + assert!( + message.contains("PMID") && message.contains("9659853"), + "message names the identifier the entry actually carries: {message:?}" + ); + assert!( + !message.contains("` and the MCP surface flattened to `NETWORK_ERROR` -- so +/// an OpenAlex 404 during expansion was `NOT_FOUND` (terminal) on one surface +/// and `NETWORK_ERROR` (retry_after) on the other, telling an agent to retry a +/// seed that will never resolve. Defined here, in the crate that owns the +/// error, so the wildcard `#[non_exhaustive]` forces on downstream crates +/// cannot silently absorb a new variant again. +impl From<&GraphError> for crate::ErrorCode { + fn from(e: &GraphError) -> crate::ErrorCode { + match e { + GraphError::CapabilityDenied => crate::ErrorCode::CapabilityDenied, + // An indexing gap upstream: the seed is a valid DOI that OpenAlex + // does not hold. + GraphError::SeedNotIndexed => crate::ErrorCode::NoOaAvailable, + // Already owns an exhaustive mapping; reuse it rather than + // restate it. + GraphError::Source(fe) => crate::ErrorCode::from(fe), + GraphError::Log(_) => crate::ErrorCode::LogError, + } + } +} + /// Expand the citation graph for `seed_doi` via OpenAlex. pub async fn expand( seed_doi: &Doi, diff --git a/crates/doiget-core/src/discovery.rs b/crates/doiget-core/src/discovery.rs index 19981878e..7817d23a3 100644 --- a/crates/doiget-core/src/discovery.rs +++ b/crates/doiget-core/src/discovery.rs @@ -1231,9 +1231,73 @@ pub async fn frontier_view( // Tests // --------------------------------------------------------------------------- +/// Advice to attach to a paper search that matched nothing (#534). +/// +/// OpenAlex free-text matching degrades sharply as a query lengthens: past +/// roughly eight terms it returns nothing at all rather than a partial match. +/// A human reading `0 results` shortens the query and tries again. An agent +/// reading `ok: true` with an empty array reads it as a fact about the world +/// and stops -- in the session that produced #534, eleven consecutive searches +/// returned zero for papers a three-to-five term query then found immediately, +/// and a known study was written off as unavailable. +/// +/// Returns `None` for short queries: a zero-result two-term search really may +/// mean the work is not indexed, and a hint on every empty result would train +/// readers to skip it. +/// +/// Lives here rather than on either front end because both surfaces call +/// [`paper_search`] and both can return the empty envelope. The first fix +/// landed only on the MCP tool, so `doiget search --mode json` went on +/// emitting the exact envelope #534 was filed about. +#[must_use] +pub fn zero_result_hint(query: &str) -> Option { + // Where OpenAlex free-text matching starts failing outright. Not a hard + // boundary -- a bound observed from the queries in #534, which is why the + // wording says "roughly". + const DEGRADES_PAST: usize = 8; + + let terms = query.split_whitespace().count(); + if terms <= DEGRADES_PAST { + return None; + } + Some(format!( + "This query has {terms} terms. OpenAlex free-text matching degrades sharply past roughly {DEGRADES_PAST} and returns nothing rather than a partial match, so zero results here is more likely to be about the query than about the literature. Retry with 3-5 distinctive terms - an author surname, a coined phrase, the distinguishing noun - before concluding the work is not indexed." + )) +} + #[cfg(test)] #[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)] mod tests { + /// #534: a long query matching nothing is far more likely to be a + /// query-length problem than an absent literature, and the agent reading + /// the envelope is the one who cannot tell the difference. + #[test] + fn a_long_zero_result_query_is_told_why_it_may_be_zero() { + let q = "lithium refractoriness after discontinuation kindling sensitization course of illness Post"; + let hint = zero_result_hint(q).expect("10 terms is past the threshold"); + assert!(hint.contains("10 terms"), "names the count: {hint}"); + assert!( + hint.contains("3-5"), + "says what to do instead, not only what went wrong: {hint}" + ); + } + + /// A short query returning nothing may genuinely mean nothing is indexed. + /// Hinting on every empty result would teach readers to skip the hint. + #[test] + fn a_short_zero_result_query_is_left_alone() { + assert!(zero_result_hint("depersonalization derealization").is_none()); + assert!(zero_result_hint("").is_none()); + } + + /// The threshold counts terms, not characters: one very long term is still + /// one term, and "shorten it" is not the advice to give. + #[test] + fn the_threshold_counts_terms_not_length() { + assert!(zero_result_hint(&"a".repeat(400)).is_none()); + assert!(zero_result_hint("a b c d e f g h i").is_some()); + } + use super::*; use std::sync::Arc; diff --git a/crates/doiget-core/src/http.rs b/crates/doiget-core/src/http.rs index 18809e3f6..9586bb8d8 100644 --- a/crates/doiget-core/src/http.rs +++ b/crates/doiget-core/src/http.rs @@ -173,6 +173,94 @@ impl SourceAllowlist { .iter() .any(|pat| host_matches_pattern(&host_lc, pat)) } + + /// Returns `true` if a request to `host` may proceed under this + /// allowlist: either it is on the list, or it is a transparent DOI + /// resolver ([`is_transparent_resolver`]). + /// + /// **This, not [`matches`](Self::matches), is what an adjudication site + /// calls.** `matches` answers "is this host on the list", which is a + /// question about the list; `permits` answers "may we go here", which is + /// the question every gate is actually asking. Keeping them separate + /// means the resolver set is not silently reported as part of any + /// source's `expected_hosts`. + /// + /// The five adjudication sites -- the pre-fetch OA-URL check in + /// `orchestrator`, the two redirect-policy closures, `probe`, and the + /// pre-check `doiget config doctor --network` runs before calling + /// `probe` -- all route through here so they cannot disagree. #533 was + /// found because only one of them was ever walked end to end, and the + /// doctor's was missed on the first pass at this very doc comment: it + /// said "four" while still calling `matches`, which would have had the + /// command that explains allowlist refusals reproduce #533 the moment a + /// resolver host entered its probe list. + #[must_use] + pub fn permits(&self, host: &str) -> bool { + is_transparent_resolver(host) || self.matches(host) + } +} + +/// Hosts that are addressing, not hosting. +/// +/// `doi.org` is the indirection layer every DOI passes through, and +/// Unpaywall routinely reports it AS the OA location: for the gold cc-by +/// paper in #533, `best_oa_location.url` is literally +/// `https://doi.org/10.1002/pcn5.205` with no `url_for_pdf`. Adjudicating +/// that hop as if it were a content host had two consequences, both wrong: +/// +/// * the chain was refused at the FIRST hop, before it ever reached the +/// publisher -- whose host, `*.wiley.com`, was already on the list; and +/// * the denial's remediation told the user to allowlist `doi.org`, which +/// does not widen the trusted surface toward one publisher. It removes +/// the bound entirely, because every DOI in existence resolves through +/// it. An agent following that advice would get the PDF and silently +/// lose the invariant the allowlist exists to hold (ADR-0027). +/// +/// These hosts are therefore FOLLOWED but never allowlisted, never named as +/// remediation, and never counted as the source of the content. The host +/// that actually serves the bytes is adjudicated exactly as before, so this +/// is transparent to the invariant rather than an exception to it. +/// +/// One edge the sentence above does not cover: a chain that TERMINATES at a +/// resolver -- a `200` straight from `doi.org` rather than the `302` it +/// exists to send -- is served by a host `permits` allowed and `matches` +/// would not. The bound still holds in practice because these three hosts are +/// operated by the DOI Foundation and CNRI and do not serve article bytes, +/// which is why the set is closed and exact; but the invariant is "the +/// resolver is trusted to redirect", not "only allowlisted hosts ever send +/// bytes". +/// +/// # Why a closed set and not "the terminal host" +/// +/// #533's first suggestion was to adjudicate only the final host of a +/// chain. That is a bigger change than it looks: it would let a chain +/// traverse ANY host so long as it ended somewhere allowed, and every hop +/// still sees the request. A named, closed set of resolvers keeps the bound. +/// +/// # Why exact hosts and no wildcards +/// +/// `*.doi.org` would sweep in `www.doi.org`, which is the DOI Foundation's +/// website, not a resolver -- and any other subdomain the Foundation ever +/// stands up. Each entry here is a host measured to 302 straight to the +/// publisher (2026-08-30, `10.1002/pcn5.205`): +/// +/// | host | what it is | +/// |---|---| +/// | `doi.org` | the canonical DOI resolver | +/// | `dx.doi.org` | its long-standing alias, still in live metadata | +/// | `hdl.handle.net` | the Handle System resolver `doi.org` proxies | +const TRANSPARENT_RESOLVER_HOSTS: &[&str] = &["doi.org", "dx.doi.org", "hdl.handle.net"]; + +/// Whether `host` is a DOI resolver rather than a content host (#533). +/// +/// Exact match against `TRANSPARENT_RESOLVER_HOSTS`, deliberately without +/// wildcard support: `evil-doi.org`, `doi.org.evil.test` and `www.doi.org` +/// are all NOT resolvers, and the first two are what an attacker would +/// register. +#[must_use] +pub fn is_transparent_resolver(host: &str) -> bool { + let host_lc = host.to_ascii_lowercase(); + TRANSPARENT_RESOLVER_HOSTS.contains(&host_lc.as_str()) } /// Returns `true` if `host` (already lowercased) matches `pattern` per @@ -610,6 +698,22 @@ pub enum HttpError { status: u16, /// The URL that produced the status. url: String, + /// The server's own `Retry-After`, in milliseconds, when it sent one + /// on the response that ended the attempt (#506). + /// + /// `parse_retry_after` already read this header, but only on the + /// retry path -- the terminal `return` discarded it, so by the time + /// the error reached a caller the number was gone and + /// `error.retry_after_ms` looked impossible to fill honestly. It is + /// not: the LAST response carries its own `Retry-After`, and that is + /// the one the caller should wait. + /// + /// `None` when the server sent no header. Deliberately not + /// substituted with `backoff_delay` -- doiget's internal backoff is a + /// guess about the server, and handing a caller a guess wearing the + /// name of a server-supplied value is the defect this field exists to + /// avoid. + retry_after_ms: Option, }, /// No allowlist entry exists for this source. The caller asked /// [`HttpClient`] to fetch on behalf of a source that wasn't passed to @@ -955,7 +1059,7 @@ impl HttpClient { .ok_or_else(|| HttpError::UnknownSource { source_key: source.to_string(), })?; - if !allow.matches(&host) { + if !allow.permits(&host) { return Err(HttpError::RedirectDenied { source_key: source.to_string(), host, @@ -1073,6 +1177,10 @@ impl HttpClient { } return Err(HttpError::HttpStatus { status: code, + // Read from the response that ENDED the attempt, so it is + // the server's number rather than ours (#506). + retry_after_ms: parse_retry_after(response.headers()) + .map(|d| u64::try_from(d.as_millis()).unwrap_or(u64::MAX)), // Issue #146: Springer Nature authenticates via an // `api_key` URL query parameter, and IEEE via // `apikey` (#430) — neither documents a header @@ -1271,7 +1379,7 @@ fn build_client_allow_http(allowlist: SourceAllowlist) -> Result Result = (&e).into(); @@ -2558,6 +2757,73 @@ mod tests { assert_eq!(&body[..], b"ok"); } + /// #506: the server's own `Retry-After` survives to the caller. + /// + /// `parse_retry_after` already read this header, but only on the RETRY + /// path -- the terminal `return` discarded it, so by the time the error + /// reached a caller the number was gone and `error.retry_after_ms` looked + /// impossible to fill honestly. It is not: the response that ends the + /// attempt carries its own header, and that is the one to wait. + #[tokio::test] + async fn a_terminal_429_carries_the_servers_retry_after() { + let server = MockServer::start().await; + // 429 on every attempt, so the retries are exhausted and the error is + // the terminal one -- the case that used to lose the header. + Mock::given(method("GET")) + .and(path("/p")) + .respond_with(ResponseTemplate::new(429).insert_header("Retry-After", "7")) + .mount(&server) + .await; + + let client = build_test_client_for_http("crossref", &host_of(&server)); + let url: Url = format!("{}/p", server.uri()).parse().unwrap(); + let err = client + .fetch_bytes("crossref", url) + .await + .expect_err("every attempt 429s"); + + match err { + HttpError::HttpStatus { + status, + retry_after_ms, + .. + } => { + assert_eq!(status, 429); + assert_eq!( + retry_after_ms, + Some(7_000), + "the SERVER's number, in ms, not our backoff" + ); + } + other => panic!("expected HttpStatus, got {other:?}"), + } + } + + /// No header, no number. Backfilling from `backoff_delay` would hand the + /// caller a guess about the server wearing the name of a value the server + /// supplied -- the defect this field exists to avoid. + #[tokio::test] + async fn a_terminal_429_without_the_header_carries_no_number() { + let server = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/p")) + .respond_with(ResponseTemplate::new(429)) + .mount(&server) + .await; + + let client = build_test_client_for_http("crossref", &host_of(&server)); + let url: Url = format!("{}/p", server.uri()).parse().unwrap(); + let err = client + .fetch_bytes("crossref", url) + .await + .expect_err("every attempt 429s"); + + match err { + HttpError::HttpStatus { retry_after_ms, .. } => assert_eq!(retry_after_ms, None), + other => panic!("expected HttpStatus, got {other:?}"), + } + } + #[tokio::test] async fn permanent_404_is_not_retried() { let server = MockServer::start().await; diff --git a/crates/doiget-core/src/lib.rs b/crates/doiget-core/src/lib.rs index e96c8ae3f..dbfdf1c3e 100644 --- a/crates/doiget-core/src/lib.rs +++ b/crates/doiget-core/src/lib.rs @@ -705,7 +705,11 @@ pub enum ErrorCode { /// message lists the candidate matches. Wire form: `"AMBIGUOUS"`. /// Raised by `doiget search`'s name-filter resolution (ADR-0031 D5). Ambiguous, - /// Filesystem write failed. + /// The local store could not serve the request: a filesystem write + /// failed, or a mutating tool was asked to change an entry that has not + /// been fetched. Deliberately not [`Self::NotFound`] in the second case -- + /// that code says a metadata source reported the id does not exist, and a + /// caller acting on it would treat a perfectly good reference as dead. StoreError, /// Provenance log write failed; the fetch was aborted. LogError, @@ -742,6 +746,98 @@ pub enum ErrorCode { TextUnavailable, } +/// What a caller should DO about a failure, as opposed to what happened. +/// +/// `docs/ERRORS.md` §2 has carried per-code retry guidance since Phase 0 and +/// it is good guidance — but it is a markdown table, and the agent making the +/// retry decision never reads it. Its only signal was the NAME of the code, +/// and several names point the wrong way: `NO_OA_AVAILABLE` is the most common +/// failure there is, and "no OA available" invites an unbounded retry loop for +/// something that will not change until the configuration does (#506). +/// +/// Three states, not two. "Retryable / not retryable" cannot express the case +/// that matters most here — the answer will not change *by itself*, but a +/// named one-line change makes it change. Facing that, an agent should neither +/// loop nor give up silently; it should surface the specific change to a human. +#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] +#[serde(rename_all = "snake_case")] +#[non_exhaustive] +pub enum Disposition { + /// The answer will not change. Do not retry, and do not wait for it. + /// + /// Includes failures a caller can act on by issuing a DIFFERENT request + /// (`INVALID_REF`, `AMBIGUOUS`, `TEXT_UNAVAILABLE`): this call is settled, + /// which is what the disposition is about. + Terminal, + /// The answer may change on its own. Retry, with backoff. + RetryAfter, + /// The answer will not change by itself, but a named change makes it. + /// Surface it; do not loop. + NeedsConfig, +} + +impl Disposition { + /// The wire token, allocation-free. + #[must_use] + pub const fn as_wire(self) -> &'static str { + match self { + Self::Terminal => "terminal", + Self::RetryAfter => "retry_after", + Self::NeedsConfig => "needs_config", + } + } +} + +impl ErrorCode { + /// What a caller should do about this code — see [`Disposition`]. + /// + /// This is the single source of truth. `docs/ERRORS.md` §2 carries a + /// Disposition column, and `errors_md_disposition_column_matches_the_code` + /// parses that table and asserts it against this function for every + /// variant, so the document and the wire cannot drift (#506; the drift + /// pattern is #493). + /// + /// An exhaustive `match` with no wildcard: a new code must decide. + #[must_use] + pub const fn disposition(self) -> Disposition { + match self { + // Settled. The same call will return the same thing. + Self::InvalidRef + | Self::NotFound + | Self::InternalError + // ERRORS.md is explicit: "do not retry". + | Self::NotImplemented + // A different request may work (narrow the name / fetch the PDF + // instead), but THIS one is answered. + | Self::Ambiguous + | Self::TextUnavailable => Disposition::Terminal, + + // May change on its own. + Self::RateLimited + | Self::NetworkError + | Self::FetchTimeout + | Self::LockTimeout => Disposition::RetryAfter, + + // Will not change by itself; a named change makes it. + // + // `NoOaAvailable` sits here rather than in `RetryAfter` on + // purpose: it is the most common failure, and ERRORS.md's "Try + // later, or enable opt-in source" reads to a machine as the + // former when it is nearly always the latter. + // + // `StoreError` / `LogError` are disk and permission problems. A + // machine cannot name the fix, but it must not loop on it either, + // and "surface this to a human" is exactly what this disposition + // means. + Self::NoOaAvailable + | Self::CapabilityDenied + | Self::SchemaTooNew + | Self::StoreError + | Self::LogError => Disposition::NeedsConfig, + } + } +} + impl ErrorCode { /// The `SCREAMING_SNAKE_CASE` wire token for this code, as a /// `&'static str`. Identical to the serde representation but @@ -906,8 +1002,84 @@ pub struct DenialContext { // ResolvedCandidate / ResolveResult (Issue #242) // --------------------------------------------------------------------------- +/// How much of the query a candidate actually matched, as something an +/// agent can branch on (#536). +/// +/// `score` alone is not judgement material. For it to work as a gate, the +/// consumer has to already know that the scorer is token overlap rather than +/// semantic similarity, that 0.5 is the FLOOR so the worst candidate the tool +/// will ever emit still looks like a positive number, and that for a citation +/// string carrying author + title + journal + volume + year, 0.5 means most of +/// it did not match. None of that is in the envelope, and an agent consuming a +/// ranked list takes the head of it. +/// +/// The reported case: a citation for a paper in *Psychiatria Danubina* came +/// back as a different 2010 paper in a different journal by a different author +/// at `score: 0.5` — `quality`, `life`, `bipolar` and `2010` were enough to +/// clear the floor — in the same shape as a `score: 1.0` identity. +/// +/// # These are bands over token overlap, not a semantic verdict +/// +/// [`Self::Exact`] means every token in the query was found somewhere in the +/// candidate record. That is a strong signal and it is still not proof: a +/// short query can match the wrong paper completely. The bands make the +/// difference between "identity" and "coincidence" legible; they do not +/// remove the need to verify before citing. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +#[non_exhaustive] +pub enum Confidence { + /// Every query token matched. Verify before citing, but this is an + /// identity rather than an overlap. + Exact, + /// At least four query tokens in five matched. + Probable, + /// Cleared the 0.5 floor and no more. For a known-item lookup this is a + /// NEGATIVE result wearing a positive number. + Weak, +} + +impl Confidence { + /// Band a token-overlap score. + /// + /// The floor is 0.5 (`MIN_CITATION_SCORE`), so the range actually in play + /// is 0.5..=1.0 and the split at 0.8 asks for four tokens in five. `Exact` + /// compares against 0.999 rather than 1.0 because the score is a division: + /// asking for bit-exact equality would band an all-tokens match as + /// `Probable` on a rounding accident. + /// A score outside `0.0..=1.0`, or `NaN`, is not a token-overlap ratio + /// and gets the lowest band rather than a confident-looking answer. The + /// only caller today guards with `MIN_CITATION_SCORE`, but that constant + /// is private to `crossref.rs` and invisible from this signature -- and + /// this is a public function on a semver-strict crate. + #[must_use] + pub fn from_score(score: f64) -> Self { + if !score.is_finite() || !(0.0..=1.0).contains(&score) { + return Self::Weak; + } + if score >= 0.999 { + Self::Exact + } else if score >= 0.8 { + Self::Probable + } else { + Self::Weak + } + } + + /// The wire token, allocation-free. + #[must_use] + pub const fn as_wire(self) -> &'static str { + match self { + Self::Exact => "exact", + Self::Probable => "probable", + Self::Weak => "weak", + } + } +} + /// A candidate paper resolved from a bibliographic citation string. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[non_exhaustive] pub struct ResolvedCandidate { /// Resolved DOI. pub doi: String, @@ -918,7 +1090,20 @@ pub struct ResolvedCandidate { /// Publication year, if resolved. pub year: Option, /// Token similarity overlap score in `0.0..=1.0`. + /// + /// Thresholded at `0.5`, so this is never below the floor — which is + /// exactly why it reads as a positive number even at its worst. Branch on + /// [`Self::confidence`] instead (#536). pub score: f64, + /// [`Self::score`] banded into something an agent can branch on without + /// knowing anything about the scorer (#536). + pub confidence: Confidence, + /// The query tokens that were found in this candidate's record. + /// + /// The evidence behind the score, so a reader can see *what* matched: in + /// the #536 case it was `quality`, `life`, `bipolar`, `2010` — none of + /// them the author or the journal, which is the whole story. + pub matched: Vec, /// Resolving metadata source (e.g. `"crossref"`). pub source: String, } @@ -2109,6 +2294,74 @@ agreed = true } } + /// #506: `docs/ERRORS.md` §2 and [`ErrorCode::disposition`] are the same + /// claim written twice, so this asserts they say the same thing. + /// + /// The issue asked for exactly this ("the ERRORS.md §2 table either + /// generated from it or asserted against it in a test — otherwise the doc + /// and the wire drift, which is the #493 pattern"). Generating the table + /// would have cost the per-code prose, which is the useful part; asserting + /// it keeps both. + /// + /// Reads the shipped document rather than a fixture copy, so a doc edit + /// that contradicts the code fails here and not in someone's agent. + #[test] + fn errors_md_disposition_column_matches_the_code() { + // Resolves relative to this file; three levels up is the workspace + // root (same reasoning as `safekey_matches_reference_vectors`). + let doc = include_str!("../../../docs/ERRORS.md"); + + let mut checked = 0usize; + for line in doc.lines() { + // `| \`CODE\` | meaning | \`disposition\` | recoverable |` + let Some(rest) = line.strip_prefix("| `") else { + continue; + }; + let Some((code_str, tail)) = rest.split_once("` | ") else { + continue; + }; + let cols: Vec<&str> = tail.split(" | ").collect(); + if cols.len() < 3 { + continue; + } + let documented = cols[1].trim().trim_matches('`'); + + let code = match code_str { + "INVALID_REF" => ErrorCode::InvalidRef, + "NO_OA_AVAILABLE" => ErrorCode::NoOaAvailable, + "RATE_LIMITED" => ErrorCode::RateLimited, + "NETWORK_ERROR" => ErrorCode::NetworkError, + "NOT_FOUND" => ErrorCode::NotFound, + "AMBIGUOUS" => ErrorCode::Ambiguous, + "STORE_ERROR" => ErrorCode::StoreError, + "LOG_ERROR" => ErrorCode::LogError, + "CAPABILITY_DENIED" => ErrorCode::CapabilityDenied, + "FETCH_TIMEOUT" => ErrorCode::FetchTimeout, + "SCHEMA_TOO_NEW" => ErrorCode::SchemaTooNew, + "LOCK_TIMEOUT" => ErrorCode::LockTimeout, + "INTERNAL_ERROR" => ErrorCode::InternalError, + "NOT_IMPLEMENTED" => ErrorCode::NotImplemented, + "TEXT_UNAVAILABLE" => ErrorCode::TextUnavailable, + // Not a §2 row (e.g. the §6 mapping tables). + _ => continue, + }; + assert_eq!( + documented, + code.disposition().as_wire(), + "docs/ERRORS.md §2 says {code_str} is `{documented}`, the code says `{}` — one of the two is wrong and an agent reads the second", + code.disposition().as_wire() + ); + checked += 1; + } + + // The guard the assertion above cannot be: a parser that silently + // matches nothing would pass every time. §2 has one row per code. + assert_eq!( + checked, 15, + "expected every ErrorCode to have a §2 row with a Disposition column; parsed {checked}. Either a code was added without documenting it, or the table's shape changed and this parser stopped seeing it." + ); + } + #[test] fn safekey_matches_reference_vectors() { // include_str! resolves relative to the file containing this macro diff --git a/crates/doiget-core/src/orchestrator.rs b/crates/doiget-core/src/orchestrator.rs index 16ff27bcb..dc644c05c 100644 --- a/crates/doiget-core/src/orchestrator.rs +++ b/crates/doiget-core/src/orchestrator.rs @@ -91,10 +91,12 @@ pub struct MetadataOnlyOutcome { /// /// # Dispatch /// -/// - `Ref::Doi(_)` → Crossref first (bibliographic metadata + OA URL -/// via `message.link[]`). If Crossref returns a usable payload the -/// call returns immediately; Unpaywall is consulted only as a fallback -/// when Crossref fails. The Unpaywall fallback surfaces a license +/// - `Ref::Doi(_)` → Crossref first, for bibliographic metadata. Crossref's +/// `message.link[]` does NOT supply an OA URL -- every entry is +/// programme-scoped (ADR-0052, #517) -- so `oa_url` stays `None` on this +/// path. Unpaywall is consulted as a fallback when Crossref fails, and +/// additionally when [`MetadataOnlyOptions::include_oa_location`] is set +/// (#539). The Unpaywall fallback surfaces a license /// string and may overwrite `oa_url` with the `best_oa_location` /// channel. /// - `Ref::Arxiv(_)` → [`ArxivSource::fetch_metadata_only`]: ONLY the @@ -141,6 +143,79 @@ pub async fn metadata_only( ref_: &Ref, profile: &CapabilityProfile, ctx: &FetchContext, +) -> Result { + metadata_only_with_options(ref_, profile, ctx, MetadataOnlyOptions::default()).await +} + +/// What a metadata-only resolve may spend beyond its single round-trip. +/// +/// One knob, and it exists because #517 removed the reason there was none. +/// The DOI path is Crossref-first, and its rationale was that Crossref's +/// `message.link[]` supplied an OA URL for free. Measurement showed every +/// entry is programme-scoped, so `extract_crossref_publisher_url` +/// correctly returns `None` for every DOI -- leaving `oa_url` permanently +/// null while `docs/MCP_TOOLS.md` s11 told callers to act on it (#539). +/// +/// Crossref-first is still the right default: it resolves nearly every DOI +/// in one request. What was wrong was promising a field it cannot fill. So +/// the caller says whether it wants the second request. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +#[non_exhaustive] +pub struct MetadataOnlyOptions { + /// Consult Unpaywall for a real OA location, filling `oa_url`, + /// `oa_status` and `license` from its record. + /// + /// Costs one extra request against Unpaywall. Unpaywall is a metadata + /// source, so this does not weaken the "never fetches a PDF, never + /// touches a publisher" guarantee: the URL is still reported and never + /// followed. + /// + /// Off by default. A caller that wants metadata should not pay for a + /// location it will not use. + /// + /// # Reading the result + /// + /// With this set, the `oa_status`/`oa_url` pair says which of three + /// things happened: + /// + /// | `oa_status` | `oa_url` | meaning | + /// |---|---|---| + /// | `"closed"` | `None` | asked; this work has no OA location | + /// | `"gold"`, `"green"`, ... | `Some` | asked; here it is | + /// | `None` | `None` | the lookup did not complete | + /// + /// The third row is why a failed Unpaywall call leaves *both* fields + /// `None` rather than filling in a plausible-looking `oa_status`: a + /// caller that paid for a lookup has to be able to tell "no OA location + /// exists" from "I could not find out", and that distinction is the + /// whole of #539. + pub include_oa_location: bool, +} + +impl MetadataOnlyOptions { + /// Ask for the OA location (or not). + /// + /// A builder rather than a struct literal because the struct is + /// `#[non_exhaustive]`: a future option must not be a breaking change + /// for `doiget-mcp`, `doiget-cli`, or anyone else downstream. + #[must_use] + pub const fn with_oa_location(mut self, include: bool) -> Self { + self.include_oa_location = include; + self + } +} + +/// [`metadata_only`], with the caller choosing what it is willing to spend. +/// +/// # Errors +/// +/// As [`metadata_only`]. A failure of the *optional* OA-location lookup is +/// not an error here: see [`MetadataOnlyOptions`]. +pub async fn metadata_only_with_options( + ref_: &Ref, + profile: &CapabilityProfile, + ctx: &FetchContext, + opts: MetadataOnlyOptions, ) -> Result { // Resolver cache (docs/CACHE.md): on a hit within TTL, return the // cached outcome without touching the network — this is what lets @@ -158,14 +233,18 @@ pub async fn metadata_only( ctx.cache_root.as_deref() }; + // Keyed by the options, not just the ref: an entry written by a default + // resolve carries `oa_url: None`, and handing that to a caller that asked + // for the location would answer a question it did not ask. See + // `resolver_cache::cache_file_with_options`. if let Some(root) = cache_root { - if let Some(cached) = crate::resolver_cache::read(root, ref_) { + if let Some(cached) = crate::resolver_cache::read_with_options(root, ref_, opts) { return Ok(cached); } } let outcome = match ref_ { - Ref::Doi(doi) => metadata_only_doi(doi, ref_, profile, ctx).await?, + Ref::Doi(doi) => metadata_only_doi(doi, ref_, profile, ctx, opts).await?, Ref::Arxiv(id) => { let arxiv = arxiv_source_from_env(); let metadata = arxiv.fetch_metadata_only(id, ctx).await?; @@ -185,7 +264,7 @@ pub async fn metadata_only( // Best-effort cache write (never fails the resolve). if let Some(root) = cache_root { - crate::resolver_cache::write(root, ref_, &outcome); + crate::resolver_cache::write_with_options(root, ref_, &outcome, opts); } Ok(outcome) } @@ -249,13 +328,27 @@ pub async fn resolve_only( profile: &CapabilityProfile, ctx: &FetchContext, ) -> Result { - // Delegating to the PURE `metadata_only` is the contract-correct - // implementation, not a placeholder: `metadata_only` never writes + resolve_only_with_options(ref_, profile, ctx, MetadataOnlyOptions::default()).await +} + +/// [`resolve_only`], with the caller choosing what it is willing to spend. +/// +/// # Errors +/// +/// As [`resolve_only`]. +pub async fn resolve_only_with_options( + ref_: &Ref, + profile: &CapabilityProfile, + ctx: &FetchContext, + opts: MetadataOnlyOptions, +) -> Result { + // Delegating to the PURE `metadata_only_with_options` is the + // contract-correct implementation, not a placeholder: it never writes // to the store (the persisting path is the separate // `metadata_only_to_store`, which this function does not call), so // `resolve_only`'s "no store mutation" guarantee holds structurally // and cannot regress (#139). - metadata_only(ref_, profile, ctx).await + metadata_only_with_options(ref_, profile, ctx, opts).await } /// Resolve a [`Ref`] to metadata **and persist the metadata TOML to the @@ -290,7 +383,24 @@ pub async fn metadata_only_to_store( ctx: &FetchContext, store: &dyn Store, ) -> Result { - let outcome = metadata_only(ref_, profile, ctx).await?; + metadata_only_to_store_with_options(ref_, profile, ctx, store, MetadataOnlyOptions::default()) + .await +} + +/// [`metadata_only_to_store`], with the caller choosing what it is willing +/// to spend. +/// +/// # Errors +/// +/// As [`metadata_only_to_store`]. +pub async fn metadata_only_to_store_with_options( + ref_: &Ref, + profile: &CapabilityProfile, + ctx: &FetchContext, + store: &dyn Store, + opts: MetadataOnlyOptions, +) -> Result { + let outcome = metadata_only_with_options(ref_, profile, ctx, opts).await?; let safekey = ref_.safekey(); let metadata = build_metadata_only_metadata(ref_, &outcome); // `pdf_src = None` => writes `/.metadata/.toml` and @@ -688,37 +798,81 @@ fn unpaywall_source_from_env(contact: &str) -> UnpaywallSource { UnpaywallSource::new(contact.to_string()) } -/// DOI branch — Crossref first, with Unpaywall as a fallback when -/// Crossref fails. Crossref's `message.link[]` array (when present) -/// supplies the OA URL hint without making a publisher request. +/// DOI branch: Crossref first, with Unpaywall as a fallback when Crossref +/// fails, and as an *addition* when the caller asked for an OA location. +/// +/// This doc used to say Crossref's `message.link[]` "supplies the OA URL +/// hint without making a publisher request". It does not, and #517 is why: +/// measured across twelve live `link[]` entries and eight captured +/// fixtures, every one was programme-scoped, so +/// [`extract_crossref_publisher_url`] correctly refuses all of them. +/// Crossref-first is still the right default -- one request resolves nearly +/// every DOI -- but the free OA hint that sentence promised never existed +/// (#539). async fn metadata_only_doi( _doi: &Doi, ref_: &Ref, profile: &CapabilityProfile, ctx: &FetchContext, + opts: MetadataOnlyOptions, ) -> Result { let contact = resolve_contact_email(); let crossref = crossref_source_from_env(&contact); match crossref.fetch(ref_, profile, ctx).await { Ok(res) => { let metadata = res.metadata_json.unwrap_or(Value::Null); - let oa_url = extract_crossref_publisher_url(&metadata); - // Pure resolver — no store write here (see `metadata_only` + // `None` for every DOI in practice -- see the fn doc and #517. + // Kept because it is the gate that stops a Similarity Check URL + // being handed out under the name `oa_url`. + let mut oa_url = extract_crossref_publisher_url(&metadata); + // Crossref reports neither an OA status nor a license; both + // channels are Unpaywall's. They stay `None` unless the caller + // asked us to go and look. + let mut oa_status = None; + let mut license = None; + + if opts.include_oa_location { + let unpaywall = unpaywall_source_from_env(&contact); + match unpaywall.fetch(ref_, profile, ctx).await { + Ok(uw) => { + let uw_metadata = uw.metadata_json.unwrap_or(Value::Null); + oa_url = extract_unpaywall_oa_url(&uw_metadata); + oa_status = extract_unpaywall_oa_status(&uw_metadata); + license = if uw.license == "unknown" { + None + } else { + Some(uw.license) + }; + } + Err(e) => { + // Deliberately NOT an error, and deliberately + // leaving `oa_status` as `None`. The caller still + // gets the Crossref metadata it also asked for, and + // a null `oa_status` is what distinguishes "could + // not find out" from a completed lookup reporting + // `"closed"`. Inventing a status here would + // recreate the exact ambiguity this option exists + // to remove (#539). + tracing::debug!( + error = %e, + "metadata_only: OA location lookup failed; leaving oa_status null to say so" + ); + } + } + } + + // `metadata` stays Crossref's and `source` says so: the OA + // lookup adds three fields, it does not change whose record + // this is. + // + // Pure resolver -- no store write here (see `metadata_only` // doc); persistence is `metadata_only_to_store`'s job. Ok(MetadataOnlyOutcome { source: crossref.name().to_string(), resolver_profile: crossref.name().to_string(), - // Crossref does not surface a license directly; the - // license channel for DOI metadata is Unpaywall's - // `best_oa_location.license`. Leave `None` here; the - // agent can call `unpaywall` (or a follow-up slice's - // chained orchestrator) if it needs a license string. - license: None, + license, oa_url, - // Crossref does not report OA status; "not determined". - // (The metadata path is Crossref-first and only consults - // Unpaywall on a Crossref failure — see the fallback arm.) - oa_status: None, + oa_status, metadata, }) } @@ -1011,6 +1165,21 @@ pub struct FetchPaperOutcome { } impl FetchPaperOutcome { + /// `true` when this outcome is a success with nothing withheld. + /// + /// A `Blocked` PDF leg is an `Ok` outcome whose payload was refused, so + /// `Result::is_ok` alone calls it a clean success. Both front ends need + /// the same boundary and had drifted: the CLI special-cased `Blocked` + /// for its exit code and its `SessionEnd` row, the MCP server only for + /// the row's `error_code`, so `doiget_fetch_paper` logged + /// `result: "ok"` beside a non-null code -- a self-contradictory row for + /// the one outcome an agent is most likely to retry, and the outcome + /// #507's repeat suppression has to be able to see. + #[must_use] + pub fn is_clean_success(&self) -> bool { + !matches!(self.pdf_leg, PdfLegStatus::Blocked { .. }) + } + /// Test-only constructor for downstream crates (`doiget-cli`, /// `doiget-mcp`) that need to drive classification / rendering /// logic without running the full orchestrator. Produces a @@ -1724,12 +1893,27 @@ async fn fetch_paper_doi( /// Which OA content URL an optional source reported, if any (#445). /// -/// Only the three sources that publish a direct document URL contribute. +/// Only the four sources that publish a direct document URL contribute. /// OpenAIRE's `urls[]` and DataCite's `url` point at a DOI resolver or a /// landing page, not a file — handing those to the OA chain would spend a /// request to arrive at a page the chain cannot read, and report a /// confusing failure. A source that contributes nothing is not a silent /// gap: it still appears in the attempt trace with its own outcome. +/// What a source said about where the work lives, when it named somewhere +/// but nothing followable (#547). +/// +/// Only OpenAlex today: it is the source that reports EVERY location rather +/// than a best one, so it is the one that can name a repository copy and still +/// hand back no URL. The others surface a single document URL or nothing, and +/// "nothing" is already what the trace says. +#[cfg(feature = "metadata")] +fn describe_optional_source_locations(source: &str, meta: &Value) -> Option<(usize, String)> { + match source { + "openalex" => crate::sources::openalex::describe_locations(meta), + _ => None, + } +} + #[cfg(feature = "metadata")] fn optional_source_oa_url<'a>(source: &str, meta: &'a Value) -> Option<&'a str> { match source { @@ -1820,6 +2004,26 @@ async fn try_optional_source_oa_fallback( } }; let Some(raw) = optional_source_oa_url(name, &meta) else { + // #547: the source ANSWERED and named somewhere the work lives; what + // it named was just not followable. That is a different fact from + // "nobody has a copy", and it used to reach only a `tracing::debug!` + // -- so a run whose OpenAlex record pointed straight at an + // institutional repository reported `no OA PDF available` and gave the + // reader nothing to act on. + // Only when OpenAlex flagged NOTHING as open. A location with + // `is_oa: true` and no `pdf_url` is a real deposit we cannot follow -- + // labelling that `NotOpenAccess` would assert the OPPOSITE of what the + // source said, on the machine-readable `consulted_not_open_access` + // token, with the contradicting evidence buried in prose. That is the + // #538 defect pointing the other way, and it is worse than saying + // nothing: found by review of this very change. + if let Some((oa, detail)) = describe_optional_source_locations(name, &meta) { + if oa == 0 { + if let Some(row) = attempts.iter_mut().find(|a| a.source == name) { + row.outcome = AttemptOutcome::NotOpenAccess { detail }; + } + } + } tracing::debug!( source = name, doi = %doi.as_str(), @@ -2290,7 +2494,10 @@ async fn try_fetch_oa_pdf( .host_str() .map(|h| h.to_ascii_lowercase()) .unwrap_or_default(); - if !allowlist.matches(&host) { + // `permits`, not `matches`: a DOI resolver is addressing, not a + // content host, and Unpaywall routinely reports `doi.org` AS the OA + // location (#533). See `http::is_transparent_resolver`. + if !allowlist.permits(&host) { let e = HttpError::RedirectDenied { source_key: SOURCE.to_string(), host: host.clone(), @@ -2692,6 +2899,22 @@ pub struct BatchOutcome { pub results: Vec, } +impl BatchOutcome { + /// Test-only constructor for downstream crates, mirroring + /// [`FetchPaperOutcome::for_test_synthetic`]. + /// + /// `BatchOutcome` is `#[non_exhaustive]`, so `doiget-mcp` cannot build one + /// to drive its envelope builders -- which is part of why the two batch + /// tools' error mapping went untested long enough to drift (#538). + /// + /// `#[doc(hidden)]`: not a stable public API. + #[doc(hidden)] + #[must_use] + pub fn for_test_synthetic(results: Vec) -> Self { + Self { results } + } +} + /// Iterate over `refs` through [`fetch_paper`], collecting one /// [`BatchResultEntry`] per ref. /// @@ -4025,6 +4248,48 @@ mod tests { assert!(extract_metadata_authors(&json!({"x": 1})).is_empty()); assert!(extract_metadata_authors(&json!({"authors": []})).is_empty()); } + /// #505: the widening advice is derived from the trace, not from a + /// second registry, so it cannot claim a source was skipped when it was + /// consulted -- or name a variable the chain does not actually read. + #[test] + fn widening_env_is_deduped_and_in_chain_order() { + let attempts = vec![ + SourceAttempt::new("crossref", AttemptOutcome::NoRecord), + SourceAttempt::new( + "hal", + AttemptOutcome::Disabled { + env: &["DOIGET_ENABLE_HAL"], + }, + ), + SourceAttempt::new( + "tdm-aps", + AttemptOutcome::Disabled { + env: &["DOIGET_KEY_APS", "DOIGET_AGREE_TDM_APS"], + }, + ), + // A second source behind the same switch must not repeat it. + SourceAttempt::new( + "hal-again", + AttemptOutcome::Disabled { + env: &["DOIGET_ENABLE_HAL"], + }, + ), + ]; + assert_eq!( + widening_env(&attempts), + vec![ + "DOIGET_ENABLE_HAL", + "DOIGET_KEY_APS", + "DOIGET_AGREE_TDM_APS" + ], + "chain order, de-duplicated, and a consulted source contributes nothing" + ); + // Nothing skipped means nothing to suggest. + assert!( + widening_env(&[SourceAttempt::new("crossref", AttemptOutcome::NoRecord)]).is_empty() + ); + assert!(widening_env(&[]).is_empty()); + } } // --------------------------------------------------------------------------- @@ -4297,6 +4562,29 @@ pub fn render_attempts(attempts: &[SourceAttempt]) -> String { .join("\n") } +/// Every environment variable the trace says would widen this search, in +/// first-seen order and de-duplicated. +/// +/// Built from the `Disabled` rows rather than from a separate registry, so +/// it cannot drift from what the run actually skipped: a source that was +/// consulted contributes nothing, and a source that was skipped contributes +/// exactly the variables its own row already names. +/// +/// Order is the chain's order, which is the order a user should set them in +/// (`required_env` documents the same for the multi-variable Tier-3 case). +#[must_use] +pub fn widening_env(attempts: &[SourceAttempt]) -> Vec<&'static str> { + let mut seen = Vec::new(); + for a in attempts { + for var in a.outcome.required_env().unwrap_or_default() { + if !seen.contains(var) { + seen.push(*var); + } + } + } + seen +} + /// True when no source in the trace was actually reached. /// /// Distinguishes "this DOI is genuinely not findable" from "nothing was @@ -4321,15 +4609,15 @@ pub fn nothing_was_consulted(attempts: &[SourceAttempt]) -> bool { fn classify_attempt(e: &FetchError) -> AttemptOutcome { match e { FetchError::NotFound { .. } => AttemptOutcome::NoRecord, - // Sources signal "found it, cannot give it to you" through - // SourceSchema with an explicit hint (OpenAIRE access rights, - // Europe PMC isOpenAccess, HAL openAccess_bool). The hint is - // carried verbatim so the reason survives into the message. - FetchError::SourceSchema { hint } if is_access_refusal(hint) => { - AttemptOutcome::NotOpenAccess { - detail: hint.clone(), - } - } + // A source saying "found it, cannot give it to you" says so in the + // type now (#538). This used to be `SourceSchema` plus a substring + // search over the hint, which meant the phrasing of an error message + // decided how the row rendered -- and #503 reworded one and silently + // turned every Europe PMC refusal into `Failed`. The detail is still + // carried verbatim, because the reason is for a reader, not a match. + FetchError::NotRetrievable { detail, .. } => AttemptOutcome::NotOpenAccess { + detail: detail.clone(), + }, // A policy refusal before the untyped fallback (#470). The // conversion already exists and yields exactly the reason / // attempted / expected / hop_index that `remediation::for_denial` @@ -4527,38 +4815,6 @@ mod attempt_denial_tests { } } -/// Whether a `SourceSchema` hint describes an access refusal rather than a -/// malformed response. -/// -/// The coupling is prose: a source says why it refused in free text and -/// this reads the text back. That is fragile, and #503 is the proof — it -/// reworded Europe PMC's refusal from "not open access" to "advertises no -/// retrievable PDF" and the row silently reclassified from -/// `NotOpenAccess` to `Failed`, which reads to an operator as "the source -/// broke" rather than "the source has it and cannot give it to us". -/// -/// It is caught rather than prevented: `an_access_refusal_is_recorded_ -/// distinctly_from_a_miss` drives the real chain, so any source whose -/// wording drifts out of this predicate fails there. A typed refusal on -/// `FetchError` would prevent it instead, at the cost of widening the -/// closed error set in `docs/ERRORS.md` §3. -#[cfg(any( - feature = "metadata", - feature = "tdm-elsevier", - feature = "tdm-aps", - feature = "tdm-springer", - feature = "tdm-ieee" -))] -fn is_access_refusal(hint: &str) -> bool { - hint.contains("not open access") - || hint.contains("openAccess") - // #503: Europe PMC refuses on "nothing retrievable is - // advertised", which is a strictly narrower claim than "outside - // the OA subset" and is still an access refusal, not a schema - // failure. - || hint.contains("no retrievable PDF") -} - /// Run the optional resolution chain and record one [`SourceAttempt`] per /// source, consulted or not. /// @@ -5265,8 +5521,10 @@ mod chain_tests { /// /// This record carries no `fullTextUrlList` at all, so after #503 it is /// refused for the narrower reason and must still classify as - /// `NotOpenAccess` — see `is_access_refusal`, whose coupling to the - /// wording this test is the only thing guarding. + /// `NotOpenAccess`. Until #538 this test was the only thing guarding + /// that, because classification read the wording back out of the hint; + /// the refusal is a `FetchError::NotRetrievable` now, so the compiler + /// guards it and this asserts the behaviour rather than the phrasing. #[tokio::test] #[serial_test::serial] async fn an_access_refusal_is_recorded_distinctly_from_a_miss() { diff --git a/crates/doiget-core/src/paper_text.rs b/crates/doiget-core/src/paper_text.rs index d87ba0c10..731ac5d4a 100644 --- a/crates/doiget-core/src/paper_text.rs +++ b/crates/doiget-core/src/paper_text.rs @@ -303,19 +303,19 @@ fn apply_max_chars(full: PaperText, max_chars: Option) -> PaperText { /// Local element names whose entire subtree is skipped (text discarded). /// `math` is in this set, but its `alttext` is captured before the subtree /// is skipped (see [`extract_alttext`]). -fn is_skip_element(local: &[u8]) -> bool { - matches!(local, b"script" | b"style" | b"math") +fn is_skip_element(local: &str) -> bool { + matches!(local, "script" | "style" | "math") } /// Heading level (1..=6) for an `h1`–`h6` local name, else `None`. -fn heading_level(local: &[u8]) -> Option { +fn heading_level(local: &str) -> Option { match local { - b"h1" => Some(1), - b"h2" => Some(2), - b"h3" => Some(3), - b"h4" => Some(4), - b"h5" => Some(5), - b"h6" => Some(6), + "h1" => Some(1), + "h2" => Some(2), + "h3" => Some(3), + "h4" => Some(4), + "h5" => Some(5), + "h6" => Some(6), _ => None, } } @@ -323,7 +323,7 @@ fn heading_level(local: &[u8]) -> Option { /// Extract a `` LaTeX source, if present. fn extract_alttext(e: &quick_xml::events::BytesStart<'_>) -> Option { for attr in e.attributes().flatten() { - if attr.key.as_ref() == b"alttext" { + if attr.key.as_ref() == "alttext" { if let Ok(v) = attr.normalized_value(quick_xml::XmlVersion::Explicit1_0) { let s = v.into_owned(); if !s.trim().is_empty() { @@ -388,7 +388,7 @@ fn parse_ar5iv(html: &[u8]) -> Result<(Option, Vec), FetchE let name = e.name(); let local = local_name(name.as_ref()); if is_skip_element(local) { - if skip == 0 && local == b"math" { + if skip == 0 && local == "math" { if let Some(alt) = extract_alttext(&e) { let frag = format!("\\({alt}\\) "); push_target( @@ -407,7 +407,7 @@ fn parse_ar5iv(html: &[u8]) -> Result<(Option, Vec), FetchE flush_section(&mut sections, &mut cur_heading, &mut cur_text); in_heading = level; heading_buf.clear(); - } else if local == b"title" && title.is_none() { + } else if local == "title" && title.is_none() { in_title = true; title_buf.clear(); } @@ -418,7 +418,7 @@ fn parse_ar5iv(html: &[u8]) -> Result<(Option, Vec), FetchE let local = local_name(name.as_ref()); // A self-closing `` contributes its alttext but // has no subtree to skip. - if skip == 0 && local == b"math" { + if skip == 0 && local == "math" { if let Some(alt) = extract_alttext(&e) { let frag = format!("\\({alt}\\) "); push_target( @@ -434,11 +434,7 @@ fn parse_ar5iv(html: &[u8]) -> Result<(Option, Vec), FetchE buf.clear(); } Ok(Event::Text(t)) => { - match t.decode().ok().and_then(|raw| { - quick_xml::escape::unescape(&raw) - .ok() - .map(|c| c.into_owned()) - }) { + match quick_xml::escape::unescape(&t).ok().map(|c| c.into_owned()) { Some(s) => { if !s.is_empty() && skip == 0 { let mut frag = s; @@ -482,7 +478,7 @@ fn parse_ar5iv(html: &[u8]) -> Result<(Option, Vec), FetchE in_heading = 0; // The body of the new section starts fresh. cur_text.clear(); - } else if local == b"title" && in_title { + } else if local == "title" && in_title { in_title = false; let t = normalize(&title_buf); if !t.is_empty() { @@ -551,10 +547,10 @@ fn flush_section( cur_text.clear(); } -/// Strip an XML namespace prefix, returning the local-part bytes -/// (`b"xhtml:p"` -> `b"p"`). Mirrors the arXiv Atom parser's helper. -fn local_name(qname: &[u8]) -> &[u8] { - match qname.iter().rposition(|&b| b == b':') { +/// Strip an XML namespace prefix, returning the local part +/// (`"xhtml:p"` -> `"p"`). Mirrors the arXiv Atom parser's helper. +fn local_name(qname: &str) -> &str { + match qname.rfind(':') { Some(idx) => &qname[idx + 1..], None => qname, } diff --git a/crates/doiget-core/src/refs.rs b/crates/doiget-core/src/refs.rs index e4f474174..765f9d174 100644 --- a/crates/doiget-core/src/refs.rs +++ b/crates/doiget-core/src/refs.rs @@ -66,6 +66,28 @@ pub enum ParseError { /// The source bibliography's citation key, when known. entry_key: Option, }, + /// The entry DOES carry an identifier, and it is one doiget recognises + /// and cannot resolve yet (#500). + /// + /// Distinct from [`Self::NoIdentifier`] because the two send a reader in + /// opposite directions. "entry has no DOI / arXiv id" is accurate about + /// what the parser did and wrong about the entry: a PubMed-exported + /// `.bib` record carrying `pmid = {9659853}` is not deficient, and a user + /// who believes it is will go and edit a bibliography that was fine. The + /// missing piece is on doiget's side. + /// + /// Surfaces as `NOT_IMPLEMENTED` rather than `INVALID_REF`: the input is + /// valid and the support is absent, and the two carry different advice -- + /// "wait for a release" versus "correct your input" (ADR-0055). + #[error("entry {entry_key:?} is identified only by {kind} {value:?}, which doiget cannot resolve yet (issue #500) -- it is NOT missing an identifier")] + UnsupportedIdentifier { + /// Human-facing name of the identifier class, e.g. `"PMID"`. + kind: &'static str, + /// The identifier as written in the entry. + value: String, + /// The source bibliography's citation key, when known. + entry_key: Option, + }, /// The identifier was present but `Ref::parse` rejected it /// (malformed DOI suffix, invalid arXiv id shape, etc.). #[error( @@ -104,6 +126,19 @@ pub enum ParseError { }, } +/// The claim [`ParseError::UnsupportedIdentifier`] makes, without the +/// `entry {entry_key:?}` prefix its `Display` carries. +/// +/// Callers that put `entry_key` in a field of its own -- the CLI `verify` +/// row and the MCP `batch_from_bibliography` envelope both do -- would +/// otherwise say it twice. One definition rather than a copy at each site, +/// because the copies drifted: two of the three carried a run of joined-line +/// whitespace into user-facing output before anything asserted the text. +#[must_use] +pub fn unsupported_identifier_claim(kind: &str, value: &str) -> String { + format!("entry is identified only by {kind} {value:?}, which doiget cannot resolve yet (issue #500); it is NOT missing an identifier") +} + /// Input-shape discriminator per ADR-0030 D4. /// /// `Auto` means "detect from path extension and/or content @@ -347,9 +382,66 @@ fn parse_csl_entry( } } } + // #500, CSL-JSON half. The BibTeX parser learned to say "this entry HAS an + // identifier I cannot use" and this one did not, so the same PubMed record + // exported as CSL-JSON still got "entry has no DOI / arXiv id" -- the claim + // #500 exists to stop doiget making, surviving in the format a Zotero user + // is most likely to hand it. + if let Some((kind, value)) = csl_unsupported_identifier(entry) { + return Err(ParseError::UnsupportedIdentifier { + kind, + value, + entry_key, + }); + } Err(ParseError::NoIdentifier { entry_key }) } +/// The CSL-JSON counterpart of [`unsupported_identifier`]: an identifier +/// doiget recognises and cannot resolve yet (#500). +/// +/// CSL-JSON has no standard PMID field, so exporters improvise. Zotero writes +/// `PMID: 9659853` into `note`; some tools emit a top-level `PMID` key. Both +/// are checked, and the note scan mirrors the arXiv one directly above it. +fn csl_unsupported_identifier(entry: &serde_json::Value) -> Option<(&'static str, String)> { + let direct = |name: &str| -> Option { + let v = entry.get(name)?; + let s = match v { + serde_json::Value::String(s) => s.trim().to_string(), + serde_json::Value::Number(n) => n.to_string(), + _ => return None, + }; + (!s.is_empty()).then_some(s) + }; + for (field, kind) in [ + ("PMID", "PMID"), + ("pmid", "PMID"), + ("PMCID", "PMCID"), + ("pmcid", "PMCID"), + ] { + if let Some(v) = direct(field) { + return Some((kind, v)); + } + } + + // Zotero's `note` carries `PMID: 9659853` / `PMCID: PMC1234567`. + let note = entry.get("note").and_then(|v| v.as_str())?; + for (needle, kind) in [("pmcid:", "PMCID"), ("pmid:", "PMID")] { + let lower = note.to_ascii_lowercase(); + if let Some(at) = lower.find(needle) { + let value: String = note[at + needle.len()..] + .trim_start() + .chars() + .take_while(|c| c.is_ascii_alphanumeric()) + .collect(); + if !value.is_empty() { + return Some((kind, value)); + } + } + } + None +} + /// Parse a BibTeX / BibLaTeX document via the `biblatex` crate /// (ADR-0030 D2). One `@entrytype{KEY, …}` produces one entry; /// `entry_key` is the citation key verbatim. @@ -423,9 +515,53 @@ fn parse_bibtex_entry( }; } } + // #500: before reporting "no identifier", check for one doiget simply + // does not support. Saying "no DOI / arXiv id" about an entry that + // carries a PMID is accurate about the parser and wrong about the entry. + if let Some((kind, value)) = unsupported_identifier(entry) { + return Err(ParseError::UnsupportedIdentifier { + kind, + value, + entry_key, + }); + } Err(ParseError::NoIdentifier { entry_key }) } +/// An identifier doiget recognises but cannot resolve yet (#500). +/// +/// Only classes doiget can *name*. An entry carrying some field this does not +/// know about still reports [`ParseError::NoIdentifier`], which stays correct +/// for it: the point is not to guess, it is to stop saying "no identifier" +/// about the cases where there demonstrably is one. +/// +/// `pmid = {...}` is what PubMed's own BibTeX export writes. The BibLaTeX +/// shape is `eprint = {...}` with `eprinttype = {pubmed}`, which +/// [`arxiv_eligible`] already refuses -- correctly, and until now silently. +fn unsupported_identifier(entry: &biblatex::Entry) -> Option<(&'static str, String)> { + let field = |name: &str| -> Option { + let v = entry.get(name)?.format_verbatim().trim().to_string(); + (!v.is_empty()).then_some(v) + }; + + if let Some(v) = field("pmid") { + return Some(("PMID", v)); + } + if let Some(v) = field("pmcid") { + return Some(("PMCID", v)); + } + let names_pubmed = entry + .get("archiveprefix") + .or_else(|| entry.get("eprinttype")) + .is_some_and(|c| c.format_verbatim().to_ascii_lowercase().contains("pubmed")); + if names_pubmed { + if let Some(v) = field("eprint") { + return Some(("PMID", v)); + } + } + None +} + /// Whether an `eprint` field should be interpreted as an arXiv id. /// True when `archivePrefix` / `eprinttype` names arXiv (case- /// insensitive) or is absent; false when it explicitly names a @@ -446,10 +582,172 @@ fn arxiv_eligible(entry: &biblatex::Entry) -> bool { #[cfg(test)] #[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)] mod tests { + /// #500, CSL-JSON. The BibTeX parser learned to distinguish "has no + /// identifier" from "has one I cannot use"; this format did not, so the + /// same PubMed record exported from Zotero still got the claim #500 exists + /// to prevent -- in the format a Zotero user is most likely to hand over. + #[test] + fn a_csl_entry_identified_only_by_a_pmid_is_not_missing_an_identifier() { + let doc = r#"[{"id":"Smith2020","type":"article-journal","PMID":"9659853"}]"#; + let out = parse_csl_json(doc); + assert_eq!(out.len(), 1); + match &out[0] { + Err(ParseError::UnsupportedIdentifier { kind, value, .. }) => { + assert_eq!(*kind, "PMID"); + assert_eq!(value, "9659853"); + } + other => panic!("expected UnsupportedIdentifier, got {other:?}"), + } + } + + /// Zotero does not emit a top-level `PMID`; it writes it into `note`. + /// A checker that only looked at the field would have missed the exporter + /// that produces most of these files. + #[test] + fn a_csl_note_carrying_a_pmid_is_read_the_same_way() { + let doc = r#"[{"id":"S","type":"article-journal","note":"PMID: 9659853; see PubMed"}]"#; + match &parse_csl_json(doc)[0] { + Err(ParseError::UnsupportedIdentifier { kind, value, .. }) => { + assert_eq!(*kind, "PMID"); + assert_eq!(value, "9659853"); + } + other => panic!("expected UnsupportedIdentifier, got {other:?}"), + } + } + + /// PMCID is the other half of the check and was written without a test. + /// Both the direct field and the Zotero `note` form. + #[test] + fn a_csl_entry_identified_only_by_a_pmcid_is_read_as_pmcid() { + for doc in [ + r#"[{"id":"S","type":"article-journal","PMCID":"PMC1234567"}]"#, + r#"[{"id":"S","type":"article-journal","pmcid":"PMC1234567"}]"#, + r#"[{"id":"S","type":"article-journal","note":"PMCID: PMC1234567"}]"#, + ] { + match &parse_csl_json(doc)[0] { + Err(ParseError::UnsupportedIdentifier { kind, value, .. }) => { + assert_eq!(*kind, "PMCID", "for {doc}"); + assert_eq!(value, "PMC1234567", "for {doc}"); + } + other => panic!("expected UnsupportedIdentifier for {doc}, got {other:?}"), + } + } + } + + /// A numeric PMID must not be coerced into anything, and a DOI alongside + /// one still wins: the check runs only after every supported identifier + /// has been tried. + #[test] + fn a_csl_doi_still_wins_over_a_pmid() { + let doc = r#"[{"id":"S","type":"article-journal","DOI":"10.1234/x","PMID":9659853}]"#; + let parsed = parse_csl_json(doc); + let entry = parsed[0].as_ref().expect("the DOI must still win"); + assert_eq!(entry.ref_.as_input_str(), "10.1234/x"); + } + + /// An entry with genuinely nothing must keep saying so -- the new check + /// must not turn every unidentifiable entry into "unsupported". + #[test] + fn a_csl_entry_with_no_identifier_at_all_still_says_so() { + let doc = r#"[{"id":"S","type":"article-journal","title":"No ids here"}]"#; + assert!(matches!( + &parse_csl_json(doc)[0], + Err(ParseError::NoIdentifier { .. }) + )); + } + use super::*; // ---- detect_format --------------------------------------------- + /// #500: the entry from PubMed's own BibTeX export. It carries `pmid`, + /// and reporting "entry has no DOI / arXiv id" about it is accurate about + /// the parser and wrong about the entry -- a user who believes it goes and + /// edits a bibliography that was fine. + /// + /// The PMID is real: `9659853` is Coryell 1998, whose DOI + /// `10.1176/ajp.155.7.895` NCBI's own esummary returns for it. + #[test] + fn a_pubmed_only_entry_says_it_has_a_pmid_not_that_it_has_nothing() { + let bib = r#"@article{coryell1998, + title = {Lithium discontinuation and subsequent effectiveness}, + author = {Coryell, William}, + year = {1998}, + pmid = {9659853}, +}"#; + let out = parse_bibtex(bib); + assert_eq!(out.len(), 1); + match &out[0] { + Err(ParseError::UnsupportedIdentifier { + kind, + value, + entry_key, + }) => { + assert_eq!(kind, &"PMID"); + assert_eq!(value, "9659853"); + assert_eq!(entry_key.as_deref(), Some("coryell1998")); + let msg = out[0].as_ref().unwrap_err().to_string(); + assert!( + msg.contains("NOT missing an identifier"), + "the message has to contradict the wrong conclusion explicitly, or the reader draws it anyway: {msg}" + ); + } + other => panic!("expected UnsupportedIdentifier, got {other:?}"), + } + } + + /// The BibLaTeX shape. `arxiv_eligible` already refused this -- correctly, + /// and until now silently, which is the whole complaint. + #[test] + fn the_biblatex_eprinttype_pubmed_shape_is_recognised_too() { + let bib = r#"@article{e, + title = {T}, + eprint = {9659853}, + eprinttype = {pubmed}, +}"#; + let out = parse_bibtex(bib); + assert!( + matches!( + &out[0], + Err(ParseError::UnsupportedIdentifier { kind: "PMID", .. }) + ), + "got {:?}", + out[0] + ); + } + + /// An entry with genuinely nothing still reports `NoIdentifier`. The point + /// is not to relabel every failure -- it is to stop saying "no identifier" + /// about the cases where there demonstrably is one. + #[test] + fn an_entry_with_no_identifier_at_all_is_unchanged() { + let bib = "@article{x, + title = {T}, + year = {2020}, +}"; + let out = parse_bibtex(bib); + assert!( + matches!(&out[0], Err(ParseError::NoIdentifier { .. })), + "got {:?}", + out[0] + ); + } + + /// A DOI still wins. The new check runs only after every supported + /// identifier has been tried, so adding it cannot divert an entry doiget + /// could actually have resolved. + #[test] + fn a_doi_alongside_a_pmid_still_resolves() { + let bib = r#"@article{both, + title = {T}, + doi = {10.1176/ajp.155.7.895}, + pmid = {9659853}, +}"#; + let out = parse_bibtex(bib); + let parsed = out[0].as_ref().expect("the DOI must still win"); + assert_eq!(parsed.ref_.as_input_str(), "10.1176/ajp.155.7.895"); + } + #[test] fn detect_by_bib_extension() { let p = Utf8Path::new("/tmp/library.bib"); @@ -703,12 +1001,20 @@ doi:10.1234/foo } #[test] - fn bibtex_non_arxiv_eprinttype_is_skipped() { - // eprinttype names a different server → not an arXiv id, and - // there is no DOI, so the entry has no resolvable identifier. + fn bibtex_non_arxiv_eprinttype_reports_the_identifier_it_found() { + // This test used to assert `NoIdentifier`, with the comment "the entry + // has no resolvable identifier". The entry has a PMID. It is not + // resolvable BY DOIGET, which is a different statement, and the one + // #500 is about -- so the test was pinning the wrong claim. let body = "@article{x, eprint = {12345678}, eprinttype = {pubmed}}"; let res = parse_bibtex(body).into_iter().next().unwrap(); - assert!(matches!(res, Err(ParseError::NoIdentifier { .. }))); + match res { + Err(ParseError::UnsupportedIdentifier { kind, value, .. }) => { + assert_eq!(kind, "PMID"); + assert_eq!(value, "12345678"); + } + other => panic!("expected UnsupportedIdentifier, got {other:?}"), + } } #[test] @@ -824,4 +1130,22 @@ doi:10.1234/foo assert_eq!(Format::CslJson.as_wire(), "csl-json"); assert_eq!(Format::Bibtex.as_wire(), "bibtex"); } + + #[test] + fn the_unsupported_identifier_claim_denies_the_wrong_reading() { + // #500's whole point: the sentence must put the gap on doiget's + // side. A reader who takes "no identifier" at face value goes and + // edits a `.bib` that was fine. + let msg = unsupported_identifier_claim("PMID", "9659853"); + assert!(msg.contains("PMID"), "names the identifier kind: {msg}"); + assert!(msg.contains("9659853"), "quotes the value: {msg}"); + assert!( + msg.contains("NOT missing an identifier"), + "denies the wrong reading: {msg}" + ); + assert!(msg.contains("#500"), "points at the issue: {msg}"); + // The `entry_key` prefix belongs to `Display`, not here -- callers + // carry it in a field of its own and would say it twice. + assert!(!msg.starts_with("entry {"), "no entry_key prefix: {msg}"); + } } diff --git a/crates/doiget-core/src/resolver_cache.rs b/crates/doiget-core/src/resolver_cache.rs index f314620db..029399a97 100644 --- a/crates/doiget-core/src/resolver_cache.rs +++ b/crates/doiget-core/src/resolver_cache.rs @@ -23,7 +23,7 @@ use camino::{Utf8Path, Utf8PathBuf}; use chrono::{DateTime, Duration, Utc}; use serde::{Deserialize, Serialize}; -use crate::orchestrator::MetadataOnlyOutcome; +use crate::orchestrator::{MetadataOnlyOptions, MetadataOnlyOutcome}; use crate::{Ref, RESOLVER_CACHE_TTL_DAYS}; /// Current cache-entry schema version (CACHE.md §2). @@ -45,10 +45,55 @@ struct CacheEntry { /// The on-disk path for a ref's cache entry: /// `/resolver/.toml`. #[must_use] -pub fn cache_file(cache_root: &Utf8Path, ref_: &Ref) -> Utf8PathBuf { - cache_root - .join("resolver") - .join(format!("{}.toml", ref_.safekey().as_str())) +// `pub(crate)`, not `pub`. Nothing outside `doiget-core` calls this module -- +// the orchestrator is the only consumer -- and it is absent from +// `docs/PUBLIC_API.md`, so every one of these was an accidental semver +// commitment, including the on-disk cache layout they encode. This cycle +// added the `_with_options` half and doubled that surface. +// +// `#[cfg(test)]` on the remaining plain wrappers is not tidying: making them +// `pub(crate)` is what revealed that production calls none of them. They +// default the options for this module's own tests and nothing else, and `pub` +// had been keeping the dead-code lint quiet about it. Two of the original +// five, `read` and `write`, turned out to have no caller anywhere -- not even +// a test -- and are gone. +#[cfg(test)] +pub(crate) fn cache_file(cache_root: &Utf8Path, ref_: &Ref) -> Utf8PathBuf { + cache_file_with_options(cache_root, ref_, MetadataOnlyOptions::default()) +} + +/// [`cache_file`], keyed by the options as well as the ref. +/// +/// A default resolve and an `include_oa_location` resolve ask different +/// questions of the network and get different answers, so they cannot share +/// an entry. Serving a default entry to an opt-in caller would answer with +/// `oa_url: None` -- indistinguishable from "Unpaywall was asked and this +/// work has no OA location", which is the one thing the caller paid a +/// request to find out. +/// +/// The other direction is just as wrong: reading the opt-in entry and +/// re-fetching whenever `oa_url` is `None` would re-fetch forever for +/// exactly the closed-access works one asks about repeatedly. +/// +/// So: two entries, separated by a SUBDIRECTORY rather than a filename +/// suffix. A `.oa.toml` suffix would collide -- [`Ref::safekey`] +/// keeps `.` (it is in the allowed set), so the DOI `10.1234/foo.oa` +/// resolved by default and the DOI `10.1234/foo` resolved with the flag +/// would both want `doi_10.1234_foo.oa.toml`. A safekey can never contain a +/// path separator (`/` is replaced with `_`), so a subdirectory cannot. +#[must_use] +pub(crate) fn cache_file_with_options( + cache_root: &Utf8Path, + ref_: &Ref, + opts: MetadataOnlyOptions, +) -> Utf8PathBuf { + let dir = cache_root.join("resolver"); + let dir = if opts.include_oa_location { + dir.join("oa") + } else { + dir + }; + dir.join(format!("{}.toml", ref_.safekey().as_str())) } /// Read a cached outcome for `ref_` if present and still within its TTL. @@ -57,12 +102,25 @@ pub fn cache_file(cache_root: &Utf8Path, ref_: &Ref) -> Utf8PathBuf { /// expired, or a `response` blob that no longer deserializes. `now` is /// injected so tests can pin expiry without touching the clock. #[must_use] -pub fn read_at( +#[cfg(test)] +pub(crate) fn read_at( cache_root: &Utf8Path, ref_: &Ref, now: DateTime, ) -> Option { - let path = cache_file(cache_root, ref_); + read_at_with_options(cache_root, ref_, now, MetadataOnlyOptions::default()) +} + +/// [`read_at`], reading the entry keyed by `opts`. See +/// [`cache_file_with_options`] for why the options are part of the key. +#[must_use] +pub(crate) fn read_at_with_options( + cache_root: &Utf8Path, + ref_: &Ref, + now: DateTime, + opts: MetadataOnlyOptions, +) -> Option { + let path = cache_file_with_options(cache_root, ref_, opts); let text = std::fs::read_to_string(&path).ok()?; let entry: CacheEntry = toml::from_str(&text).ok()?; let fetched: DateTime = DateTime::parse_from_rfc3339(&entry.fetched_at) @@ -75,20 +133,42 @@ pub fn read_at( serde_json::from_str(&entry.response).ok() } -/// Read using the current wall clock. See [`read_at`]. +/// [`read`], reading the entry keyed by `opts`. #[must_use] -pub fn read(cache_root: &Utf8Path, ref_: &Ref) -> Option { - read_at(cache_root, ref_, Utc::now()) +pub(crate) fn read_with_options( + cache_root: &Utf8Path, + ref_: &Ref, + opts: MetadataOnlyOptions, +) -> Option { + read_at_with_options(cache_root, ref_, Utc::now(), opts) } /// Write `outcome` to the cache for `ref_`. Best-effort: returns `false` /// (after a `tracing::debug!`) on any I/O or serialization failure rather /// than propagating, since a cache write must never fail a resolve. -pub fn write_at( +#[cfg(test)] +pub(crate) fn write_at( cache_root: &Utf8Path, ref_: &Ref, outcome: &MetadataOnlyOutcome, now: DateTime, +) -> bool { + write_at_with_options( + cache_root, + ref_, + outcome, + now, + MetadataOnlyOptions::default(), + ) +} + +/// [`write_at`], writing the entry keyed by `opts`. +pub(crate) fn write_at_with_options( + cache_root: &Utf8Path, + ref_: &Ref, + outcome: &MetadataOnlyOutcome, + now: DateTime, + opts: MetadataOnlyOptions, ) -> bool { let response = match serde_json::to_string(outcome) { Ok(s) => s, @@ -111,23 +191,33 @@ pub fn write_at( return false; } }; - let path = cache_file(cache_root, ref_); + let path = cache_file_with_options(cache_root, ref_, opts); if let Some(parent) = path.parent() { if let Err(e) = std::fs::create_dir_all(parent) { tracing::debug!(error = %e, dir = %parent, "resolver cache: mkdir failed; skipping write"); return false; } } - if let Err(e) = std::fs::write(&path, toml_text) { + // tmp + rename, not a plain write. A reader racing a plain write sees a + // half-written file, `toml::from_str` fails, and the entry degrades to a + // miss -- safe, per this module's best-effort contract, but it is a + // re-fetch nobody asked for and a `debug!` line that looks like + // corruption. The store next door already had the helper. + if let Err(e) = crate::store::atomic_write(&path, toml_text.as_bytes()) { tracing::debug!(error = %e, path = %path, "resolver cache: write failed"); return false; } true } -/// Write using the current wall clock. See [`write_at`]. -pub fn write(cache_root: &Utf8Path, ref_: &Ref, outcome: &MetadataOnlyOutcome) -> bool { - write_at(cache_root, ref_, outcome, Utc::now()) +/// [`write()`], writing the entry keyed by `opts`. +pub(crate) fn write_with_options( + cache_root: &Utf8Path, + ref_: &Ref, + outcome: &MetadataOnlyOutcome, + opts: MetadataOnlyOptions, +) -> bool { + write_at_with_options(cache_root, ref_, outcome, Utc::now(), opts) } #[cfg(test)] @@ -159,6 +249,64 @@ mod tests { assert_eq!(got.metadata["DOI"], "10.1234/x"); } + /// #539: the options are part of the key, not just the request. + /// + /// A default resolve caches `oa_url: None` because it never asked. Serving + /// that entry to a caller who DID ask would answer the one question it + /// paid a round-trip for, with a value that means something else. + #[test] + fn an_opt_in_read_does_not_hit_the_default_entry() { + let dir = tempfile::TempDir::new().unwrap(); + let root = Utf8Path::from_path(dir.path()).unwrap(); + let r = Ref::parse("10.1234/x").unwrap(); + let now = Utc::now(); + let with_oa = MetadataOnlyOptions::default().with_oa_location(true); + + assert!(write_at(root, &r, &outcome(), now)); + assert!( + read_at_with_options(root, &r, now, with_oa).is_none(), + "the default entry must not satisfy an opt-in read" + ); + // ... and the converse, so a warm opt-in cache does not start + // answering default calls with a field they did not ask for. + let dir2 = tempfile::TempDir::new().unwrap(); + let root2 = Utf8Path::from_path(dir2.path()).unwrap(); + assert!(write_at_with_options(root2, &r, &outcome(), now, with_oa)); + assert!(read_at(root2, &r, now).is_none()); + assert!(read_at_with_options(root2, &r, now, with_oa).is_some()); + } + + /// The first version of this used a `.oa.toml` SUFFIX, which + /// collides: [`Ref::safekey`] keeps `.` (it is in the allowed character + /// set), so the DOI `10.1234/foo.oa` resolved by default and the DOI + /// `10.1234/foo` resolved with the flag both wanted + /// `doi_10.1234_foo.oa.toml` -- one silently serving the other's answer. + /// A subdirectory cannot collide, because a safekey can never contain a + /// path separator. + #[test] + fn a_dot_oa_doi_cannot_collide_with_an_opt_in_entry() { + let dir = tempfile::TempDir::new().unwrap(); + let root = Utf8Path::from_path(dir.path()).unwrap(); + let plain = Ref::parse("10.1234/foo").unwrap(); + let dotted = Ref::parse("10.1234/foo.oa").unwrap(); + + // Guard the premise: if safekey ever starts escaping `.`, this test + // is no longer testing what it says it is. + assert!( + dotted.safekey().as_str().ends_with(".oa"), + "premise: safekey keeps '.', so a '.oa' suffix is reachable" + ); + + assert_ne!( + cache_file_with_options(root, &dotted, MetadataOnlyOptions::default()), + cache_file_with_options( + root, + &plain, + MetadataOnlyOptions::default().with_oa_location(true) + ), + ); + } + #[test] fn miss_when_absent() { let dir = tempfile::TempDir::new().unwrap(); diff --git a/crates/doiget-core/src/source.rs b/crates/doiget-core/src/source.rs index e8b3e4c7f..ab8330b98 100644 --- a/crates/doiget-core/src/source.rs +++ b/crates/doiget-core/src/source.rs @@ -136,6 +136,35 @@ pub enum FetchError { /// source receives a borrowed string from upstream and re-validates). #[error("invalid ref: {0}")] InvalidRef(#[from] RefParseError), + /// A source found the record and **cannot supply a copy** — an access + /// refusal, not a failure. + /// + /// The distinction is the whole point: "the source has it and cannot + /// give it to us" and "the source broke" lead an operator to different + /// conclusions, and only the second is a bug to chase. + /// + /// This exists as a variant because it used to be a *substring search*. + /// A refusal was [`Self::SourceSchema`] with an explanatory hint, and + /// `orchestrator::is_access_refusal` read the hint back looking for + /// "not open access" / "openAccess" / "no retrievable PDF". #503 + /// reworded Europe PMC's refusal for good reasons, the hint fell out of + /// that list, and every Europe PMC refusal silently became + /// `AttemptOutcome::Failed`. Nothing in the source said the wording was + /// load-bearing, and `hal` matched on `openAccess` — a JSON *field + /// name*, not prose anyone chose (#538). + /// + /// Collapses to [`crate::ErrorCode::NoOaAvailable`], which is an + /// EXISTING wire code: "found it, no free copy" is exactly what that + /// means, so the closed set in `docs/ERRORS.md` §3 does not widen. See + /// ADR-0054. + #[error("{source_key} has the record but no retrievable copy: {detail}")] + NotRetrievable { + /// Which source refused. + source_key: String, + /// Why, in the source's own terms — the flags or codes a reader + /// checks next. Displayed, never parsed: that is the point. + detail: String, + }, /// Source-side schema mismatch (unexpected JSON shape, missing /// required field). Surfaces to [`crate::ErrorCode::InternalError`] /// at the public boundary. @@ -231,9 +260,43 @@ impl From<&FetchError> for crate::ErrorCode { FetchError::Http(HttpError::HttpStatus { status: 401 | 403, .. }) => crate::ErrorCode::CapabilityDenied, - FetchError::Http(_) => crate::ErrorCode::NetworkError, + // Exhaustive over `HttpError`, not `Http(_)`. The wildcard sent + // six deterministic outcomes to `NETWORK_ERROR`, whose disposition + // is `retry_after` -- so an agent was told to back off and retry an + // allowlist refusal, an http:// downgrade, a size cap, a + // wrong content type, an unregistered source key and a malformed + // header, none of which a retry can change. That is the defect + // ADR-0055 exists to remove, in the mapping every surface routes + // through. The `DenialContext` impl 100 lines down already matches + // all eight variants; this one opted out of the same protection. + FetchError::Http(e) => match e { + // Policy decisions. Settled until the configuration changes, + // which is what `needs_config` means -- and each of these + // carries a `DenialContext` naming the fix. + HttpError::RedirectDenied { .. } | HttpError::InsecureRedirect { .. } => { + crate::ErrorCode::CapabilityDenied + } + // The response arrived and was not what was asked for. + // Re-requesting returns the same bytes. + HttpError::OversizedBody { .. } | HttpError::NotAPdf { .. } => { + crate::ErrorCode::NoOaAvailable + } + // The caller asked for a source the client was never given. + // A build/wiring fault, not the network (#454, #462). + HttpError::UnknownSource { .. } | HttpError::InvalidHeader { .. } => { + crate::ErrorCode::InternalError + } + // Genuinely transient: transport failures, and the statuses + // the arms above did not claim. + HttpError::Network(_) | HttpError::HttpStatus { .. } => { + crate::ErrorCode::NetworkError + } + }, FetchError::Log(_) => crate::ErrorCode::LogError, FetchError::InvalidRef(_) => crate::ErrorCode::InvalidRef, + // An access refusal is not an internal error. Before #538 it + // was reported as one, because it travelled as `SourceSchema`. + FetchError::NotRetrievable { .. } => crate::ErrorCode::NoOaAvailable, FetchError::SourceSchema { .. } => crate::ErrorCode::InternalError, // Slice 2: a too-large batch is a request-shape failure, so // collapse to `INVALID_REF` (closest closed-set fit). The @@ -260,6 +323,23 @@ impl From<&FetchError> for crate::ErrorCode { /// still needs for `error.message` and the `From for /// ErrorCode` collapse above. The `Http` arm delegates to the /// `From<&HttpError> for Option` impl in [`crate::http`]. +/// The server's own `Retry-After` for this failure, in milliseconds (#506). +/// +/// `None` when the server sent no header, which is most failures. Deliberately +/// NOT backfilled from doiget's internal backoff: that is a guess about the +/// server, and a guess wearing the name of a server-supplied value is exactly +/// the defect `error.disposition` and this field exist to remove. +/// +/// Pairs with [`crate::Disposition::RetryAfter`] -- the disposition says +/// "retry", and this says how long the server asked you to wait before you do. +#[must_use] +pub fn retry_after_ms(e: &FetchError) -> Option { + match e { + FetchError::Http(HttpError::HttpStatus { retry_after_ms, .. }) => *retry_after_ms, + _ => None, + } +} + impl From<&FetchError> for Option { fn from(e: &FetchError) -> Self { use crate::{DenialContext, DenialReason}; @@ -282,6 +362,12 @@ impl From<&FetchError> for Option { // `TooManyRefs` is a request-shape failure, not a denial — // adding it to the None arm keeps the mapping table consistent.) FetchError::NoOaAvailable + // #538: a source refusing because the work is not open there is + // NOT a denial in the ADR-0023 sense. Nothing was withheld by + // policy, so there is no capability to grant and no allowlist to + // widen -- a `DenialContext` would send a reader after a + // configuration change that does not exist. + | FetchError::NotRetrievable { .. } | FetchError::NotFound { .. } | FetchError::Ambiguous { .. } | FetchError::Log(_) @@ -481,6 +567,60 @@ mod tests { assert!(res.metadata_json.is_none()); } + /// A deterministic HTTP outcome must not be advertised as retriable. + /// + /// `FetchError::Http(_) => NetworkError` was a wildcard over all eight + /// `HttpError` variants, and `NetworkError`'s disposition is + /// `retry_after`. Six of them cannot change on a retry, so the mapping + /// every surface routes through was telling agents to back off and try + /// again on an allowlist refusal, a size cap and an unregistered source + /// key -- the exact advice ADR-0055 exists to stop giving. + #[test] + fn a_deterministic_http_failure_is_not_advertised_as_retriable() { + let cases: Vec<(HttpError, crate::Disposition)> = vec![ + ( + HttpError::RedirectDenied { + source_key: "oa-publisher".into(), + host: "evil.example.com".into(), + expected_hosts: vec!["*.wiley.com".to_string()], + }, + crate::Disposition::NeedsConfig, + ), + ( + HttpError::UnknownSource { + source_key: "tdm-aps".into(), + }, + crate::Disposition::Terminal, + ), + ]; + for (he, want) in cases { + let code: ErrorCode = FetchError::Http(he).into(); + assert_ne!( + code, + ErrorCode::NetworkError, + "a policy/wiring outcome is not a network error: {code:?}" + ); + assert_eq!( + code.disposition(), + want, + "and its disposition must not say retry_after: {code:?}" + ); + } + } + + /// The transient ones keep saying retry, so the fix did not overshoot. + #[test] + fn a_transient_http_failure_still_says_retry() { + let code: ErrorCode = FetchError::Http(HttpError::HttpStatus { + status: 503, + retry_after_ms: None, + url: "https://api.crossref.org/works/10.5555/x".into(), + }) + .into(); + assert_eq!(code, ErrorCode::NetworkError); + assert_eq!(code.disposition(), crate::Disposition::RetryAfter); + } + #[test] fn fetch_error_collapses_to_error_code() { // Mirrors `docs/PUBLIC_API.md` §4 / PR #55 boundary collapse. @@ -494,17 +634,27 @@ mod tests { let e: ErrorCode = FetchError::NoOaAvailable.into(); assert_eq!(e, ErrorCode::NoOaAvailable); + // `UnknownSource` is "the caller asked HttpClient to fetch for a + // source it was never given" -- a wiring fault. This asserted + // `NetworkError` because the mapping used to be `Http(_) => + // NetworkError`, i.e. it pinned the wildcard rather than a decision: + // retrying cannot register a missing source, and `NetworkError`'s + // `retry_after` disposition told an agent to try anyway. It is the + // error #462's TDM reproduction actually hit, and calling it a network + // problem is part of why it read as one. let e: ErrorCode = FetchError::Http(HttpError::UnknownSource { source_key: "mock".into(), }) .into(); - assert_eq!(e, ErrorCode::NetworkError); + assert_eq!(e, ErrorCode::InternalError); + assert_eq!(e.disposition(), crate::Disposition::Terminal); // 404 / 410 / 451 from a metadata source are authoritative "id does // not exist" → NotFound (network-independent), NOT NetworkError. for status in [404u16, 410, 451] { let e: ErrorCode = FetchError::Http(HttpError::HttpStatus { status, + retry_after_ms: None, url: "https://api.crossref.org/works/10.5555/absent".into(), }) .into(); @@ -514,6 +664,31 @@ mod tests { "status {status} should map to NotFound" ); } + // ...and a `Retry-After` on that response does not change it. #506 + // added `retry_after_ms` to this variant, and the arm above briefly + // matched `retry_after_ms: None`, which silently sent a 404 carrying + // the header to `NETWORK_ERROR` -- disposition `retry_after` -- so an + // agent was told to retry a DOI that will never resolve. The header + // says how long to wait IF you retry; it does not make an + // authoritative absence provisional. + for status in [404u16, 410, 451] { + let e: ErrorCode = FetchError::Http(HttpError::HttpStatus { + status, + retry_after_ms: Some(30_000), + url: "https://api.crossref.org/works/10.5555/absent".into(), + }) + .into(); + assert_eq!( + e, + ErrorCode::NotFound, + "status {status} with Retry-After is still NotFound" + ); + assert_eq!( + e.disposition(), + crate::Disposition::Terminal, + "and stays terminal, so nothing tells the agent to retry it" + ); + } // A non-HTTP authoritative absence (e.g. arXiv's empty Atom feed) // also maps to NotFound. let e: ErrorCode = FetchError::NotFound { @@ -525,6 +700,7 @@ mod tests { // `doiget verify` tolerates it rather than failing a live id. let e: ErrorCode = FetchError::Http(HttpError::HttpStatus { status: 503, + retry_after_ms: None, url: "https://api.crossref.org/works/10.5555/down".into(), }) .into(); diff --git a/crates/doiget-core/src/sources/arxiv.rs b/crates/doiget-core/src/sources/arxiv.rs index 12376f783..37a0470e6 100644 --- a/crates/doiget-core/src/sources/arxiv.rs +++ b/crates/doiget-core/src/sources/arxiv.rs @@ -417,10 +417,10 @@ pub(crate) fn parse_atom_feed(xml: &[u8]) -> Result { loop { match reader.read_event_into(&mut buf) { Ok(Event::Start(e)) => { - let name_bytes = e.name(); - let local = local_name(name_bytes.as_ref()); + let name = e.name(); + let local = local_name(name.as_ref()); if !in_entry { - if local == b"entry" { + if local == "entry" { in_entry = true; saw_entry = true; depth = 0; @@ -432,33 +432,33 @@ pub(crate) fn parse_atom_feed(xml: &[u8]) -> Result { // Depth==1 means a direct child of ``. if depth == 1 { match local { - b"title" => target = Some(Target::Title), - b"summary" => target = Some(Target::Summary), - b"published" => target = Some(Target::Published), - b"updated" => target = Some(Target::Updated), + "title" => target = Some(Target::Title), + "summary" => target = Some(Target::Summary), + "published" => target = Some(Target::Published), + "updated" => target = Some(Target::Updated), // arXiv namespace; `local_name` strips the `arxiv:` // prefix, so these match `` / // ``. - b"doi" => target = Some(Target::Doi), - b"journal_ref" => target = Some(Target::JournalRef), - b"author" => { + "doi" => target = Some(Target::Doi), + "journal_ref" => target = Some(Target::JournalRef), + "author" => { in_author = true; authors.push(String::new()); } _ => {} } - } else if depth == 2 && in_author && local == b"name" { + } else if depth == 2 && in_author && local == "name" { target = Some(Target::AuthorName); } buf.clear(); } Ok(Event::Empty(e)) => { - let name_bytes = e.name(); - let local = local_name(name_bytes.as_ref()); - if in_entry && depth == 0 && local == b"category" { + let name = e.name(); + let local = local_name(name.as_ref()); + if in_entry && depth == 0 && local == "category" { // — extract `term`. for attr in e.attributes().flatten() { - if attr.key.as_ref() == b"term" { + if attr.key.as_ref() == "term" { // quick-xml 0.40: `unescape_value()` is // deprecated in favour of `normalized_value()` // (attribute-value normalization resolves the @@ -475,15 +475,12 @@ pub(crate) fn parse_atom_feed(xml: &[u8]) -> Result { } Ok(Event::Text(t)) => { if let Some(tg) = target { - // quick-xml 0.40 removed `BytesText::unescape`. - // Reproduce the old behaviour: decode the bytes, then - // unescape XML entities via `quick_xml::escape::unescape`. - // Best-effort — skip the text on decode/unescape error. - if let Some(s) = t.decode().ok().and_then(|raw| { - quick_xml::escape::unescape(&raw) - .ok() - .map(|c| c.into_owned()) - }) { + // quick-xml 0.40 removed `BytesText::unescape`, and 0.42 + // made the reader UTF-8 throughout, so the decode step is + // gone too -- `BytesText` derefs to `str`. Entities still + // need `quick_xml::escape::unescape`, best-effort: skip + // the text if it fails. + if let Some(s) = quick_xml::escape::unescape(&t).ok().map(|c| c.into_owned()) { match tg { Target::Title => title.get_or_insert_with(String::new).push_str(&s), Target::Summary => { @@ -512,9 +509,9 @@ pub(crate) fn parse_atom_feed(xml: &[u8]) -> Result { buf.clear(); continue; } - let name_bytes = e.name(); - let local = local_name(name_bytes.as_ref()); - if depth == 0 && local == b"entry" { + let name = e.name(); + let local = local_name(name.as_ref()); + if depth == 0 && local == "entry" { // Done with the first entry — stop. We deliberately // ignore any subsequent entries since the orchestrator // always queries a single id. @@ -522,7 +519,7 @@ pub(crate) fn parse_atom_feed(xml: &[u8]) -> Result { } depth -= 1; if depth == 0 { - if local == b"author" { + if local == "author" { in_author = false; // Drop empty author names (defensive). if let Some(last) = authors.last() { @@ -532,7 +529,7 @@ pub(crate) fn parse_atom_feed(xml: &[u8]) -> Result { } } target = None; - } else if depth == 1 && in_author && local == b"name" { + } else if depth == 1 && in_author && local == "name" { target = None; } buf.clear(); @@ -625,13 +622,13 @@ pub(crate) fn parse_atom_feed(xml: &[u8]) -> Result { } /// Strip an XML namespace prefix from a qualified name, returning the -/// local-part bytes. `b"atom:entry"` -> `b"entry"`. Atom uses the default +/// local part. `"atom:entry"` -> `"entry"`. Atom uses the default /// namespace so most names arrive unprefixed; this helper makes the /// parser robust to either form without depending on quick-xml's /// namespace resolver (which would require us to thread a /// `NsReader` and explicit prefix bindings through every event). -fn local_name(qname: &[u8]) -> &[u8] { - match qname.iter().rposition(|&b| b == b':') { +fn local_name(qname: &str) -> &str { + match qname.rfind(':') { Some(idx) => &qname[idx + 1..], None => qname, } diff --git a/crates/doiget-core/src/sources/crossref.rs b/crates/doiget-core/src/sources/crossref.rs index 9fe1006e7..5cc35957e 100644 --- a/crates/doiget-core/src/sources/crossref.rs +++ b/crates/doiget-core/src/sources/crossref.rs @@ -179,12 +179,19 @@ impl CrossrefSource { .filter(|s| !s.is_empty()) .collect(); - let matched = query_tokens + // Collected, not counted: the tokens ARE the evidence, and #536 + // is a report about a caller being handed a number with no way to + // tell an identity from a coincidence. In that case the matches + // were `quality`, `life`, `bipolar`, `2010` -- not the author, not + // the journal, which is the whole story and was not in the + // envelope. + let matched: Vec = query_tokens .iter() .filter(|q| candidate_tokens.contains(*q)) - .count(); + .cloned() + .collect(); - let score = matched as f64 / query_tokens.len() as f64; + let score = matched.len() as f64 / query_tokens.len() as f64; if score >= MIN_CITATION_SCORE { let first_author = fields.authors.first().cloned().unwrap_or_default(); @@ -194,6 +201,8 @@ impl CrossrefSource { author: first_author, year: fields.year, score, + confidence: crate::Confidence::from_score(score), + matched, source: "crossref".to_string(), }); } @@ -581,6 +590,82 @@ mod tests { assert_eq!(cand.author, "Onsager, Lars"); assert_eq!(cand.year, Some(1944)); assert_eq!(cand.score, 1.0); + // #536: an identity and a coincidence used to arrive in the same + // shape. Every query token was found, so this is the top band. + assert_eq!(cand.confidence, crate::Confidence::Exact); + let mut got = cand.matched.clone(); + got.sort(); + assert_eq!( + got, + vec!["1944".to_string(), "onsager".to_string()], + "the evidence behind the score, not just the score" + ); + } + + /// #536: the reported near-miss and the reported identity, banded. + /// + /// A citation for a paper in *Psychiatria Danubina* came back as a + /// different 2010 paper, in a different journal, by a different author, at + /// `score: 0.5` -- `quality`, `life`, `bipolar` and `2010` cleared the + /// floor -- in the SAME SHAPE as a `score: 1.0` identity. 0.5 is the floor + /// (`MIN_CITATION_SCORE`), so the worst candidate the tool can emit still + /// looks like a positive number. + #[test] + fn the_floor_bands_as_weak_and_a_full_match_bands_as_exact() { + use crate::Confidence; + assert_eq!( + Confidence::from_score(MIN_CITATION_SCORE), + Confidence::Weak, + "the worst candidate the tool can emit must not read as a match" + ); + assert_eq!(Confidence::from_score(1.0), Confidence::Exact); + + // Four tokens in five is the `probable` boundary; below it, `weak`. + assert_eq!(Confidence::from_score(0.8), Confidence::Probable); + assert_eq!(Confidence::from_score(0.79), Confidence::Weak); + + // `Exact` compares against 0.999, not 1.0: the score is a division, + // and banding an all-tokens match as `Probable` on a rounding + // accident would be the same defect in miniature. + assert_eq!(Confidence::from_score(7.0 / 7.0), Confidence::Exact); + assert_eq!(Confidence::from_score(0.9999), Confidence::Exact); + } + + /// A value that is not a token-overlap ratio gets the lowest band, not a + /// confident-looking one. `from_score` is public on a semver-strict crate + /// and its only caller's floor (`MIN_CITATION_SCORE`) is private to this + /// file, so the guard has to live in the function, not in the caller. + #[test] + fn a_score_outside_the_ratio_range_is_never_confident() { + use crate::Confidence; + for bad in [-5.0, 1.5, 50.0, f64::NAN, f64::INFINITY, f64::NEG_INFINITY] { + assert_eq!( + Confidence::from_score(bad), + Confidence::Weak, + "{bad} is not a ratio and must not band as a match" + ); + } + } + + /// The bands must stay ordered with the score they band, or the enum says + /// something the number contradicts. + #[test] + fn confidence_is_monotonic_in_the_score() { + use crate::Confidence; + let rank = |c| match c { + Confidence::Weak => 0, + Confidence::Probable => 1, + // No wildcard: `#[non_exhaustive]` binds DOWNSTREAM crates, not + // this one, so a new band has to be ranked here before it compiles. + Confidence::Exact => 2, + }; + let mut prev = 0; + for i in 50..=100 { + let r = rank(Confidence::from_score(f64::from(i) / 100.0)); + assert!(r >= prev, "score {i}/100 banded below a lower score"); + prev = r; + } + assert_eq!(prev, 2, "the top of the range must reach Exact"); } #[tokio::test] diff --git a/crates/doiget-core/src/sources/europepmc.rs b/crates/doiget-core/src/sources/europepmc.rs index 4296a0f42..df9d0a982 100644 --- a/crates/doiget-core/src/sources/europepmc.rs +++ b/crates/doiget-core/src/sources/europepmc.rs @@ -186,9 +186,12 @@ impl Source for EuropePmcSource { // one line before the code written to find them. The flags stay // in the refusal because they are what a reader checks next. if open_access_pdf_url(record).is_none() { - return Err(FetchError::SourceSchema { - hint: format!( - "europepmc record advertises no retrievable PDF: no \ + return Err(FetchError::NotRetrievable { + // Must match `Source::name()`, which is hyphenated; this said + // "europepmc" and only ever surfaced in a message string. + source_key: "europe-pmc".to_string(), + detail: format!( + "no \ fullTextUrlList entry has documentStyle = pdf with \ availabilityCode F or OA (isOpenAccess = {}, inEPMC = {})", flag(record, "isOpenAccess").unwrap_or("absent"), @@ -489,19 +492,32 @@ mod tests { .await .expect_err("a non-OA record must be refused, not returned"); let msg = err.to_string(); + // #538: the CATEGORY, not "some schema error". `SourceSchema` + // collapses to INTERNAL_ERROR at the boundary, which said the source + // broke; a refusal is `NoOaAvailable`, which says the work is not + // free here. assert!( - matches!(err, FetchError::SourceSchema { .. }), - "got {err:?}" + matches!(err, FetchError::NotRetrievable { .. }), + "an access refusal is its own variant, not a schema failure: {err:?}" + ); + assert_eq!( + crate::ErrorCode::from(&err), + crate::ErrorCode::NoOaAvailable, + "and it must not surface as an internal error: {err:?}" ); assert!( msg.contains("isOpenAccess = N") && msg.contains("inEPMC = Y"), "the refusal must say WHY, naming both flags; got: {msg}" ); + // #503's point survives, but it is no longer asserted through the + // wording. That the refusal is about RETRIEVABILITY rather than + // subset membership is now carried by the variant -- checked above -- + // and by the criterion the detail names. Asserting the old phrase + // "no retrievable PDF" would rebuild in the test exactly the + // prose-coupling #538 removed from the classifier. assert!( - msg.contains("no retrievable PDF"), - "the refusal must be about retrievability, not subset membership \ - — the two are distinguishable and #503 was the case where they \ - differ; got: {msg}" + msg.contains("documentStyle = pdf") && msg.contains("availabilityCode"), + "the refusal must name the criterion it applied, which is per-entry retrievability and not the isOpenAccess subset; got: {msg}" ); } diff --git a/crates/doiget-core/src/sources/hal.rs b/crates/doiget-core/src/sources/hal.rs index a6c56e270..4b6850ff5 100644 --- a/crates/doiget-core/src/sources/hal.rs +++ b/crates/doiget-core/src/sources/hal.rs @@ -176,8 +176,9 @@ impl Source for HalSource { .and_then(serde_json::Value::as_bool) != Some(true) { - return Err(FetchError::SourceSchema { - hint: "hal deposit is not open access (openAccess_bool != true)".to_string(), + return Err(FetchError::NotRetrievable { + source_key: "hal".to_string(), + detail: "deposit is not open access (openAccess_bool != true)".to_string(), }); } @@ -425,9 +426,18 @@ mod tests { .fetch(&ref_, &profile(true), &ctx) .await .expect_err("closed deposits must not be returned"); + // #538: the CATEGORY, not "some schema error". `SourceSchema` + // collapses to INTERNAL_ERROR at the boundary, which said the source + // broke; a refusal is `NoOaAvailable`, which says the work is not + // free here. assert!( - matches!(err, FetchError::SourceSchema { .. }), - "got {err:?}" + matches!(err, FetchError::NotRetrievable { .. }), + "an access refusal is its own variant, not a schema failure: {err:?}" + ); + assert_eq!( + crate::ErrorCode::from(&err), + crate::ErrorCode::NoOaAvailable, + "and it must not surface as an internal error: {err:?}" ); } diff --git a/crates/doiget-core/src/sources/openaire.rs b/crates/doiget-core/src/sources/openaire.rs index e961159f6..9bca84fc3 100644 --- a/crates/doiget-core/src/sources/openaire.rs +++ b/crates/doiget-core/src/sources/openaire.rs @@ -173,9 +173,10 @@ impl Source for OpenAireSource { })?; if !is_open_access(record) { - return Err(FetchError::SourceSchema { - hint: format!( - "openaire record is not open access (bestAccessRight.code = {})", + return Err(FetchError::NotRetrievable { + source_key: "openaire".to_string(), + detail: format!( + "record is not open access (bestAccessRight.code = {})", access_right_code(record).unwrap_or("absent") ), }); @@ -426,9 +427,18 @@ mod tests { .fetch(&ref_, &profile(true), &ctx) .await .expect_err("restricted records must not be returned"); + // #538: the CATEGORY, not "some schema error". `SourceSchema` + // collapses to INTERNAL_ERROR at the boundary, which said the source + // broke; a refusal is `NoOaAvailable`, which says the work is not + // free here. assert!( - matches!(err, FetchError::SourceSchema { .. }), - "got {err:?}" + matches!(err, FetchError::NotRetrievable { .. }), + "an access refusal is its own variant, not a schema failure: {err:?}" + ); + assert_eq!( + crate::ErrorCode::from(&err), + crate::ErrorCode::NoOaAvailable, + "and it must not surface as an internal error: {err:?}" ); } diff --git a/crates/doiget-core/src/sources/openalex.rs b/crates/doiget-core/src/sources/openalex.rs index 9911584b7..be7f706ea 100644 --- a/crates/doiget-core/src/sources/openalex.rs +++ b/crates/doiget-core/src/sources/openalex.rs @@ -255,6 +255,72 @@ pub fn open_access_pdf_url(record: &serde_json::Value) -> Option<&str> { }) } +/// What OpenAlex actually said about where this work lives, for the case +/// where [`open_access_pdf_url`] found nothing usable (#547). +/// +/// That function needs `is_oa == true` AND a non-empty `pdf_url`, and a +/// location can fail both while still being a real repository copy. The +/// reported DOI has two locations, the second of which IS the institutional +/// deposit -- but its `landing_page_url` is an author-listing page +/// (`/view/author/70486.html`), not an item, and `is_oa` is `false`. So the +/// extractor correctly returns `None`, the run reports "no OA PDF available", +/// and the fact that a repository was NAMED goes nowhere. +/// +/// "OpenAlex named 1 repository location; its URL is not an item page" points +/// the reader at the repository. "no OA PDF available" points them at giving +/// up. Returns `None` when there is nothing to say. +#[must_use] +// `pub(crate)`, not `pub`: one caller, `orchestrator::describe_optional_source_locations`, +// and the signature takes a raw `&serde_json::Value` -- publishing it would put +// OpenAlex's wire shape under this crate's semver guarantee for no consumer. +pub(crate) fn describe_locations(record: &serde_json::Value) -> Option<(usize, String)> { + let locations = record + .get("locations") + .and_then(serde_json::Value::as_array)?; + if locations.is_empty() { + return None; + } + + let mut oa = 0usize; + let mut with_pdf = 0usize; + let mut named: Vec<&str> = Vec::new(); + for loc in locations { + if loc.get("is_oa").and_then(serde_json::Value::as_bool) == Some(true) { + oa += 1; + } + if loc + .get("pdf_url") + .and_then(serde_json::Value::as_str) + .is_some_and(|u| !u.is_empty()) + { + with_pdf += 1; + } + if let Some(host) = loc + .get("source") + .and_then(|s| s.get("display_name")) + .and_then(serde_json::Value::as_str) + .filter(|s| !s.is_empty()) + { + if !named.contains(&host) { + named.push(host); + } + } + } + + let hosts = if named.is_empty() { + String::new() + } else { + format!(" ({})", named.join("; ")) + }; + Some(( + oa, + format!( + "openalex named {} location(s){hosts}: {oa} flagged open access, {with_pdf} with a PDF URL. A location without a PDF URL may still be a real deposit whose landing page is not an item page", + locations.len() + ), + )) +} + fn truncate_for_hint(body: &[u8]) -> String { const MAX: usize = 200; let s = String::from_utf8_lossy(body); @@ -272,6 +338,102 @@ fn truncate_for_hint(body: &[u8]) -> String { #[cfg(test)] #[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)] mod tests { + + /// #547, with the shape the report measured for `10.1109/tsp.2023.3269664`. + /// + /// Two locations. The second IS the Strathprints deposit -- and it has no + /// `pdf_url`, `is_oa: false`, and a `landing_page_url` pointing at an + /// author-listing page rather than an item. `open_access_pdf_url` + /// correctly returns `None`; the run then reported `no OA PDF available`, + /// which is a different claim from what OpenAlex actually said. + #[test] + fn a_named_but_unusable_location_is_described_rather_than_dropped() { + let record = serde_json::json!({ + "locations": [ + { + "is_oa": false, + "pdf_url": serde_json::Value::Null, + "landing_page_url": "https://doi.org/10.1109/tsp.2023.3269664", + "source": { "display_name": "IEEE Transactions on Signal Processing" } + }, + { + "is_oa": false, + "pdf_url": serde_json::Value::Null, + "landing_page_url": "https://strathprints.strath.ac.uk/view/author/70486.html", + "source": { "display_name": "Strathprints: The University of Strathclyde" } + } + ] + }); + + assert!( + open_access_pdf_url(&record).is_none(), + "premise: the extractor still finds nothing followable" + ); + + let (oa, d) = describe_locations(&record).expect("locations were named"); + assert_eq!( + oa, 0, + "the caller reclassifies to NotOpenAccess only when this is 0, so it is part of the contract, not a detail" + ); + assert!(d.contains("2 location"), "says how many: {d}"); + assert!( + d.contains("Strathprints"), + "NAMES the repository, which is what the reader can act on: {d}" + ); + assert!( + d.contains("0 with a PDF URL"), + "and why none was followed: {d}" + ); + } + + /// The control from the report: a proper item PDF URL at the same + /// repository still resolves, so this describes a gap rather than + /// papering over one. + #[test] + fn a_usable_location_still_resolves_and_needs_no_description() { + let record = serde_json::json!({ + "locations": [{ + "is_oa": true, + "pdf_url": "https://strathprints.strath.ac.uk/91130/7/Khattak-etal.pdf", + "source": { "display_name": "Strathprints: The University of Strathclyde" } + }] + }); + assert_eq!( + open_access_pdf_url(&record), + Some("https://strathprints.strath.ac.uk/91130/7/Khattak-etal.pdf") + ); + } + + /// A location OpenAlex flagged OPEN, with no followable URL, must not be + /// reported as "not open access" -- the count is what stops the caller + /// asserting the opposite of what the source said. + #[test] + fn an_open_location_without_a_pdf_url_is_counted_as_open() { + let record = serde_json::json!({ + "locations": [{ + "is_oa": true, + "pdf_url": serde_json::Value::Null, + "landing_page_url": "https://repo.example/view/author/1.html", + "source": { "display_name": "Some Repository" } + }] + }); + assert!( + open_access_pdf_url(&record).is_none(), + "premise: still nothing followable" + ); + let (oa, d) = describe_locations(&record).expect("a location was named"); + assert_eq!( + oa, 1, + "OpenAlex said this is open; labelling it `NotOpenAccess` would assert the opposite on the machine-readable token: {d}" + ); + } + + /// Nothing to say when nothing was named. + #[test] + fn no_locations_means_no_description() { + assert!(describe_locations(&serde_json::json!({})).is_none()); + assert!(describe_locations(&serde_json::json!({"locations": []})).is_none()); + } use super::*; use std::sync::Arc; diff --git a/crates/doiget-core/src/store/fs_store.rs b/crates/doiget-core/src/store/fs_store.rs index 9dc2b58e0..bb78d99b5 100644 --- a/crates/doiget-core/src/store/fs_store.rs +++ b/crates/doiget-core/src/store/fs_store.rs @@ -36,8 +36,8 @@ use camino::{Utf8Path, Utf8PathBuf}; use fs2::FileExt; use tracing::warn; -use super::metadata::{DoigetExtension, Metadata}; -use super::{EntryInfo, Store, StoreError}; +use super::metadata::{DoigetExtension, Metadata, LICENSE_UNDETERMINED}; +use super::{EntryInfo, Store, StoreError, UserFields}; use crate::{Safekey, SCHEMA_VERSION}; /// Subdirectory under `` that holds metadata TOML files and their @@ -203,6 +203,72 @@ impl Store for FsStore { } fn write(&self, key: &Safekey, m: &Metadata, pdf: Option<&Utf8Path>) -> Result<(), StoreError> { + self.write_with_impl(key, m, pdf, UserFields::Preserve) + } + + fn write_user_authored( + &self, + key: &Safekey, + m: &Metadata, + pdf: Option<&Utf8Path>, + ) -> Result<(), StoreError> { + self.write_with_impl(key, m, pdf, UserFields::Authored) + } + + fn list_recent(&self, limit: usize) -> Result, StoreError> { + let mut entries = read_all_entries(&self.metadata_dir)?; + // Most-recent first by [doiget].fetched_at; entries with no + // `[doiget]` table sort last (None < Some via Reverse). + entries.sort_by_key(|e| std::cmp::Reverse(e.fetched_at)); + entries.truncate(limit); + Ok(entries) + } + + /// Phase 1 search is a linear scan over all metadata files. Phase 2 will + /// add a tantivy / sqlite-fts index when the corpus grows past the point + /// where O(N) per query becomes noticeable in CLI latency. + fn search(&self, query: &str, limit: usize) -> Result, StoreError> { + let q = query.to_lowercase(); + let mut hits = Vec::new(); + for path in metadata_files(&self.metadata_dir)? { + let raw = std::fs::read_to_string(path.as_std_path())?; + let Ok(md) = toml::from_str::(&raw) else { + // Malformed entries are skipped rather than failing the + // whole query. A future audit task will surface them. + continue; + }; + let haystacks = [ + md.title.to_lowercase(), + md.authors.join(" ").to_lowercase(), + md.venue.clone().unwrap_or_default().to_lowercase(), + md.publisher.clone().unwrap_or_default().to_lowercase(), + ]; + if haystacks.iter().any(|h| h.contains(&q)) { + let safekey = safekey_from_metadata_filename(&path); + hits.push(EntryInfo { + safekey, + title: md.title, + year: md.year, + fetched_at: md.doiget.as_ref().map(|d| d.fetched_at), + size_bytes: md.doiget.as_ref().map(|d| d.size_bytes), + }); + if hits.len() >= limit { + break; + } + } + } + Ok(hits) + } +} + +impl FsStore { + fn write_with_impl( + &self, + key: &Safekey, + m: &Metadata, + pdf: Option<&Utf8Path>, + user_fields: UserFields, + ) -> Result<(), StoreError> { let meta_path = self.metadata_path(key)?; let lock_path = self.lock_path(key)?; let lock_file = open_or_create_lock_file(&lock_path)?; @@ -217,7 +283,7 @@ impl Store for FsStore { let raw = std::fs::read_to_string(meta_path.as_std_path())?; let existing: Metadata = toml::from_str(&raw)?; check_schema_version_for_write(&existing.schema_version)?; - merge_metadata(existing, m.clone()) + merge_metadata(existing, m.clone(), user_fields) } else { m.clone() }; @@ -254,51 +320,6 @@ impl Store for FsStore { let _ = ::unlock(&lock_file); Ok(()) } - - fn list_recent(&self, limit: usize) -> Result, StoreError> { - let mut entries = read_all_entries(&self.metadata_dir)?; - // Most-recent first by [doiget].fetched_at; entries with no - // `[doiget]` table sort last (None < Some via Reverse). - entries.sort_by_key(|e| std::cmp::Reverse(e.fetched_at)); - entries.truncate(limit); - Ok(entries) - } - - /// Phase 1 search is a linear scan over all metadata files. Phase 2 will - /// add a tantivy / sqlite-fts index when the corpus grows past the point - /// where O(N) per query becomes noticeable in CLI latency. - fn search(&self, query: &str, limit: usize) -> Result, StoreError> { - let q = query.to_lowercase(); - let mut hits = Vec::new(); - for path in metadata_files(&self.metadata_dir)? { - let raw = std::fs::read_to_string(path.as_std_path())?; - let Ok(md) = toml::from_str::(&raw) else { - // Malformed entries are skipped rather than failing the - // whole query. A future audit task will surface them. - continue; - }; - let haystacks = [ - md.title.to_lowercase(), - md.authors.join(" ").to_lowercase(), - md.venue.clone().unwrap_or_default().to_lowercase(), - md.publisher.clone().unwrap_or_default().to_lowercase(), - ]; - if haystacks.iter().any(|h| h.contains(&q)) { - let safekey = safekey_from_metadata_filename(&path); - hits.push(EntryInfo { - safekey, - title: md.title, - year: md.year, - fetched_at: md.doiget.as_ref().map(|d| d.fetched_at), - size_bytes: md.doiget.as_ref().map(|d| d.size_bytes), - }); - if hits.len() >= limit { - break; - } - } - } - Ok(hits) - } } // --------------------------------------------------------------------------- @@ -333,7 +354,19 @@ fn guard_safekey(s: &str) -> Result<(), StoreError> { /// list/search results; the safekey we emit here originated as a stored /// safekey, so it has already passed `guard_safekey` at write time. fn safekey_from_metadata_filename(p: &Utf8Path) -> Safekey { - Safekey(p.file_stem().unwrap_or("").to_string()) + let stem = p.file_stem().unwrap_or(""); + // The safety argument here is "the filesystem only holds names that + // already passed `guard_safekey` at write time" -- true, and a claim + // about the world rather than something the type enforces. Every other + // `Safekey` in the crate is minted through the guard; this one trusts a + // directory listing. Assert it in debug builds so a future write path + // that skips the guard is caught by the test suite instead of by whatever + // reads the store afterwards. + debug_assert!( + guard_safekey(stem).is_ok(), + "store contains a metadata file whose stem is not a valid safekey: {stem:?}" + ); + Safekey(stem.to_string()) } /// Lock mode for [`acquire_lock`]. @@ -451,7 +484,7 @@ fn parse_schema_version(s: &str) -> Result<(u32, u32), StoreError> { /// in `existing` not present in `incoming` are kept; otherwise `incoming` /// wins (callers usually leave `other` empty on a re-fetch, so existing /// fields survive intact). -fn merge_metadata(existing: Metadata, incoming: Metadata) -> Metadata { +fn merge_metadata(existing: Metadata, incoming: Metadata, user_fields: UserFields) -> Metadata { let mut out = incoming.clone(); // schema_version: never downgrade. The §6 exception explicitly allows a @@ -532,11 +565,47 @@ fn merge_metadata(existing: Metadata, incoming: Metadata) -> Metadata { out.arxiv_categories = existing.arxiv_categories; } - // [doiget]: doiget owns this table; incoming wins (already in `out`). - // If incoming has no [doiget] but existing did, keep the existing one - // so a metadata-only re-write doesn't silently drop a fetch record. - if out.doiget.is_none() && existing.doiget.is_some() { - out.doiget = existing.doiget; + // [doiget]: doiget owns this table, so a re-write wins (STORE.md §6) -- + // except for the two fields whose "absent" value is a marker rather than + // a reading. `oa_status` is omitted when not determined (#281) and + // `license` falls back to `LICENSE_UNDETERMINED`. A `metadata_only` call + // without `include_oa_location` never runs the OA lookup, so it carries + // exactly those markers; letting them win replaces an answer with the + // absence of one. STORE.md §6 permits a `[doiget]` downgrade because it + // is reported to the operator (#118) -- on this path nothing is, which is + // what ADR-0056 closes. Preserving cannot suppress real news: a paper that + // stops being open access reports `Some("closed")`, and a license that + // changes reports the new string. + match (existing.doiget, out.doiget.as_mut()) { + (Some(existing_d), Some(incoming_d)) => { + if incoming_d.oa_status.is_none() { + incoming_d.oa_status = existing_d.oa_status; + } + if incoming_d.license == LICENSE_UNDETERMINED { + incoming_d.license = existing_d.license; + } + // `tags` / `collections` / `annotation` are USER-AUTHORED. A + // fetch never writes them -- all three orchestrator construction + // sites hard-code `Vec::new()` / `None` -- so letting the + // incoming side win meant `doiget tag X --add priority` followed + // by any `doiget fetch X` silently discarded the tag. Same defect + // ADR-0056 closed for `oa_status`/`license` two lines up, on the + // fields where the loss is the user's own data rather than a + // re-derivable reading. + // + // `UserFields::Authored` is how `doiget tag` / `doiget annotate` + // say they mean it, including meaning an EMPTY list: without that + // distinction, removing the last tag would be a silent no-op. + if matches!(user_fields, UserFields::Preserve) { + incoming_d.tags = existing_d.tags; + incoming_d.collections = existing_d.collections; + incoming_d.annotation = existing_d.annotation; + } + } + // Incoming carries no [doiget] at all: keep the existing fetch record + // rather than dropping it. + (Some(existing_d), None) => out.doiget = Some(existing_d), + (None, _) => {} } // `other` (unknown tables / fields): union, prefer EXISTING on key @@ -688,7 +757,7 @@ fn toml_value_inline(value: &toml::Value) -> Result { /// A crash mid-write leaves either the old file intact (if before the /// rename) or the new file fully written (if after). It never leaves a /// partially-visible new file. -fn atomic_write(dst: &Utf8Path, bytes: &[u8]) -> std::io::Result<()> { +pub(crate) fn atomic_write(dst: &Utf8Path, bytes: &[u8]) -> std::io::Result<()> { let file_name = dst.file_name().ok_or_else(|| { std::io::Error::new( std::io::ErrorKind::InvalidInput, @@ -870,6 +939,107 @@ mod tests { FsStore::new(root).expect("FsStore::new") } + #[test] + fn a_default_rewrite_does_not_downgrade_a_known_oa_status_or_license() { + // Issue #583. `metadata_only` without `include_oa_location` never + // runs the OA lookup, so it carries `oa_status: None` and + // `license: "unknown"` -- the not-determined markers, not readings. + // Letting them win would replace an answer with the absence of one, + // and STORE.md §6 only permits a [doiget] downgrade that is reported. + let mut existing = sample_metadata(); + existing.url = Some("https://example.org/paper.pdf".to_string()); + let d = existing.doiget.as_mut().expect("sample has [doiget]"); + d.oa_status = Some("gold".to_string()); + d.license = "CC-BY-4.0".to_string(); + + let mut incoming = sample_metadata(); + incoming.url = None; + let d = incoming.doiget.as_mut().expect("sample has [doiget]"); + d.oa_status = None; + d.license = LICENSE_UNDETERMINED.to_string(); + + let out = merge_metadata(existing, incoming, UserFields::Preserve); + let d = out.doiget.expect("[doiget] survives"); + assert_eq!(d.oa_status.as_deref(), Some("gold")); + assert_eq!(d.license, "CC-BY-4.0"); + // `url` was already protected by `merge_opt!`; pinned so the two + // halves of #583 cannot drift apart. + assert_eq!(out.url.as_deref(), Some("https://example.org/paper.pdf")); + } + + #[test] + fn a_rewrite_that_determined_a_new_oa_status_or_license_still_wins() { + // The other half: preserving must not suppress real news. A paper + // that stops being open access reports `Some("closed")`, not `None`. + let mut existing = sample_metadata(); + let d = existing.doiget.as_mut().expect("sample has [doiget]"); + d.oa_status = Some("gold".to_string()); + d.license = "CC-BY-4.0".to_string(); + + let mut incoming = sample_metadata(); + let d = incoming.doiget.as_mut().expect("sample has [doiget]"); + d.oa_status = Some("closed".to_string()); + d.license = "CC-BY-NC-4.0".to_string(); + + let out = merge_metadata(existing, incoming, UserFields::Preserve); + let d = out.doiget.expect("[doiget] survives"); + assert_eq!(d.oa_status.as_deref(), Some("closed")); + assert_eq!(d.license, "CC-BY-NC-4.0"); + } + + /// A tag the user added must survive a re-fetch. + /// + /// Every `DoigetExtension` the orchestrator builds hard-codes + /// `tags: Vec::new()`, so before `UserFields::Preserve` the incoming + /// empty list won and `doiget tag X --add priority` followed by any + /// `doiget fetch X` discarded the tag with no warning, no log row and no + /// exit-code effect. ADR-0056 closed exactly this for `oa_status` and + /// `license` two fields over. + #[test] + fn a_fetch_does_not_discard_the_user_tags_it_never_authored() { + let mut existing = sample_metadata(); + let d = existing.doiget.as_mut().expect("doiget table"); + d.tags = vec!["priority".to_string()]; + d.collections = vec!["to-read".to_string()]; + d.annotation = Some("check the appendix".to_string()); + + // What a re-fetch hands to the store. + let incoming = sample_metadata(); + assert!( + incoming + .doiget + .as_ref() + .is_some_and(|d| d.tags.is_empty() && d.annotation.is_none()), + "the fixture must model a fetch, which authors none of these" + ); + + let out = merge_metadata(existing, incoming, UserFields::Preserve); + let d = out.doiget.expect("doiget table"); + assert_eq!(d.tags, vec!["priority".to_string()], "tag survived"); + assert_eq!(d.collections, vec!["to-read".to_string()]); + assert_eq!(d.annotation.as_deref(), Some("check the appendix")); + } + + /// ...and `doiget tag --remove` of the last tag still empties it. + /// + /// This is why the policy is a parameter rather than "preserve when the + /// incoming value is empty": for an authored write the empty list IS the + /// intent, and collapsing the two would make removal a silent no-op -- + /// trading one silent data problem for another. + #[test] + fn an_authored_write_can_empty_the_user_fields() { + let mut existing = sample_metadata(); + let d = existing.doiget.as_mut().expect("doiget table"); + d.tags = vec!["priority".to_string()]; + d.annotation = Some("old note".to_string()); + + let incoming = sample_metadata(); // what `tag --remove` writes back + let out = merge_metadata(existing, incoming, UserFields::Authored); + let d = out.doiget.expect("doiget table"); + assert!(d.tags.is_empty(), "removal is honoured: {:?}", d.tags); + assert_eq!(d.annotation, None, "clear is honoured"); + } + #[test] fn merge_metadata_preserves_existing_arxiv_categories() { // Issue #303 / review #318: a later metadata-only re-write that did @@ -878,7 +1048,7 @@ mod tests { existing.arxiv_categories = vec!["cond-mat.str-el".to_string()]; let mut incoming = sample_metadata(); incoming.arxiv_categories = vec![]; // re-write without categories - let merged = merge_metadata(existing, incoming); + let merged = merge_metadata(existing, incoming, UserFields::Preserve); assert_eq!(merged.arxiv_categories, vec!["cond-mat.str-el".to_string()]); } diff --git a/crates/doiget-core/src/store/metadata.rs b/crates/doiget-core/src/store/metadata.rs index efba2c4a0..5d5f169d5 100644 --- a/crates/doiget-core/src/store/metadata.rs +++ b/crates/doiget-core/src/store/metadata.rs @@ -97,6 +97,14 @@ pub struct Metadata { pub other: std::collections::BTreeMap, } +/// The value `[doiget].license` carries when no license was determined. +/// +/// It is a marker for "the lookup did not produce one", not a reading. A +/// resolver that genuinely reports a license writes that license; nothing +/// reports `"unknown"` as news. `merge_metadata` relies on that to tell an +/// absent answer from a new one. +pub const LICENSE_UNDETERMINED: &str = "unknown"; + /// doiget-specific extension table (`[doiget]`). /// /// Per `docs/STORE.md` §6, doiget owns this table outright and may diff --git a/crates/doiget-core/src/store/mod.rs b/crates/doiget-core/src/store/mod.rs index 2a996c7bf..fd763e5c8 100644 --- a/crates/doiget-core/src/store/mod.rs +++ b/crates/doiget-core/src/store/mod.rs @@ -25,6 +25,10 @@ pub mod metadata; pub mod render; pub use fs_store::FsStore; + +/// Crash-consistent write (tmp + fsync + rename), shared with the resolver +/// cache so both write the same way. See `docs/STORE.md` §5. +pub(crate) use fs_store::atomic_write; pub use metadata::{DoigetExtension, Metadata}; pub use render::{to_bibtex, to_csl_array}; @@ -137,6 +141,27 @@ pub enum StoreError { }, } +/// Who is authoritative for the user-authored `[doiget]` fields +/// (`tags`, `collections`, `annotation`) on a write. +/// +/// A fetch never authors them: every `DoigetExtension` the orchestrator +/// builds hard-codes `Vec::new()` / `None`. Letting that win silently +/// discarded a user's tags on any re-fetch, which is the loss ADR-0056 +/// closed for `oa_status` / `license` and left open here. +/// +/// The distinction has to be explicit rather than "is the incoming value +/// empty", because `doiget tag --remove` and `doiget annotate --clear` +/// legitimately mean the empty value. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[non_exhaustive] +pub enum UserFields { + /// The caller did not author them; keep whatever is on disk. Fetches. + Preserve, + /// The caller means exactly what it wrote, empty included. `doiget tag`, + /// `doiget annotate`, and their MCP equivalents. + Authored, +} + /// Filesystem-shaped metadata store, semver-locked per `docs/PUBLIC_API.md` /// §2. /// @@ -163,8 +188,21 @@ pub trait Store: Send + Sync { /// `/.pdf` via the same atomic-rename dance as the /// metadata file. The caller is responsible for emitting the /// `event=store_write` provenance row (see `docs/PROVENANCE_LOG.md` §3). + /// Write a fetch result. User-authored `[doiget]` fields already on disk + /// are preserved ([`UserFields::Preserve`]) -- a fetch does not author + /// them, and silently dropping them is data loss. fn write(&self, key: &Safekey, m: &Metadata, pdf: Option<&Utf8Path>) -> Result<(), StoreError>; + /// Write on behalf of a caller that DID author the user fields, so an + /// empty `tags` / `collections` or a `None` annotation means exactly that + /// ([`UserFields::Authored`]). `doiget tag` / `doiget annotate` only. + fn write_user_authored( + &self, + key: &Safekey, + m: &Metadata, + pdf: Option<&Utf8Path>, + ) -> Result<(), StoreError>; + /// Return up to `limit` entries, most-recent first by `[doiget].fetched_at`. fn list_recent(&self, limit: usize) -> Result, StoreError>; diff --git a/crates/doiget-mcp/src/lib.rs b/crates/doiget-mcp/src/lib.rs index a26078716..be778aee7 100644 --- a/crates/doiget-mcp/src/lib.rs +++ b/crates/doiget-mcp/src/lib.rs @@ -52,8 +52,8 @@ use doiget_core::http::{ }; use doiget_core::orchestrator::{ batch_fetch as core_batch_fetch, batch_fetch_plans, fetch_paper as core_fetch_paper, - metadata_only_to_store, resolve_only as core_resolve_only, FetchPaperOutcome, - MetadataOnlyOutcome, PdfLegStatus, + metadata_only_to_store_with_options, resolve_only_with_options as core_resolve_only, + FetchPaperOutcome, MetadataOnlyOptions, MetadataOnlyOutcome, PdfLegStatus, }; use doiget_core::provenance::{Capability, LogEvent, LogResult, ProvenanceLog, RowInput}; use doiget_core::rate_limiter::RateLimiter; @@ -98,12 +98,39 @@ pub struct Server { /// the per-instance trimming visible to `tools/list` and `tools/call` /// (issue #379). Do not switch the handler back to the associated fn. tool_router: ToolRouter, + /// ONE per process, shared by every tool call. + /// + /// `RateLimiter`'s state -- the rolling global window and the + /// per-source next-allowed instants -- lives in its own `Arc>` + /// fields, so it paces only the calls that share the instance. Building + /// a fresh one inside every tool handler, which is what this server did, + /// gave each call an empty `per_source_next` and no memory of the last: + /// arXiv's 3 s spacing, which `docs/LEGAL.md` treats as an obligation + /// rather than politeness, was not enforced between MCP calls at all. + /// The type calls itself "process-wide"; nothing made it so. + rate_limiter: Arc, + /// ONE per process, opened lazily because `Server::new` is infallible. + /// + /// `ProvenanceLog::open` documents its own contract: the `session_id` + /// "MUST be a 26-char ULID generated once per process", and a long-lived + /// handle reuses it. Opening per tool call broke that thirteen times + /// over, and worse: `open` seeds `(next_seq, last_hash)` by reading the + /// file, so two overlapping calls both read the same state and both + /// append rows claiming the same `ts_seq` with a `prev_hash` that does + /// not match the row actually before them -- which is precisely what + /// `doiget audit-log --verify` reports as a broken chain. + log: std::sync::OnceLock>, + /// The process ULID the log above is opened with. + session_id: String, } #[tool_router] impl Server { /// Construct a server with the given runtime capability profile. pub fn new(profile: CapabilityProfile) -> Self { + let rate_limiter = Arc::new(RateLimiter::new(RateLimits::HARD_CODED)); + let session_id = ulid::Ulid::generate().to_string(); + let log = std::sync::OnceLock::new(); let mut tool_router = Self::tool_router(); // Issue #379 / #373(b): a tool that can only ever answer // NOT_IMPLEMENTED is worse than an absent one — an agent will @@ -126,9 +153,42 @@ impl Server { Self { profile, tool_router, + rate_limiter, + log, + session_id, } } + /// The per-call [`FetchContext`], built from the process-lifetime + /// rate limiter and provenance log this server owns. + /// + /// Was a free `self.fetch_context()` that constructed both fresh on + /// every tool call. See the field docs on [`Server`] for what that cost: + /// no rate pacing between MCP calls, and a provenance hash chain that + /// two overlapping calls could break. + fn fetch_context(&self) -> anyhow::Result { + let log = match self.log.get() { + Some(l) => Arc::clone(l), + None => { + let opened = Arc::new(open_provenance_log(self.session_id.clone())?); + // A racing caller may have won; prefer whichever landed so + // every context in this process shares ONE handle, which is + // what serialises `append` and keeps the chain intact. + let _ = self.log.set(Arc::clone(&opened)); + self.log.get().map_or(opened, Arc::clone) + } + }; + Ok(FetchContext { + http: Arc::new(build_http_client_for_fetch()?), + rate_limiter: Arc::clone(&self.rate_limiter), + log, + session_id: self.session_id.clone(), + // Resolver cache disabled on the MCP path for now; the resolve + // cache (docs/CACHE.md) is wired through `doiget verify` first. + cache_root: None, + }) + } + /// Run the MCP server until stdin reaches EOF. /// /// Returns once the underlying rmcp service loop exits — that happens @@ -248,11 +308,11 @@ impl Server { /// [`FetchPlan`]: doiget_core::dry_run::FetchPlan #[tool( description = "WHEN TO USE: User wants metadata for a DOI / arXiv id without paying for or being noticed by a PDF download.\n\ - INPUTS: ref (DOI or arXiv id), dry_run (optional bool).\n\ - OUTPUTS: { ok: true, ref, source, license?, oa_url, metadata } OR { ok: true, dry_run: true, ref, plan, rate_limit_budget } OR { ok:false, error }.\n\ - COSTS: 1-2 s metadata round-trip (or 0 when dry_run).\n\ + INPUTS: ref (DOI or arXiv id), dry_run (optional bool), include_oa_location (optional bool).\n\ + OUTPUTS: { ok: true, ref, source, license?, oa_url, oa_status, metadata } OR { ok: true, dry_run: true, ref, plan, rate_limit_budget } OR { ok:false, error }.\n\ + COSTS: 1-2 s metadata round-trip (or 0 when dry_run; roughly doubled when include_oa_location).\n\ SIDE EFFECTS: Appends a 'metadata-only' provenance row (unless dry_run). Writes the metadata TOML to the store. Never fetches PDF.\n\ - LIMITS: Subject to the same rate cap as fetch_paper (5/sec). The OA URL is reported but never followed.", + LIMITS: Subject to the same rate cap as fetch_paper (5/sec). The OA URL is reported but never followed. Crossref alone cannot supply an oa_url, because its link[] entries are scoped to a licensed programme (Similarity Check / TDM / syndication) rather than being general-purpose. A null oa_url is therefore NOT evidence that the work is closed. oa_url is null on the default path whenever Crossref answered, which is nearly every DOI. It can be non-null WITHOUT the flag in one case: Crossref failed and Unpaywall answered instead, and then source is unpaywall rather than crossref - so source tells you which happened. With include_oa_location set, oa_status says which answer you got: closed means the lookup completed and found no OA location, null means the lookup did not complete.", annotations( read_only_hint = false, destructive_hint = false, @@ -300,7 +360,7 @@ impl Server { // selection and per-leg politeness; we own the per-call // session boundary (SessionStart / SessionEnd bookend rows) and // the wire envelope shape (`docs/MCP_TOOLS.md` §11). - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(metadata_only_error_envelope( @@ -368,12 +428,19 @@ impl Server { ))); } - let outcome = metadata_only_to_store(&ref_, &self.profile, &ctx, &store).await; + let opts = MetadataOnlyOptions::default().with_oa_location(input.include_oa_location); + let outcome = + metadata_only_to_store_with_options(&ref_, &self.profile, &ctx, &store, opts).await; // SessionEnd bookend. Best-effort: if this append fails we still // surface the orchestrator's outcome (a fresh log error here // would mask the more informative orchestrator error). let session_ok = outcome.is_ok(); + // #507: the bookend recorded THAT the call failed and not WHAT it + // failed with, so the provenance log could not answer "what did this + // session tell the caller about this ref?" -- which is the question + // repeat suppression has to ask before it can suppress anything. + let session_err = outcome.as_ref().err().map(|e| ErrorCode::from(e).as_wire()); let _ = ctx.log.append(RowInput { event: LogEvent::SessionEnd, result: if session_ok { @@ -384,7 +451,7 @@ impl Server { capability: Capability::Metadata, ref_: Some(input.ref_.as_str()), source: None, - error_code: None, + error_code: session_err, size_bytes: None, license: None, store_path: None, @@ -434,11 +501,11 @@ impl Server { /// `doiget_metadata_only` with `dry_run: true` instead. #[tool( description = "WHEN TO USE: User wants metadata for a DOI / arXiv id with no local persistence (audit log row only).\n\ - INPUTS: ref (DOI or arXiv id).\n\ - OUTPUTS: { ok: true, ref, source, resolver_profile, license?, oa_url, metadata, schema_version } OR { ok:false, ref, error }.\n\ - COSTS: 1-2 s metadata round-trip.\n\ + INPUTS: ref (DOI or arXiv id), include_oa_location (optional bool).\n\ + OUTPUTS: { ok: true, ref, source, resolver_profile, license?, oa_url, oa_status, metadata, schema_version } OR { ok:false, ref, error }.\n\ + COSTS: 1-2 s metadata round-trip (roughly doubled when include_oa_location).\n\ SIDE EFFECTS: Appends one provenance row per consulted resolver. NEVER writes a metadata TOML to the store. NEVER fetches PDF.\n\ - LIMITS: Subject to the same rate cap as metadata_only (5/sec). The OA URL is reported but never followed. dry_run is not supported; use metadata_only with dry_run for a preview.", + LIMITS: Subject to the same rate cap as metadata_only (5/sec). The OA URL is reported but never followed. Crossref alone cannot supply an oa_url, because its link[] entries are scoped to a licensed programme (Similarity Check / TDM / syndication) rather than being general-purpose. A null oa_url is therefore NOT evidence that the work is closed. oa_url is null on the default path whenever Crossref answered, which is nearly every DOI. It can be non-null WITHOUT the flag in one case: Crossref failed and Unpaywall answered instead, and then source is unpaywall rather than crossref - so source tells you which happened. With include_oa_location set, oa_status says which answer you got: closed means the lookup completed and found no OA location, null means the lookup did not complete. dry_run is not supported; use metadata_only with dry_run for a preview.", annotations( read_only_hint = false, destructive_hint = false, @@ -465,7 +532,7 @@ impl Server { // Step 2: build the per-call context. Failures here surface as // INTERNAL_ERROR per the metadata_only pattern. - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(metadata_only_error_envelope( @@ -501,10 +568,16 @@ impl Server { ))); } - let outcome = core_resolve_only(&ref_, &self.profile, &ctx).await; + let opts = MetadataOnlyOptions::default().with_oa_location(input.include_oa_location); + let outcome = core_resolve_only(&ref_, &self.profile, &ctx, opts).await; // SessionEnd bookend. Best-effort. let session_ok = outcome.is_ok(); + // #507: the bookend recorded THAT the call failed and not WHAT it + // failed with, so the provenance log could not answer "what did this + // session tell the caller about this ref?" -- which is the question + // repeat suppression has to ask before it can suppress anything. + let session_err = outcome.as_ref().err().map(|e| ErrorCode::from(e).as_wire()); let _ = ctx.log.append(RowInput { event: LogEvent::SessionEnd, result: if session_ok { @@ -515,7 +588,7 @@ impl Server { capability: Capability::Metadata, ref_: Some(input.ref_.as_str()), source: None, - error_code: None, + error_code: session_err, size_bytes: None, license: None, store_path: None, @@ -589,7 +662,7 @@ impl Server { // Step 3: non-dry-run path. Build foundation modules + open // FsStore + dispatch through core orchestrator. - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(fetch_paper_error_envelope( @@ -645,7 +718,34 @@ impl Server { let outcome = core_fetch_paper(&ref_, &self.profile, &ctx, &store, &store_root).await; - let session_ok = outcome.is_ok(); + // #507, second surface. `core_fetch_paper` returns `Ok` with a FAILED + // leg when an OA URL was found and refused, so `Result::is_ok` alone + // calls that a clean success -- for the one outcome an agent is most + // likely to retry. The CLI sibling special-cases exactly this; the + // first pass at this fix ported only the `error_code` half below, so + // the row said `result: "ok"` AND carried a code. Both halves now go + // through `FetchPaperOutcome::is_clean_success`. + let session_ok = outcome + .as_ref() + .is_ok_and(FetchPaperOutcome::is_clean_success); + let blocked_code = match outcome.as_ref() { + Ok(o) => match &o.pdf_leg { + doiget_core::orchestrator::PdfLegStatus::Blocked { code, .. } => { + Some(code.as_wire()) + } + _ => None, + }, + Err(_) => None, + }; + // #507: the bookend recorded THAT the call failed and not WHAT it + // failed with, so the provenance log could not answer "what did this + // session tell the caller about this ref?" -- which is the question + // repeat suppression has to ask before it can suppress anything. + let session_err = outcome + .as_ref() + .err() + .map(|e| ErrorCode::from(e).as_wire()) + .or(blocked_code); let _ = ctx.log.append(RowInput { event: LogEvent::SessionEnd, result: if session_ok { @@ -656,7 +756,7 @@ impl Server { capability: Capability::Oa, ref_: Some(input.ref_.as_str()), source: None, - error_code: None, + error_code: session_err, size_bytes: None, license: None, store_path: None, @@ -754,7 +854,7 @@ impl Server { // Step 4: non-dry-run — stand up the shared FetchContext + // store, emit a single SessionStart row, fan out via the core // orchestrator. - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(batch_fetch_error_envelope( @@ -806,10 +906,19 @@ impl Server { let batch_outcome = core_batch_fetch(&parsed, &self.profile, &ctx, &store, &store_root).await; - let session_ok = batch_outcome - .as_ref() - .map(|b| b.results.iter().all(|r| r.outcome.is_ok())) - .unwrap_or(false); + // #507, third and fourth surfaces. `.is_ok()` on a `BatchResultEntry` + // calls a Blocked PDF leg a success, exactly as the single-ref tool + // did before this release fixed it -- and the first pass at that fix + // stopped at `doiget_fetch_paper`, leaving the two batch tools writing + // `result: ok` for a call where every entry was refused. One boundary, + // `FetchPaperOutcome::is_clean_success`, for all four. + let session_ok = batch_outcome.as_ref().is_ok_and(|b| { + b.results.iter().all(|r| { + r.outcome + .as_ref() + .is_ok_and(FetchPaperOutcome::is_clean_success) + }) + }); let _ = ctx.log.append(RowInput { event: LogEvent::SessionEnd, @@ -959,10 +1068,32 @@ impl Server { "entry_key": entry_key, "ref": raw, "ok": false, - "error": { - "code": ErrorCode::InvalidRef, - "message": source.to_string(), - }, + "error": error_object(ErrorCode::InvalidRef, source.to_string()), + })); + } + // #500: NOT_IMPLEMENTED, not INVALID_REF. The entry is fine; + // doiget is what is missing, and the two codes carry opposite + // advice -- "wait for a release" versus "correct your input". + // Without this arm it fell into the wildcard below and became + // "unhandled bibliography parse error", losing the identifier + // it had just identified. + Err(doiget_core::refs::ParseError::UnsupportedIdentifier { + kind, + value, + entry_key, + }) => { + let msg = doiget_core::refs::unsupported_identifier_claim(kind, &value); + if input.strict { + return Ok(CallToolResult::structured(batch_fetch_error_envelope( + ErrorCode::NotImplemented, + &format!("{msg} (strict mode aborts)"), + ))); + } + parse_errors.push(json!({ + "entry_key": entry_key, + "ref": Value::Null, + "ok": false, + "error": error_object(ErrorCode::NotImplemented, msg), })); } Err(doiget_core::refs::ParseError::NoIdentifier { entry_key }) => { @@ -979,10 +1110,7 @@ impl Server { "entry_key": entry_key, "ref": Value::Null, "ok": false, - "error": { - "code": ErrorCode::InvalidRef, - "message": "entry has no DOI / arXiv id", - }, + "error": error_object(ErrorCode::InvalidRef, "entry has no DOI / arXiv id"), })); } Err(_) => { @@ -993,10 +1121,7 @@ impl Server { "entry_key": Value::Null, "ref": Value::Null, "ok": false, - "error": { - "code": ErrorCode::InvalidRef, - "message": "unhandled bibliography parse error", - }, + "error": error_object(ErrorCode::InvalidRef, "unhandled bibliography parse error"), })); } } @@ -1017,7 +1142,7 @@ impl Server { // Step 5: stand up the shared context + store (mirrors // `doiget_batch_fetch`). A context-init failure aborts the // whole call. - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(batch_fetch_error_envelope( @@ -1070,11 +1195,14 @@ impl Server { let entry_keys: Vec> = to_fetch.iter().map(|(_, k)| k.clone()).collect(); let batch_outcome = core_batch_fetch(&refs, &self.profile, &ctx, &store, &store_root).await; - let session_ok = batch_outcome - .as_ref() - .map(|b| b.results.iter().all(|r| r.outcome.is_ok())) - .unwrap_or(false) - && parse_errors.is_empty(); + // Same boundary as the sibling batch tool above (#507). + let session_ok = batch_outcome.as_ref().is_ok_and(|b| { + b.results.iter().all(|r| { + r.outcome + .as_ref() + .is_ok_and(FetchPaperOutcome::is_clean_success) + }) + }) && parse_errors.is_empty(); let _ = ctx.log.append(RowInput { event: LogEvent::SessionEnd, @@ -1282,7 +1410,8 @@ impl Server { OUTPUTS: { ok: true, scope: \"external\", query, total_results, count, results: [{ doi, openalex_id, arxiv, title, authors, year, venue, abstract, cited_by_count, oa_status, source }] } OR { ok:false, error }.\n\ COSTS: 1 OpenAlex request, plus 1 per supplied author/venue/publisher name to resolve.\n\ SIDE EFFECTS: Emits Metadata provenance rows. NEVER writes the store. NEVER fetches a PDF.\n\ - LIMITS: Tier-1, always-on (no DOIGET_ENABLE_OPENALEX gate). An ambiguous author/venue/publisher name → AMBIGUOUS (candidates listed); no match → NOT_FOUND.", + LIMITS: Tier-1, always-on (no DOIGET_ENABLE_OPENALEX gate). An ambiguous author/venue/publisher name → AMBIGUOUS (candidates listed); no match → NOT_FOUND.\n\ + ZERO RESULTS ARE NOT EVIDENCE OF ABSENCE: OpenAlex free-text matching degrades sharply past roughly 8 terms and returns nothing rather than a partial match. On total_results: 0, retry with 3-5 distinctive terms before concluding the work is not indexed; the envelope carries a `hint` saying so.", annotations( read_only_hint = false, destructive_hint = false, @@ -1348,7 +1477,7 @@ impl Server { let contact_email = doiget_core::orchestrator::configured_contact_email().unwrap_or_default(); - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(read_path_error_envelope( @@ -1380,6 +1509,11 @@ impl Server { let outcome = doiget_core::discovery::paper_search(&base, &contact_email, &q, &ctx).await; let session_ok = outcome.is_ok(); + // #507: the bookend recorded THAT the call failed and not WHAT it + // failed with, so the provenance log could not answer "what did this + // session tell the caller about this ref?" -- which is the question + // repeat suppression has to ask before it can suppress anything. + let session_err = outcome.as_ref().err().map(|e| ErrorCode::from(e).as_wire()); let _ = ctx.log.append(RowInput { event: LogEvent::SessionEnd, result: if session_ok { @@ -1390,7 +1524,7 @@ impl Server { capability: Capability::Metadata, ref_: None, source: None, - error_code: None, + error_code: session_err, size_bytes: None, license: None, store_path: None, @@ -1398,14 +1532,32 @@ impl Server { }); match outcome { - Ok(results) => Ok(CallToolResult::structured(json!({ - "ok": true, - "scope": "external", - "query": input.query, - "total_results": results.total_results, - "count": results.results.len(), - "results": results.results, - }))), + Ok(results) => { + let mut envelope = json!({ + "ok": true, + "scope": "external", + "query": input.query, + "total_results": results.total_results, + "count": results.results.len(), + "results": results.results, + }); + // A zero-result search is a SUCCESS envelope, so #506's work on + // error dispositions does not reach it. An agent reads + // `ok: true` with an empty array as a fact about the world -- + // the paper is not indexed -- and stops looking. That happened: + // eleven consecutive searches returned 0 for papers a shorter + // query then found on the first try (#534). + // + // The cause is query length, not absence, so the envelope says + // so at the exact point an agent would otherwise conclude + // absence. + if results.results.is_empty() { + if let Some(hint) = doiget_core::discovery::zero_result_hint(&input.query) { + envelope["hint"] = json!(hint); + } + } + Ok(CallToolResult::structured(envelope)) + } // Canonical FetchError -> ErrorCode (AMBIGUOUS / NOT_FOUND / // NETWORK_ERROR / …) so an agent can branch on the code. Err(e) => Ok(CallToolResult::structured(read_path_error_envelope( @@ -1420,7 +1572,7 @@ impl Server { /// #281 "read" step; ADR-0032). Fetches the ar5iv LaTeXML-XHTML /// rendering of an **arXiv** paper and returns it as sectioned plain /// text. The PDF blob is never opened (ADR-0032 D1). Tier-1 OA, - /// always-on. A bare DOI returns `NO_OA_AVAILABLE` (DOI→arXiv linking + /// always-on. A bare DOI returns `NOT_IMPLEMENTED` (DOI→arXiv linking /// is #281 item 5). #[tool( description = "WHEN TO USE: Read a paper's full text (arXiv only) without an external pdf-to-text tool — the 'read' step after discovery/fetch.\n\ @@ -1428,7 +1580,7 @@ impl Server { OUTPUTS: { ok: true, arxiv_id, source: \"ar5iv\", title, sections: [{ heading, text }], char_count, truncated, retrieved_from } OR { ok:false, error }.\n\ COSTS: 1 ar5iv HTTP request (HTML), then parse; large papers can be sizeable — use max_chars to bound.\n\ SIDE EFFECTS: Emits an OA provenance row. NEVER opens the PDF blob; NEVER writes the store.\n\ - LIMITS: arXiv only (a DOI → NO_OA_AVAILABLE; pass the arXiv id). A paper not converted by ar5iv → NOT_FOUND. Best-effort extraction (truncation flagged on `truncated`).", + LIMITS: arXiv only (a DOI → NOT_IMPLEMENTED; pass the arXiv id). A paper not converted by ar5iv → NOT_FOUND. Best-effort extraction (truncation flagged on `truncated`).", annotations( read_only_hint = true, destructive_hint = false, @@ -1441,13 +1593,22 @@ impl Server { Parameters(input): Parameters, ) -> Result { // Validate the ref. A DOI has no full-text source in this slice; - // report NO_OA_AVAILABLE rather than silently failing (ADR-0032 D5). + // report the absence rather than silently failing (ADR-0032 D5). let id = match doiget_core::Ref::parse(&input.ref_) { Ok(doiget_core::Ref::Arxiv(a)) => a, Ok(doiget_core::Ref::Doi(_)) => { return Ok(CallToolResult::structured(read_path_error_envelope( Some(&input.ref_), - ErrorCode::NoOaAvailable, + // Not `NO_OA_AVAILABLE`. That code's disposition is + // `needs_config` -- "a named change makes it" -- and there + // is no config knob here: this tool is arXiv-only and + // DOI to arXiv linking is not built. `NOT_IMPLEMENTED` is + // terminal and says the true thing (ERRORS.md: "wait for + // next minor release; do not retry"), and it is the code + // `verify` / `batch_from_bibliography` already give the + // same situation for PMIDs (#500): valid input, absent + // support. + ErrorCode::NotImplemented, "no full-text source for a DOI — pass the arXiv id if a preprint exists \ (DOI→arXiv linking is #281 item 5)", ))); @@ -1476,7 +1637,7 @@ impl Server { // for now, mirroring `build_fetch_context`'s resolver-cache note; // enabling it here is a follow-up. Correctness is unaffected — a // cache miss just re-fetches. - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(read_path_error_envelope( @@ -1549,7 +1710,7 @@ impl Server { OUTPUTS: { ok: true, arxiv_id, main_file, tex_source, char_count, truncated, retrieved_from } OR { ok:false, error }.\n\ COSTS: 1 arXiv source API request (gzip'd tar download); large papers can be sizeable — use max_chars to bound.\n\ SIDE EFFECTS: Emits an OA provenance row. NEVER writes the store; NEVER opens a PDF blob.\n\ - LIMITS: arXiv only (a DOI → NO_OA_AVAILABLE; pass the arXiv id). PDF-only submissions → TEXT_UNAVAILABLE.", + LIMITS: arXiv only (a DOI → NOT_IMPLEMENTED; pass the arXiv id). PDF-only submissions → TEXT_UNAVAILABLE.", annotations( read_only_hint = true, destructive_hint = false, @@ -1566,7 +1727,9 @@ impl Server { Ok(doiget_core::Ref::Doi(_)) => { return Ok(CallToolResult::structured(read_path_error_envelope( Some(&input.ref_), - ErrorCode::NoOaAvailable, + // Same reasoning as `doiget_paper_text` above: absent + // support, not absent open access. + ErrorCode::NotImplemented, "no TeX source for a DOI — pass the arXiv id if a preprint exists", ))); } @@ -1590,7 +1753,7 @@ impl Server { } }; - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(read_path_error_envelope( @@ -1723,7 +1886,7 @@ impl Server { let contact_email = doiget_core::orchestrator::configured_contact_email().unwrap_or_default(); - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(read_path_error_envelope( @@ -1977,10 +2140,10 @@ impl Server { let _ = input; return Ok(CallToolResult::structured(json!({ "ok": false, - "error": { - "code": ErrorCode::NotImplemented, - "message": "doiget_expand_citation_graph requires the `citation` Cargo feature; this binary was built without it", - }, + "error": error_object( + ErrorCode::NotImplemented, + "doiget_expand_citation_graph requires the `citation` Cargo feature; this binary was built without it", + ), }))); } #[cfg(feature = "citation")] @@ -2006,7 +2169,7 @@ impl Server { } }; - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(read_path_error_envelope( @@ -2066,6 +2229,17 @@ impl Server { doiget_core::citation_graph::expand(&doi, caps, &source, &self.profile, &ctx).await; let session_ok = outcome.is_ok(); + // #507: the bookend recorded THAT the call failed and not WHAT it + // failed with, so the provenance log could not answer "what did this + // session tell the caller about this ref?" -- which is the question + // repeat suppression has to ask before it can suppress anything. + // `GraphError`, not `FetchError` -- it has no `From<&_> for + // ErrorCode`, and this site is behind `citation`, so an oa-only + // build never compiled it. Same mapping the response arm uses. + let session_err = outcome + .as_ref() + .err() + .map(|e| graph_error_code(e).as_wire()); let _ = ctx.log.append(RowInput { event: LogEvent::SessionEnd, result: if session_ok { @@ -2076,7 +2250,7 @@ impl Server { capability: Capability::Metadata, ref_: Some(input.ref_.as_str()), source: None, - error_code: None, + error_code: session_err, size_bytes: None, license: None, store_path: None, @@ -2093,21 +2267,11 @@ impl Server { "truncated": graph.truncated, "total_visited": graph.total_visited, }))), + // Was a third copy of the same match, inline, which did not + // even call the helper directly above it. Err(e) => Ok(CallToolResult::structured(read_path_error_envelope( Some(&input.ref_), - match &e { - doiget_core::citation_graph::GraphError::CapabilityDenied => { - ErrorCode::CapabilityDenied - } - doiget_core::citation_graph::GraphError::SeedNotIndexed => { - ErrorCode::NoOaAvailable - } - doiget_core::citation_graph::GraphError::Log(_) => ErrorCode::LogError, - doiget_core::citation_graph::GraphError::Source(_) => { - ErrorCode::NetworkError - } - _ => ErrorCode::InternalError, - }, + graph_error_code(&e), &format!("citation graph expansion failed: {e}"), ))), } @@ -2172,10 +2336,10 @@ impl Server { #[tool( description = "WHEN TO USE: Resolve a free-form bibliographic citation string (e.g. 'Onsager 1944') to ranked DOI candidates.\n\ INPUTS: query (bibliographic citation query string), limit (maximum number of candidates to return, default: 5).\n\ - OUTPUTS: { ok: true, query, candidates: [ { doi, title, author, year, score, source } ] } OR { ok: false, error }.\n\ + OUTPUTS: { ok: true, query, candidates: [ { doi, title, author, year, score, confidence, matched, source } ] } OR { ok: false, error }.\n\ COSTS: 1-2 s round-trip.\n\ SIDE EFFECTS: none.\n\ - LIMITS: Returns candidates with similarity score >= 0.5.", + LIMITS: Returns candidates with similarity score >= 0.5. BRANCH ON confidence, NOT score: the score is token overlap against your query string and 0.5 is the FLOOR, so the worst candidate this tool can emit still looks like a positive number. confidence is exact (every query token matched), probable (four in five), or weak (cleared the floor and no more) - for a known-item lookup a weak candidate is a near-miss, not a match, so verify it with doiget_resolve_paper before citing. matched lists which of your tokens were found, which is how you see whether the author and the journal were among them.", annotations( read_only_hint = true, destructive_hint = false, @@ -2187,12 +2351,15 @@ impl Server { &self, Parameters(input): Parameters, ) -> Result { - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(serde_json::json!({ "ok": false, - "error": format!("context initialization failed: {e}"), + "error": error_object( + ErrorCode::InternalError, + format!("context initialization failed: {e}"), + ), }))); } }; @@ -2202,7 +2369,7 @@ impl Server { Err(e) => { return Ok(CallToolResult::structured(serde_json::json!({ "ok": false, - "error": e, + "error": error_object(ErrorCode::CapabilityDenied, e), }))); } }; @@ -2218,7 +2385,7 @@ impl Server { }))), Err(e) => Ok(CallToolResult::structured(serde_json::json!({ "ok": false, - "error": format!("resolve failed: {e}"), + "error": error_object(ErrorCode::from(&e), format!("resolve failed: {e}")), }))), } } @@ -2227,10 +2394,10 @@ impl Server { #[tool( description = "WHEN TO USE: Resolve multiple free-form bibliographic citation strings in batch.\n\ INPUTS: queries (array of query strings), limit (maximum number of candidates per query, default: 5).\n\ - OUTPUTS: { ok: true, results: [ { query, candidates: [ { doi, title, author, year, score, source } ] } ] } OR { ok: false, error }.\n\ + OUTPUTS: { ok: true, results: [ { query, candidates: [ { doi, title, author, year, score, confidence, matched, source } ] } ] } OR { ok: false, error }.\n\ COSTS: 1-2 s round-trip per query.\n\ SIDE EFFECTS: none.\n\ - LIMITS: Returns candidates with similarity score >= 0.5. At most 50 queries per call.", + LIMITS: Returns candidates with similarity score >= 0.5. At most 50 queries per call. BRANCH ON confidence, NOT score: the score is token overlap against your query string and 0.5 is the FLOOR, so the worst candidate this tool can emit still looks like a positive number. confidence is exact (every query token matched), probable (four in five), or weak (cleared the floor and no more) - for a known-item lookup a weak candidate is a near-miss, not a match, so verify it with doiget_resolve_paper before citing. matched lists which of your tokens were found, which is how you see whether the author and the journal were among them.", annotations( read_only_hint = true, destructive_hint = false, @@ -2245,16 +2412,22 @@ impl Server { if input.queries.len() > 50 { return Ok(CallToolResult::structured(serde_json::json!({ "ok": false, - "error": "At most 50 queries per call.", + "error": error_object( + ErrorCode::InvalidRef, + "At most 50 queries per call.", + ), }))); } - let ctx = match build_fetch_context() { + let ctx = match self.fetch_context() { Ok(c) => c, Err(e) => { return Ok(CallToolResult::structured(serde_json::json!({ "ok": false, - "error": format!("context initialization failed: {e}"), + "error": error_object( + ErrorCode::InternalError, + format!("context initialization failed: {e}"), + ), }))); } }; @@ -2264,7 +2437,7 @@ impl Server { Err(e) => { return Ok(CallToolResult::structured(serde_json::json!({ "ok": false, - "error": e, + "error": error_object(ErrorCode::CapabilityDenied, e), }))); } }; @@ -2281,7 +2454,10 @@ impl Server { Err(e) => { results.push(serde_json::json!({ "query": query, - "error": format!("resolve failed: {e}"), + "error": error_object( + ErrorCode::from(&e), + format!("resolve failed: {e}"), + ), })); } } @@ -2355,7 +2531,26 @@ impl Server { Ok(None) => { return Ok(CallToolResult::structured(json!({ "ok": false, - "error": format!("no store entry for {}; fetch the paper first", input.ref_), + // NOT `NOT_FOUND`. docs/ERRORS.md defines that as "a + // metadata source authoritatively reported the id does not + // exist ... doiget verify treats it as a definite dead + // reference", so an agent reading it would conclude the DOI + // is retracted or mistyped when the truth is that nobody has + // fetched it yet. `doiget_info` / `doiget_paper_pdf_path` / + // `doiget_search_local` all decline to call a store miss + // NOT_FOUND for exactly this reason; they answer ok:true with + // a null payload. These two mutate an entry, so they cannot, + // but they must not make the stronger claim either. + // + // `STORE_ERROR` is the closest fit in the closed set and its + // disposition -- `needs_config`, "will not change by itself; + // a named change makes it" -- is exactly right: the entry + // appears when you fetch it, which the message says. The code + // is approximate and the closed set has a real gap here. + "error": error_object( + ErrorCode::StoreError, + format!("no store entry for {}; fetch the paper first", input.ref_), + ), }))); } Err(e) => { @@ -2372,7 +2567,13 @@ impl Server { None => { return Ok(CallToolResult::structured(json!({ "ok": false, - "error": format!("entry {} has no [doiget] table; fetch it first", input.ref_), + // Same reasoning as the store-miss arm above: the entry + // exists and is incomplete, which is not "the id does not + // exist". + "error": error_object( + ErrorCode::StoreError, + format!("entry {} has no [doiget] table; fetch it first", input.ref_), + ), }))); } }; @@ -2397,7 +2598,7 @@ impl Server { let tags = ext.tags.clone(); let collections = ext.collections.clone(); - match store.write(&safekey, &metadata, None) { + match store.write_user_authored(&safekey, &metadata, None) { Ok(()) => Ok(CallToolResult::structured(json!({ "ok": true, "ref": input.ref_, @@ -2473,7 +2674,26 @@ impl Server { Ok(None) => { return Ok(CallToolResult::structured(json!({ "ok": false, - "error": format!("no store entry for {}; fetch the paper first", input.ref_), + // NOT `NOT_FOUND`. docs/ERRORS.md defines that as "a + // metadata source authoritatively reported the id does not + // exist ... doiget verify treats it as a definite dead + // reference", so an agent reading it would conclude the DOI + // is retracted or mistyped when the truth is that nobody has + // fetched it yet. `doiget_info` / `doiget_paper_pdf_path` / + // `doiget_search_local` all decline to call a store miss + // NOT_FOUND for exactly this reason; they answer ok:true with + // a null payload. These two mutate an entry, so they cannot, + // but they must not make the stronger claim either. + // + // `STORE_ERROR` is the closest fit in the closed set and its + // disposition -- `needs_config`, "will not change by itself; + // a named change makes it" -- is exactly right: the entry + // appears when you fetch it, which the message says. The code + // is approximate and the closed set has a real gap here. + "error": error_object( + ErrorCode::StoreError, + format!("no store entry for {}; fetch the paper first", input.ref_), + ), }))); } Err(e) => { @@ -2490,7 +2710,13 @@ impl Server { None => { return Ok(CallToolResult::structured(json!({ "ok": false, - "error": format!("entry {} has no [doiget] table; fetch it first", input.ref_), + // Same reasoning as the store-miss arm above: the entry + // exists and is incomplete, which is not "the id does not + // exist". + "error": error_object( + ErrorCode::StoreError, + format!("entry {} has no [doiget] table; fetch it first", input.ref_), + ), }))); } }; @@ -2501,20 +2727,26 @@ impl Server { if text.is_empty() { return Ok(CallToolResult::structured(json!({ "ok": false, - "error": "annotation text must not be empty; set clear:true to remove it", + "error": error_object( + ErrorCode::InvalidRef, + "annotation text must not be empty; set clear:true to remove it", + ), }))); } ext.annotation = Some(text.clone()); } else { return Ok(CallToolResult::structured(json!({ "ok": false, - "error": "provide 'text' to set an annotation, or 'clear: true' to remove it", + "error": error_object( + ErrorCode::InvalidRef, + "provide 'text' to set an annotation, or 'clear: true' to remove it", + ), }))); } let annotation = ext.annotation.clone(); - match store.write(&safekey, &metadata, None) { + match store.write_user_authored(&safekey, &metadata, None) { Ok(()) => Ok(CallToolResult::structured(json!({ "ok": true, "ref": input.ref_, @@ -2558,6 +2790,32 @@ pub struct MetadataOnlyInput { /// either omit the field or pass `false`. #[serde(default)] pub dry_run: bool, + /// Consult Unpaywall for a real OA location, filling `oa_url`, + /// `oa_status` and `license`. Costs one extra metadata round-trip. + /// + /// Defaults to `false`. On the default path `oa_url` is `null` whenever + /// Crossref answered -- which is nearly every DOI -- because Crossref's + /// `link[]` is a programme-scoped channel (Similarity Check, TDM, + /// syndication), never a general-purpose OA URL (#517). Before #539 the + /// field was advertised without that caveat, so an agent could read a + /// permanent `null` as "this work has no OA location". + /// + /// It is NOT unconditionally null without the flag, and an earlier + /// version of this doc said it was. `metadata_only_doi` keeps a + /// pre-existing fallback: when Crossref FAILS, Unpaywall is consulted + /// regardless of this flag, and `oa_url` / `oa_status` come from that + /// record. `source` distinguishes the two -- `crossref` means the + /// default path answered, `unpaywall` means the fallback did. + /// + /// Set it and `oa_status` tells you which answer you got: `"closed"` + /// means the lookup completed and there is no OA location, while + /// `null` means the lookup itself did not complete. + /// + /// Plain `bool` rather than `Option` for the same reason as + /// `dry_run`: a wire `null` should be rejected at deserialize time + /// rather than silently meaning "no". + #[serde(default)] + pub include_oa_location: bool, } /// JSON-schema-derived input for the `doiget_resolve_paper` MCP tool. @@ -2577,6 +2835,32 @@ pub struct ResolvePaperInput { #[serde(rename = "ref")] #[schemars(rename = "ref")] pub ref_: String, + /// Consult Unpaywall for a real OA location, filling `oa_url`, + /// `oa_status` and `license`. Costs one extra metadata round-trip. + /// + /// Defaults to `false`. On the default path `oa_url` is `null` whenever + /// Crossref answered -- which is nearly every DOI -- because Crossref's + /// `link[]` is a programme-scoped channel (Similarity Check, TDM, + /// syndication), never a general-purpose OA URL (#517). Before #539 the + /// field was advertised without that caveat, so an agent could read a + /// permanent `null` as "this work has no OA location". + /// + /// It is NOT unconditionally null without the flag, and an earlier + /// version of this doc said it was. `metadata_only_doi` keeps a + /// pre-existing fallback: when Crossref FAILS, Unpaywall is consulted + /// regardless of this flag, and `oa_url` / `oa_status` come from that + /// record. `source` distinguishes the two -- `crossref` means the + /// default path answered, `unpaywall` means the fallback did. + /// + /// Set it and `oa_status` tells you which answer you got: `"closed"` + /// means the lookup completed and there is no OA location, while + /// `null` means the lookup itself did not complete. + /// + /// Plain `bool` rather than `Option` for the same reason as + /// `dry_run`: a wire `null` should be rejected at deserialize time + /// rather than silently meaning "no". + #[serde(default)] + pub include_oa_location: bool, } // --------------------------------------------------------------------------- @@ -2687,7 +2971,7 @@ pub struct PaperSearchInput { #[schemars(deny_unknown_fields)] pub struct PaperTextInput { /// arXiv id (e.g. "arxiv:2401.12345"), validated via `Ref::parse`. A - /// bare DOI returns `NO_OA_AVAILABLE` (pass the arXiv id; + /// bare DOI returns `NOT_IMPLEMENTED` (pass the arXiv id; /// DOI→arXiv linking is #281 item 5). #[serde(rename = "ref")] #[schemars(rename = "ref")] @@ -2704,7 +2988,7 @@ pub struct PaperTextInput { #[schemars(deny_unknown_fields)] pub struct PaperTexSourceInput { /// arXiv id (e.g. "arxiv:2401.12345"), validated via `Ref::parse`. A - /// bare DOI returns `NO_OA_AVAILABLE` (pass the arXiv id). + /// bare DOI returns `NOT_IMPLEMENTED` (pass the arXiv id). #[serde(rename = "ref")] #[schemars(rename = "ref")] pub ref_: String, @@ -2991,7 +3275,7 @@ impl Server { Err(e) => { entries.push(json!({ "ref": r, - "error": { "code": ErrorCode::InvalidRef, "message": format!("invalid ref: {e}") }, + "error": error_object(ErrorCode::InvalidRef, format!("invalid ref: {e}")), })); continue; } @@ -3028,7 +3312,7 @@ impl Server { Err(e) => { entries.push(json!({ "ref": r, - "error": { "code": ErrorCode::StoreError, "message": format!("store read failed: {e}") }, + "error": error_object(ErrorCode::StoreError, format!("store read failed: {e}")), })); } } @@ -3083,14 +3367,44 @@ fn entry_info_to_json(entry: &EntryInfo) -> Value { /// per-ref context). This shape-symmetry with the success envelopes /// (which always carry `"ref"`) means consumers can pattern-match /// uniformly across `ok:true` / `ok:false` envelopes. +/// The closed-set code a `GraphError` is reported as (#507). +/// +/// Extracted from the inline `match` in `doiget_expand_citation_graph`'s +/// response arm so the SessionEnd bookend and the caller-facing envelope +/// cannot say different things about the same error. +#[cfg(feature = "citation")] +fn graph_error_code(e: &doiget_core::citation_graph::GraphError) -> ErrorCode { + // Delegate. This copy flattened `Source(_)` to `NETWORK_ERROR`, so an + // OpenAlex 404 during expansion arrived here as `retry_after` while the + // CLI called the same condition `NOT_FOUND` -- one error, two answers, + // and the retriable one was wrong. + ErrorCode::from(e) +} + +/// The `error` object every failure envelope carries (#506). +/// +/// One builder rather than ten literals, because the point of `disposition` +/// is that an agent can rely on it being there. A field present on some +/// failures and absent on others is worse than no field: it teaches the reader +/// to fall back to guessing from the code's name, which is the habit this +/// exists to replace. +/// +/// `disposition` is derived by [`doiget_core::ErrorCode::disposition`], which +/// is the same function `docs/ERRORS.md` §2's Disposition column is asserted +/// against. +fn error_object(code: ErrorCode, message: impl Into) -> Value { + json!({ + "code": code, + "message": message.into(), + "disposition": code.disposition().as_wire(), + }) +} + fn read_path_error_envelope(ref_str: Option<&str>, code: ErrorCode, message: &str) -> Value { json!({ "ok": false, "ref": ref_str.map(Value::from).unwrap_or(Value::Null), - "error": { - "code": code, - "message": message, - }, + "error": error_object(code, message), }) } @@ -3115,15 +3429,11 @@ fn metadata_only_error_envelope(ref_str: Option<&str>, code: ErrorCode, message: // ref to surface) for shape-symmetry with the success // envelopes and `read_path_error_envelope`. "ref": ref_str.map(Value::from).unwrap_or(Value::Null), - "error": { - "code": code, - "message": message, - // denial_context is intentionally absent for these envelope - // shapes (parse-error / not-implemented); ADR-0023 §1 says - // the field is optional and consumers MUST tolerate it - // being absent (§3 covers the per-subfield optionality - // rules that apply when denial_context IS present). - }, + // denial_context is intentionally absent for these envelope shapes + // (parse-error / not-implemented); ADR-0023 §1 says the field is + // optional and consumers MUST tolerate it being absent (§3 covers the + // per-subfield optionality rules that apply when it IS present). + "error": error_object(code, message), }) } @@ -3196,6 +3506,17 @@ fn metadata_only_fetch_error_envelope(err: &FetchError, ref_str: &str) -> Value let mut error_obj = serde_json::Map::new(); error_obj.insert("code".into(), json!(code)); error_obj.insert("message".into(), json!(message)); + // #506: these five objects are assembled by hand rather than through + // `error_object`, so the first pass at the disposition missed them -- + // including the two most common failures an agent sees. A field that is + // present on some failures and absent on others is worse than none. + error_obj.insert("disposition".into(), json!(code.disposition().as_wire())); + // #506: only when the SERVER sent one. Absent otherwise rather than + // backfilled from our own backoff, which would be a guess wearing the + // name of a measurement. + if let Some(ms) = doiget_core::source::retry_after_ms(err) { + error_obj.insert("retry_after_ms".into(), json!(ms)); + } if let Some(dc) = denial { // `DenialContext` is `Serialize` (`#[serde(deny_unknown_fields)]`, // optional fields) and `serde_json::to_value` cannot fail on a @@ -3208,6 +3529,16 @@ fn metadata_only_fetch_error_envelope(err: &FetchError, ref_str: &str) -> Value "denial_context".into(), denial_context_to_value(&dc, "metadata_only"), ); + // #506: `denial_context` says what was refused; this says what to do + // about it. `docs/ERRORS.md` §3 used to state outright that + // remediation "belongs to the ok:true envelope" -- so the one field + // naming the fix was present when the call succeeded with a blocked + // leg and absent when the call actually failed. Same core function + // the blocked leg and the CLI `= help:` block use. + let r = doiget_core::remediation::for_denial(&dc); + if !r.is_empty() { + error_obj.insert("remediation".into(), json!(r)); + } } json!({ "ok": false, @@ -3270,6 +3601,11 @@ fn pdf_leg_json(leg: &PdfLegStatus) -> Value { o.insert("status".into(), json!("blocked")); o.insert("code".into(), json!(code)); o.insert("message".into(), json!(message)); + // #506: these five objects are assembled by hand rather than through + // `error_object`, so the first pass at the disposition missed them -- + // including the two most common failures an agent sees. A field that is + // present on some failures and absent on others is worse than none. + o.insert("disposition".into(), json!(code.disposition().as_wire())); if let Some(dc) = denial { // Route through the logged helper (#154): a bare // `json!(dc)` here would silently coerce a future @@ -3367,38 +3703,50 @@ fn fetch_paper_error_envelope(ref_str: Option<&str>, code: ErrorCode, message: & if let Some(r) = ref_str { obj.insert("ref".into(), json!(r)); } - obj.insert( - "error".into(), - json!({ - "code": code, - "message": message, - }), - ); + obj.insert("error".into(), error_object(code, message)); Value::Object(obj) } /// Build the `{ok:false, error:{code, message, denial_context?}}` /// envelope for orchestrator failures in `doiget_fetch_paper`. fn fetch_paper_fetch_error_envelope(err: &FetchError, ref_str: &str) -> Value { - let code: ErrorCode = match err { - FetchError::NotEligible { .. } => ErrorCode::CapabilityDenied, - FetchError::NoOaAvailable => ErrorCode::NoOaAvailable, - FetchError::Http(_) => ErrorCode::NetworkError, - FetchError::Log(_) => ErrorCode::LogError, - FetchError::InvalidRef(_) => ErrorCode::InvalidRef, - FetchError::SourceSchema { .. } => ErrorCode::InternalError, - FetchError::TooManyRefs { .. } => ErrorCode::InvalidRef, - _ => ErrorCode::InternalError, - }; + // Delegate, do not re-implement. This was a hand-rolled copy of + // `From<&FetchError> for ErrorCode` ending in `_ => InternalError`, so + // every variant it had not enumerated -- including `NotFound`, which a + // mistyped DOI produces on the shipped build -- was reported to the caller + // as an internal error. The canonical mapping is exhaustive and is what + // the rest of this file already calls. + let code: ErrorCode = ErrorCode::from(err); let denial: Option = err.into(); let mut error_obj = serde_json::Map::new(); error_obj.insert("code".into(), json!(code)); error_obj.insert("message".into(), json!(err.to_string())); + // #506: these five objects are assembled by hand rather than through + // `error_object`, so the first pass at the disposition missed them -- + // including the two most common failures an agent sees. A field that is + // present on some failures and absent on others is worse than none. + error_obj.insert("disposition".into(), json!(code.disposition().as_wire())); + // #506: only when the SERVER sent one. Absent otherwise rather than + // backfilled from our own backoff, which would be a guess wearing the + // name of a measurement. + if let Some(ms) = doiget_core::source::retry_after_ms(err) { + error_obj.insert("retry_after_ms".into(), json!(ms)); + } if let Some(dc) = denial { error_obj.insert( "denial_context".into(), denial_context_to_value(&dc, "fetch_paper"), ); + // #506: `denial_context` says what was refused; this says what to do + // about it. `docs/ERRORS.md` §3 used to state outright that + // remediation "belongs to the ok:true envelope" -- so the one field + // naming the fix was present when the call succeeded with a blocked + // leg and absent when the call actually failed. Same core function + // the blocked leg and the CLI `= help:` block use. + let r = doiget_core::remediation::for_denial(&dc); + if !r.is_empty() { + error_obj.insert("remediation".into(), json!(r)); + } } json!({ "ok": false, @@ -3487,25 +3835,46 @@ fn build_bibliography_envelope( "pdf": pdf_leg_json(&outcome.pdf_leg), }), Err(err) => { - let code: ErrorCode = match err { - FetchError::NotEligible { .. } => ErrorCode::CapabilityDenied, - FetchError::NoOaAvailable => ErrorCode::NoOaAvailable, - FetchError::Http(_) => ErrorCode::NetworkError, - FetchError::Log(_) => ErrorCode::LogError, - FetchError::InvalidRef(_) => ErrorCode::InvalidRef, - FetchError::SourceSchema { .. } => ErrorCode::InternalError, - FetchError::TooManyRefs { .. } => ErrorCode::InvalidRef, - _ => ErrorCode::InternalError, - }; + // Delegate, do not re-implement (#538). This was the same + // hand-rolled copy `fetch_paper_fetch_error_envelope` shed one + // screen up, left behind on the two batch tools: ending in + // `_ => InternalError` it reported `NotFound`, `Ambiguous` and + // -- once #538 landed -- `NotRetrievable` as internal errors, + // and collapsed every `Http` status to `NETWORK_ERROR`, so a + // 404 told an agent to retry a DOI that will never resolve. + // `disposition` is derived from this code below, so a wrong + // code was also a wrong disposition. + let code: ErrorCode = ErrorCode::from(err); let denial: Option = err.into(); let mut error_obj = serde_json::Map::new(); error_obj.insert("code".into(), json!(code)); error_obj.insert("message".into(), json!(err.to_string())); + // #506: these five objects are assembled by hand rather than through + // `error_object`, so the first pass at the disposition missed them -- + // including the two most common failures an agent sees. A field that is + // present on some failures and absent on others is worse than none. + error_obj.insert("disposition".into(), json!(code.disposition().as_wire())); + // #506: only when the SERVER sent one. Absent otherwise rather than + // backfilled from our own backoff, which would be a guess wearing the + // name of a measurement. + if let Some(ms) = doiget_core::source::retry_after_ms(err) { + error_obj.insert("retry_after_ms".into(), json!(ms)); + } if let Some(dc) = denial { error_obj.insert( "denial_context".into(), denial_context_to_value(&dc, "batch_from_bibliography"), ); + // #506: `denial_context` says what was refused; this says what to do + // about it. `docs/ERRORS.md` §3 used to state outright that + // remediation "belongs to the ok:true envelope" -- so the one field + // naming the fix was present when the call succeeded with a blocked + // leg and absent when the call actually failed. Same core function + // the blocked leg and the CLI `= help:` block use. + let r = doiget_core::remediation::for_denial(&dc); + if !r.is_empty() { + error_obj.insert("remediation".into(), json!(r)); + } } else { error_obj.insert("denial_context".into(), Value::Null); } @@ -3574,25 +3943,46 @@ fn batch_fetch_success_envelope( "pdf": pdf_leg_json(&outcome.pdf_leg), }), Err(err) => { - let code: ErrorCode = match err { - FetchError::NotEligible { .. } => ErrorCode::CapabilityDenied, - FetchError::NoOaAvailable => ErrorCode::NoOaAvailable, - FetchError::Http(_) => ErrorCode::NetworkError, - FetchError::Log(_) => ErrorCode::LogError, - FetchError::InvalidRef(_) => ErrorCode::InvalidRef, - FetchError::SourceSchema { .. } => ErrorCode::InternalError, - FetchError::TooManyRefs { .. } => ErrorCode::InvalidRef, - _ => ErrorCode::InternalError, - }; + // Delegate, do not re-implement (#538). This was the same + // hand-rolled copy `fetch_paper_fetch_error_envelope` shed one + // screen up, left behind on the two batch tools: ending in + // `_ => InternalError` it reported `NotFound`, `Ambiguous` and + // -- once #538 landed -- `NotRetrievable` as internal errors, + // and collapsed every `Http` status to `NETWORK_ERROR`, so a + // 404 told an agent to retry a DOI that will never resolve. + // `disposition` is derived from this code below, so a wrong + // code was also a wrong disposition. + let code: ErrorCode = ErrorCode::from(err); let denial: Option = err.into(); let mut error_obj = serde_json::Map::new(); error_obj.insert("code".into(), json!(code)); error_obj.insert("message".into(), json!(err.to_string())); + // #506: these five objects are assembled by hand rather than through + // `error_object`, so the first pass at the disposition missed them -- + // including the two most common failures an agent sees. A field that is + // present on some failures and absent on others is worse than none. + error_obj.insert("disposition".into(), json!(code.disposition().as_wire())); + // #506: only when the SERVER sent one. Absent otherwise rather than + // backfilled from our own backoff, which would be a guess wearing the + // name of a measurement. + if let Some(ms) = doiget_core::source::retry_after_ms(err) { + error_obj.insert("retry_after_ms".into(), json!(ms)); + } if let Some(dc) = denial { error_obj.insert( "denial_context".into(), denial_context_to_value(&dc, "batch_fetch"), ); + // #506: `denial_context` says what was refused; this says what to do + // about it. `docs/ERRORS.md` §3 used to state outright that + // remediation "belongs to the ok:true envelope" -- so the one field + // naming the fix was present when the call succeeded with a blocked + // leg and absent when the call actually failed. Same core function + // the blocked leg and the CLI `= help:` block use. + let r = doiget_core::remediation::for_denial(&dc); + if !r.is_empty() { + error_obj.insert("remediation".into(), json!(r)); + } } else { // Per the Slice 2 spec: transport (NETWORK_ERROR) // entries carry `denial_context: null` so an agent @@ -3653,10 +4043,7 @@ fn build_batch_dry_run_envelope(plans: &[(Ref, doiget_core::dry_run::FetchPlan)] fn batch_fetch_error_envelope(code: ErrorCode, message: &str) -> Value { json!({ "ok": false, - "error": { - "code": code, - "message": message, - }, + "error": error_object(code, message), }) } @@ -3720,7 +4107,7 @@ fn paper_text_success_envelope(t: &doiget_core::paper_text::PaperText) -> Value /// if missing. /// - `session_id` — fresh 26-char ULID per call (one tool call = one /// logical session, per `docs/PROVENANCE_LOG.md` §3). -fn build_fetch_context() -> anyhow::Result { +fn open_provenance_log(session_id: String) -> anyhow::Result { let log_path = resolve_log_path()?; if let Some(parent) = log_path.parent() { if !parent.as_str().is_empty() { @@ -3728,23 +4115,8 @@ fn build_fetch_context() -> anyhow::Result { .map_err(|e| anyhow::anyhow!("creating log dir {parent}: {e}"))?; } } - let session_id = ulid::Ulid::generate().to_string(); - let log = Arc::new( - ProvenanceLog::open(log_path, session_id.clone()) - .map_err(|e| anyhow::anyhow!("opening provenance log: {e}"))?, - ); - let http = Arc::new(build_http_client_for_fetch()?); - let rate_limiter = Arc::new(RateLimiter::new(RateLimits::HARD_CODED)); - Ok(FetchContext { - http, - rate_limiter, - log, - session_id, - // Resolver cache disabled on the MCP path for now; the resolve - // cache (docs/CACHE.md) is wired through `doiget verify` first. - // Enabling it here for metadata_only / resolve_paper is a follow-up. - cache_root: None, - }) + ProvenanceLog::open(log_path, session_id) + .map_err(|e| anyhow::anyhow!("opening provenance log: {e}")) } /// Build a [`CrossrefSource`] from environment variables @@ -3784,6 +4156,14 @@ fn build_http_client_for_fetch() -> anyhow::Result { // `DOIGET_ARXIV_BASE` is absent. let arxiv_src = std::env::var("DOIGET_ARXIV_SRC_BASE").ok(); + #[cfg(feature = "tdm-aps")] + let tdm_aps = std::env::var("DOIGET_APS_BASE").ok(); + #[cfg(feature = "tdm-elsevier")] + let tdm_elsevier = std::env::var("DOIGET_ELSEVIER_BASE").ok(); + #[cfg(feature = "tdm-springer")] + let tdm_springer = std::env::var("DOIGET_SPRINGER_BASE").ok(); + #[cfg(feature = "tdm-ieee")] + let tdm_ieee = std::env::var("DOIGET_IEEE_BASE").ok(); if arxiv.is_none() && arxiv_src.is_none() && crossref.is_none() @@ -3875,8 +4255,30 @@ fn build_http_client_for_fetch() -> anyhow::Result { // When DOIGET_ARXIV_BASE is set use it; otherwise fall back to // DOIGET_ARXIV_SRC_BASE (they share the "arxiv" HTTP source key). let arxiv_entry = arxiv.as_deref().or(arxiv_src.as_deref()); + // Tier-3 test bases. A wiremock e2e could not reach the TDM-fetched + // route at all before this: the override branch built its allowlist + // from a fixed table of Tier-1/2 keys, so `tdm-aps` was simply not in + // the client's map and every attempt died as + // `no allowlist registered for source tdm-aps`. That was read as + // "#454's shape, reachable again" and filed as a possible production + // regression -- the production branch above extends with + // `tier_3_allowlists()` and was correct all along. The defect was + // here, in the harness, which is why only a route assertion found it. + // + // Not added to the production-branch test above on purpose: setting + // only `DOIGET_APS_BASE` (which `tdm_ieee.rs` documents as a way to + // replay a recorded fixture) must NOT drop the process into the + // allow-http test client. for (source, base) in [ ("arxiv", arxiv_entry), + #[cfg(feature = "tdm-aps")] + ("tdm-aps", tdm_aps.as_deref()), + #[cfg(feature = "tdm-elsevier")] + ("tdm-elsevier", tdm_elsevier.as_deref()), + #[cfg(feature = "tdm-springer")] + ("tdm-springer", tdm_springer.as_deref()), + #[cfg(feature = "tdm-ieee")] + ("tdm-ieee", tdm_ieee.as_deref()), ("crossref", crossref.as_deref()), ("unpaywall", unpaywall.as_deref()), ("oa-publisher", oa_publisher.as_deref()), @@ -3969,7 +4371,14 @@ impl ServerHandler for Server { "doiget v{VERSION} \u{2014} Open Access paper fetcher (stdio MCP). \ Tier 1 sources are always-on; Tier 2/3 require build features and \ env-var grants. Call `doiget_capability_profile` for the runtime \ - view; call `doiget_health` for an operational sanity check." + view; call `doiget_health` for an operational sanity check. \ + RETRY CONTRACT: every failure carries `error.disposition`. \ + `terminal` = the answer will not change, do not retry. \ + `retry_after` = it may change on its own, retry with backoff. \ + `needs_config` = it will not change by itself, but a named \ + change makes it — surface that to the user instead of \ + looping. Read that field, not the error code's name: \ + NO_OA_AVAILABLE is `needs_config`, not something to wait out." )) } } @@ -3978,6 +4387,15 @@ impl ServerHandler for Server { // Helpers // --------------------------------------------------------------------------- +/// Whether a `DOIGET_STORE_ROOT` value is usable as a path: non-empty and not +/// an unexpanded `${...}` placeholder. A Desktop-Extension config left blank +/// passes the literal `${user_config.store_root}`, which must never become a +/// filesystem path (it produced `os error 5` access-denied). See #369. +fn store_root_env_is_usable(value: &str) -> bool { + let v = value.trim(); + !v.is_empty() && !v.contains("${") +} + /// Resolve the on-disk store root using the same precedence the CLI /// applies (`docs/CONFIG.md` §4): /// @@ -3995,15 +4413,6 @@ impl ServerHandler for Server { /// `doiget-cli -> doiget-mcp` wiring established by this PR and pull /// `clap` etc. into the MCP crate. Lifting this helper into `doiget-core` /// is a viable Phase-3 follow-up but is out of scope for this foundation. -/// Whether a `DOIGET_STORE_ROOT` value is usable as a path: non-empty and not -/// an unexpanded `${...}` placeholder. A Desktop-Extension config left blank -/// passes the literal `${user_config.store_root}`, which must never become a -/// filesystem path (it produced `os error 5` access-denied). See #369. -fn store_root_env_is_usable(value: &str) -> bool { - let v = value.trim(); - !v.is_empty() && !v.contains("${") -} - fn resolve_store_root() -> Option { if let Ok(s) = std::env::var("DOIGET_STORE_ROOT") { if store_root_env_is_usable(&s) { @@ -4049,7 +4458,7 @@ fn store_root_from_config() -> Option { tracing::warn!( path = %path, error = %e, - "config.toml could not be read; [store] root ignored and the default store root used instead" + "config.toml could not be read; [store] root ignored and the default store root used instead" ); return None; } @@ -4178,6 +4587,150 @@ fn capability_profile_to_json(profile: &CapabilityProfile) -> Value { #[cfg(test)] #[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)] mod tests { + /// The rate limiter and the provenance log are ONE per process. + /// + /// Every tool handler used to call a free `build_fetch_context()` that + /// constructed both fresh. `RateLimiter`'s pacing state is instance-local, + /// so arXiv's 3 s spacing -- an obligation, not politeness -- was not + /// enforced between MCP calls; and `ProvenanceLog::open` seeds its + /// `(next_seq, last_hash)` by reading the file, so two overlapping calls + /// could append rows whose `prev_hash` does not match the row before + /// them, which `audit-log --verify` reports as a broken chain. + /// + /// Asserted by pointer identity, because that is the property: two + /// contexts from one server must share the same allocation, not merely + /// equal configuration. + #[test] + fn two_tool_calls_share_one_rate_limiter_and_one_log() { + let td = tempfile::TempDir::new().expect("tempdir"); + let root = camino::Utf8Path::from_path(td.path()).expect("utf-8"); + std::env::set_var("DOIGET_LOG_PATH", root.join("log.jsonl").as_str()); + std::env::set_var("DOIGET_STORE_ROOT", root.join("papers").as_str()); + + let server = Server::new(CapabilityProfile::from_env().expect("profile")); + let a = server.fetch_context().expect("first context"); + let b = server.fetch_context().expect("second context"); + + assert!( + Arc::ptr_eq(&a.rate_limiter, &b.rate_limiter), + "a fresh RateLimiter per call has an empty per-source window, so nothing paces the second request behind the first" + ); + assert!( + Arc::ptr_eq(&a.log, &b.log), + "a fresh ProvenanceLog per call re-reads the chain state, so two overlapping appends claim the same ts_seq" + ); + assert_eq!( + a.session_id, b.session_id, + "PROVENANCE_LOG.md: the session ULID is generated once per PROCESS" + ); + + std::env::remove_var("DOIGET_LOG_PATH"); + std::env::remove_var("DOIGET_STORE_ROOT"); + } + + /// #538 on the two batch tools. + /// + /// `fetch_paper_fetch_error_envelope` shed a hand-rolled + /// `From<&FetchError>` copy ending in `_ => InternalError`. Two verbatim + /// copies of that match stayed behind in `build_bibliography_envelope` and + /// `batch_fetch_success_envelope`, so on those two surfaces a refused + /// paper was an INTERNAL_ERROR and -- worse -- a 404 was a NETWORK_ERROR + /// with disposition `retry_after`, telling an agent to keep retrying a DOI + /// that will never resolve. `disposition` is derived from the code, so the + /// wrong code was also the wrong advice. + /// + /// Driven through the real envelope builders, not a re-implementation. + #[test] + fn a_batch_entry_gets_the_canonical_code_not_internal_error() { + use doiget_core::http::HttpError; + use doiget_core::orchestrator::{BatchOutcome, BatchResultEntry}; + use doiget_core::source::FetchError; + + let cases: Vec<(FetchError, &str, &str)> = vec![ + ( + FetchError::NotRetrievable { + source_key: "europepmc".into(), + detail: "record has no open-access full text".into(), + }, + "NO_OA_AVAILABLE", + "needs_config", + ), + ( + FetchError::Http(HttpError::HttpStatus { + status: 404, + url: "https://api.crossref.org/works/10.5555/absent".into(), + retry_after_ms: None, + }), + "NOT_FOUND", + "terminal", + ), + // The replaced comment names three variants the wildcard swallowed; + // the first pass at this test drove one of them. + ( + FetchError::NotFound { + hint: "no Crossref record".into(), + }, + "NOT_FOUND", + "terminal", + ), + ( + FetchError::Ambiguous { + hint: "3 authors matched".into(), + }, + "AMBIGUOUS", + "terminal", + ), + ]; + + for (err, want_code, want_disposition) in cases { + let ref_ = Ref::parse("10.1234/x").expect("valid doi"); + let batch = BatchOutcome::for_test_synthetic(vec![BatchResultEntry { + ref_: ref_.clone(), + outcome: Err(err), + }]); + + let bib = build_bibliography_envelope( + &batch, + std::slice::from_ref(&ref_), + &[None], + Vec::new(), + ); + let e = &bib["results"][0]["error"]; + assert_eq!( + e["code"], + json!(want_code), + "batch_from_bibliography: {bib:?}" + ); + assert_eq!(e["disposition"], json!(want_disposition), "{bib:?}"); + + let raw = vec![ref_.as_input_str().to_string()]; + let bf = batch_fetch_success_envelope(&batch, &raw); + let e = &bf["results"][0]["error"]; + assert_eq!(e["code"], json!(want_code), "batch_fetch: {bf:?}"); + assert_eq!(e["disposition"], json!(want_disposition), "{bf:?}"); + } + } + + /// One error, one answer, on both front ends. The CLI delegated + /// `GraphError::Source` to `From<&FetchError>` and this crate flattened it + /// to `NETWORK_ERROR`, so an OpenAlex 404 during expansion was terminal on + /// one surface and retriable on the other. + #[cfg(feature = "citation")] + #[test] + fn a_graph_source_error_keeps_the_wrapped_code() { + use doiget_core::citation_graph::GraphError; + use doiget_core::http::HttpError; + use doiget_core::source::FetchError; + + let e = GraphError::Source(FetchError::Http(HttpError::HttpStatus { + status: 404, + url: "https://api.openalex.org/works/doi:10.5555/absent".into(), + retry_after_ms: None, + })); + assert_eq!(graph_error_code(&e), ErrorCode::NotFound); + assert_eq!(graph_error_code(&e).disposition().as_wire(), "terminal"); + } + use super::*; /// #406: the store-writability probe MUST NOT create anything. diff --git a/crates/doiget-mcp/tests/error_envelope_shape.rs b/crates/doiget-mcp/tests/error_envelope_shape.rs new file mode 100644 index 000000000..601c42e77 --- /dev/null +++ b/crates/doiget-mcp/tests/error_envelope_shape.rs @@ -0,0 +1,122 @@ +//! Every `ok: false` envelope carries a structured `error` OBJECT (ADR-0055). +//! +//! `docs/ERRORS.md` §3 names `every_bare_string_error_site_is_a_known_one` as +//! the guard that pins which tools have not got there yet, "so it can shrink +//! but not grow". The guard did not exist. The document was accurate about +//! the four tools it listed and wrong about the mechanism keeping the list +//! honest -- a claim about the world resting on code that was never written, +//! which is the defect class that document defines. +//! +//! It exists now, and the set it pins is EMPTY: `doiget_resolve_citation`, +//! `doiget_batch_resolve_citations`, `doiget_tag` and `doiget_annotate` build +//! `error_object` like every other tool. Kept as a guard rather than deleted +//! along with the exemption, because the failure mode is a new `json!` literal +//! typing `"error": format!(...)` -- which is how the four got there. + +#![allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)] + +use camino::Utf8PathBuf; + +fn router_src() -> Utf8PathBuf { + Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("src/lib.rs") +} + +/// Every place the router names an `error` field, in BOTH forms it uses. +/// +/// Two forms exist, and the first version of this file knew about one: +/// +/// ```text +/// json!({ ..., "error": }) // matched +/// map.insert("error".into(), ) // INVISIBLE +/// ``` +/// +/// `fetch_paper_error_envelope` uses the second, and it backs the +/// `INVALID_REF` / `STORE_ERROR` / `LOG_ERROR` / `INTERNAL_ERROR` arms of +/// `doiget_fetch_paper` -- among the most reachable failures in the crate. +/// So the guard whose docstring called the exempt set EMPTY could not see the +/// busiest envelope builder in the file. Found by review. +fn error_field_value(line: &str) -> Option<&str> { + for key in [ + "\"error\":", + "insert(\"error\".into(),", + "insert(\"error\".to_string(),", + ] { + if let Some((_, rest)) = line.split_once(key) { + return Some(rest); + } + } + None +} + +/// The accepted right-hand sides for an `error` field. +/// +/// `error_object(..)` is the builder; `error_obj` is the `serde_json::Map` +/// the hand-assembled envelopes fill in and insert. Both are objects. +fn is_structured(value: &str) -> bool { + let v = value.trim_start(); + if v.starts_with("error_object(") { + return true; + } + // Exact identifier, not a prefix. `v.starts_with("error_obj")` also + // accepted `error_obj_msg`, so a bare-string regression could be waved + // through by naming the local variable carefully -- the guard defeated by + // spelling, which is the failure it exists to prevent. + v.strip_prefix("error_obj") + .is_some_and(|rest| !rest.starts_with(|c: char| c.is_alphanumeric() || c == '_')) +} + +#[test] +fn every_bare_string_error_site_is_a_known_one() { + let src = std::fs::read_to_string(router_src()).expect("router source is readable"); + + let mut bare = Vec::new(); + for (lineno, line) in src.lines().enumerate() { + let trimmed = line.trim_start(); + // Prose is not an envelope. + if trimmed.starts_with("//") { + continue; + } + let Some(rest) = error_field_value(trimmed) else { + continue; + }; + // A key with its value on the next line is written `"error":` alone; + // the builder call is what follows, so look there instead. + let value = if rest.trim().is_empty() { + src.lines().nth(lineno + 1).unwrap_or_default() + } else { + rest + }; + if !is_structured(value) { + bare.push(format!("src/lib.rs:{}: {}", lineno + 1, trimmed)); + } + } + + // One assertion, not two. The first version kept a + // `KNOWN_BARE_STRING_TOOLS` exemption list and told a failing + // contributor to add their tool to it -- advice that could not work, + // because the second assertion checked `bare.is_empty()` + // unconditionally and never consulted the list. An escape hatch that + // does not open is worse than none: it sends the next person down a + // path ending in rewriting the test anyway. + assert!( + bare.is_empty(), + "an ok:false envelope answers with a bare string instead of error_object(..), so the caller has no code to branch on and no disposition to decide a retry from (ADR-0055). Build the object; if a tool genuinely cannot, say so in docs/ERRORS.md and change this test deliberately: + {}", + bare.join(" + ") + ); +} + +/// The document must not still tell readers that four tools are exempt when +/// none are. This is the half of the original pairing that was doing real +/// work -- a doc describing a state the code left behind is the historical +/// defect -- and it needs no exemption list to detect it. +#[test] +fn the_document_does_not_claim_exemptions_that_no_longer_exist() { + let errors_md = Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../docs/ERRORS.md"); + let doc = std::fs::read_to_string(&errors_md).expect("docs/ERRORS.md is readable"); + assert!( + !doc.contains("do not yet emit that object at all"), + "docs/ERRORS.md section 3 still says some tools answer with a bare string in error, but every_bare_string_error_site_is_a_known_one finds none. Update the document, or this guard is vouching for prose nobody checked." + ); +} diff --git a/crates/doiget-mcp/tests/fetch_paper_e2e.rs b/crates/doiget-mcp/tests/fetch_paper_e2e.rs index 7ec87dc32..972fbb340 100644 --- a/crates/doiget-mcp/tests/fetch_paper_e2e.rs +++ b/crates/doiget-mcp/tests/fetch_paper_e2e.rs @@ -54,6 +54,11 @@ const ENV_KEYS: &[&str] = &[ "DOIGET_OA_PUBLISHER_BASE", "DOIGET_CONTACT_EMAIL", "DOIGET_UNPAYWALL_EMAIL", + // #462: the Tier-3 route. Cleared for every test so an APS grant can + // never leak from one into another. + "DOIGET_APS_BASE", + "DOIGET_KEY_APS", + "DOIGET_AGREE_TDM_APS", ]; async fn boot_in_memory_server() -> anyhow::Result<( @@ -185,6 +190,16 @@ async fn fetch_paper_arxiv_happy_path_writes_pdf_and_returns_envelope() -> anyho serde_json::json!(true), "envelope: {structured:?}" ); + // #462: WHICH ROUTE produced this, not merely that something did. + // Four "unreachable source" bugs shipped with green unit tests + // because nothing asserted the route, and a measurement over this + // suite found only one of the five `PdfLegStatus` routes asserted + // anywhere. See `route_coverage_e2e.rs`. + assert_eq!( + structured["pdf"]["status"], + serde_json::json!("fetched"), + "the arXiv happy path must report the `fetched` route: {structured:?}" + ); assert_eq!(structured["source"], serde_json::json!("arxiv")); assert_eq!(structured["ref"], serde_json::json!("2401.12345")); assert_eq!(structured["license"], serde_json::json!("arxiv-default")); @@ -475,6 +490,15 @@ async fn batch_fetch_partial_failure_emits_per_ref_outcomes() -> anyhow::Result< "transport per-ref error must surface denial_context (as null) per Slice 2 spec; got: {:?}", results[1], ); + // #506: this envelope is one of the five assembled by hand, and review + // found `disposition` was inserted at six sites and asserted at two -- + // deleting this one's insert failed no test. The error object is already + // in hand here, so the assertion costs nothing. + assert!( + results[1]["error"]["disposition"].is_string(), + "every failure envelope carries a disposition, including this one: {:?}", + results[1]["error"] + ); assert!( results[1]["error"]["denial_context"].is_null(), "denial_context must be null for NETWORK_ERROR; got: {:?}", @@ -616,3 +640,350 @@ async fn fetch_paper_doi_blocked_pdf_includes_suggested_arxiv_id() -> anyhow::Re drop(td); Ok(()) } + +#[tokio::test] +#[serial_test::serial] +async fn fetch_paper_doi_falls_back_to_the_arxiv_preprint() -> anyhow::Result<()> { + use wiremock::matchers::{method, path}; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + let server = MockServer::start().await; + + // Crossref metadata — minimal envelope. + // Crossref uses `Url::join("/works/")` which does NOT percent-encode + // the `/` inside the DOI suffix, so wiremock matches the raw path. + Mock::given(method("GET")) + .and(path("/works/10.1234/suggest-test")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "status": "ok", + "message": { + "title": ["Suggestion Test Paper"], + "author": [{"family": "Doe", "given": "Jane"}], + "issued": {"date-parts": [[2024, 1, 1]]} + } + }))) + .mount(&server) + .await; + + // Unpaywall metadata — `best_oa_location` points to a versioned arXiv URL. + // The arXiv host is off the `oa-publisher` allowlist (which only permits + // the wiremock host), so the PDF leg will be denied at the pre-fetch + // allowlist check, triggering PdfLegStatus::Blocked with a suggestion. + // Unpaywall uses `path_segments_mut().push()` which percent-encodes `/`. + Mock::given(method("GET")) + .and(path("/v2/10.1234%2Fsuggest-test")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "doi": "10.1234/suggest-test", + "is_oa": true, + // `is_oa:true` + `oa_status:"closed"` is deliberately + // contradictory (real Unpaywall never pairs these); the + // orchestrator does not cross-validate the two, so this isolates + // pure oa_status passthrough onto the fetch envelope. + "oa_status": "closed", + "best_oa_location": { + "url_for_pdf": "https://arxiv.org/pdf/2401.99999v2.pdf", + "url": "https://arxiv.org/abs/2401.99999v2", + "license": "cc-by" + }, + "oa_locations": [ + { + "url_for_pdf": "https://arxiv.org/pdf/2401.99999v2.pdf", + "url": "https://arxiv.org/abs/2401.99999v2" + } + ] + }))) + .mount(&server) + .await; + + // The preprint the suggestion points at, actually served. + let arxiv = MockServer::start().await; + Mock::given(method("GET")) + .respond_with(ResponseTemplate::new(200).set_body_bytes(SAMPLE_PDF_BODY.to_vec())) + .mount(&arxiv) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let temp_root = camino::Utf8Path::from_path(td.path()) + .expect("tempdir is utf-8") + .to_path_buf(); + let store_root = temp_root.join("papers"); + let log_path = temp_root.join("log.jsonl"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_STORE_ROOT", store_root.as_str()); + env.set("DOIGET_LOG_PATH", log_path.as_str()); + env.set("DOIGET_CROSSREF_BASE", &server.uri()); + env.set("DOIGET_UNPAYWALL_BASE", &format!("{}/v2", server.uri())); + // Register only the wiremock host for oa-publisher. arxiv.org is absent + // so the arXiv OA candidate is denied → PdfLegStatus::Blocked. + env.set("DOIGET_OA_PUBLISHER_BASE", &server.uri()); + // The one difference from the blocked test: the #325 fallback fetches + // through the arXiv SOURCE, not oa-publisher, so it needs its own base. + env.set("DOIGET_ARXIV_BASE", &arxiv.uri()); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert("ref".to_string(), serde_json::json!("10.1234/suggest-test")); + + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_fetch_paper").with_arguments(args)) + .await?; + let structured = result + .structured_content + .as_ref() + .expect("doiget_fetch_paper uses CallToolResult::structured"); + + // Metadata fetch succeeds (ok:true) but PDF leg is blocked. + assert_eq!( + structured["ok"], + serde_json::json!(true), + "envelope should be ok:true (metadata was written); got: {structured:?}" + ); + assert_eq!( + structured["pdf"]["status"], + serde_json::json!("preprint_fallback"), + "#462: the SUGGESTION and the FALLBACK are different routes, one field apart in the envelope, and only the first had ever been asserted: {structured:?}" + ); + assert_eq!( + structured["source"], + serde_json::json!("arxiv"), + "the bytes came from arXiv, and `source` has to say so: {structured:?}" + ); + assert!( + structured["size_bytes"].as_u64().unwrap_or(0) > 0, + "a fallback that reports success must have written bytes: {structured:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} + +#[tokio::test] +#[serial_test::serial] +async fn fetch_paper_doi_with_no_oa_anywhere_reports_the_no_oa_url_route() -> anyhow::Result<()> { + use wiremock::matchers::{method, path}; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + let server = MockServer::start().await; + + // Crossref metadata — minimal envelope. + // Crossref uses `Url::join("/works/")` which does NOT percent-encode + // the `/` inside the DOI suffix, so wiremock matches the raw path. + Mock::given(method("GET")) + .and(path("/works/10.1234/suggest-test")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "status": "ok", + "message": { + "title": ["Suggestion Test Paper"], + "author": [{"family": "Doe", "given": "Jane"}], + "issued": {"date-parts": [[2024, 1, 1]]} + } + }))) + .mount(&server) + .await; + + // Unpaywall metadata — `best_oa_location` points to a versioned arXiv URL. + // The arXiv host is off the `oa-publisher` allowlist (which only permits + // the wiremock host), so the PDF leg will be denied at the pre-fetch + // allowlist check, triggering PdfLegStatus::Blocked with a suggestion. + // Unpaywall uses `path_segments_mut().push()` which percent-encodes `/`. + Mock::given(method("GET")) + .and(path("/v2/10.1234%2Fsuggest-test")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "doi": "10.1234/suggest-test", + "is_oa": false, + "oa_status": "closed", + // No `best_oa_location` and no `oa_locations`: Unpaywall knows the + // work and has nothing free for it. + "best_oa_location": serde_json::Value::Null, + "oa_locations": [] + }))) + .mount(&server) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let temp_root = camino::Utf8Path::from_path(td.path()) + .expect("tempdir is utf-8") + .to_path_buf(); + let store_root = temp_root.join("papers"); + let log_path = temp_root.join("log.jsonl"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_STORE_ROOT", store_root.as_str()); + env.set("DOIGET_LOG_PATH", log_path.as_str()); + env.set("DOIGET_CROSSREF_BASE", &server.uri()); + env.set("DOIGET_UNPAYWALL_BASE", &format!("{}/v2", server.uri())); + // Register only the wiremock host for oa-publisher. arxiv.org is absent + // so the arXiv OA candidate is denied → PdfLegStatus::Blocked. + env.set("DOIGET_OA_PUBLISHER_BASE", &server.uri()); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert("ref".to_string(), serde_json::json!("10.1234/suggest-test")); + + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_fetch_paper").with_arguments(args)) + .await?; + let structured = result + .structured_content + .as_ref() + .expect("doiget_fetch_paper uses CallToolResult::structured"); + + // Metadata fetch succeeds (ok:true) but PDF leg is blocked. + assert_eq!( + structured["ok"], + serde_json::json!(true), + "envelope should be ok:true (metadata was written); got: {structured:?}" + ); + assert_eq!( + structured["pdf"]["status"], + serde_json::json!("no_oa_url"), + "#462: nowhere to fetch FROM is a different route than being refused AT somewhere, and this one had no assertion anywhere: {structured:?}" + ); + assert_eq!( + structured["ok"], + serde_json::json!(true), + "metadata-only is a success, not a failure: {structured:?}" + ); + assert_eq!( + structured["oa_status"], + serde_json::json!("closed"), + "and the envelope says WHY there was nowhere to go: {structured:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} + +/// The Tier-3 TDM-fetched route, asserted end to end over MCP. +/// +/// Written as an `#[ignore]`d reproduction first, because it failed with: +/// +/// ```text +/// "detail": "network error: no allowlist registered for source tdm-aps" +/// ``` +/// +/// -- `HttpError::UnknownSource`, the source key absent from the client's map. +/// That was diagnosed as "`tier_3_allowlists()` is `#[cfg]`-gated and the MCP +/// server does extend its allowlists with it, so the two disagree somewhere +/// between construction and use", i.e. #454's shape reachable again, and it was +/// raised as something to decide before cutting a release. +/// +/// The diagnosis was wrong, and wrong in this file's own subject matter: a +/// statement accurate about the code and false about the world. Both client +/// builders have two branches. The production branch does extend with +/// `tier_3_allowlists()` and was correct throughout. The test-override branch +/// -- taken whenever ANY `DOIGET_*_BASE` is set, which every wiremock test +/// does -- built its allowlist from a fixed table of Tier-1/2 keys with no +/// Tier-3 entry, so no e2e on either surface could reach this route. The +/// defect was in the harness. Registering the Tier-3 keys there is what this +/// test now proves, by passing. +/// +/// What survives from that diagnosis, and is still true: `fetch_content` is +/// implemented by APS alone. Elsevier, Springer and IEEE inherit the default +/// `Ok(None)` and are metadata-only, so three of the four Tier-3 sources +/// cannot reach the route the tier exists for. Read this as APS coverage, not +/// as Tier-3 coverage. +/// +/// #462: the Tier-3 route, which had no assertion anywhere -- which is how +/// #458, "the Tier-3 chain is skipped whenever Crossref answers", shipped. +#[cfg(feature = "tdm-aps")] +#[tokio::test] +#[serial_test::serial] +async fn fetch_paper_doi_served_by_the_publisher_reports_the_tdm_fetched_route( +) -> anyhow::Result<()> { + use wiremock::matchers::{header, method, path}; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + const APS_DOI: &str = "10.1103/PhysRevX.10.011001"; + const KEY: &str = "test-aps-key"; + + let server = MockServer::start().await; + Mock::given(method("GET")) + .and(path(format!("/works/{APS_DOI}"))) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "status": "ok", + "message": { "title": ["An APS article"], "DOI": APS_DOI } + }))) + .mount(&server) + .await; + // An OA location that the allowlist refuses, so the CONTENT leg is blocked + // -- which is the trigger #458 gave the Tier-3 chain. + Mock::given(method("GET")) + .and(path("/v2/10.1103%2FPhysRevX.10.011001")) + .respond_with(ResponseTemplate::new(200).set_body_json(serde_json::json!({ + "doi": APS_DOI, + "is_oa": true, + "oa_status": "closed", + "best_oa_location": { "url_for_pdf": "https://not-allowlisted.example/x.pdf" } + }))) + .mount(&server) + .await; + + // The publisher's own copy, under the agreement. + let aps = MockServer::start().await; + Mock::given(method("GET")) + .and(header("x-api-key", KEY)) + .respond_with(ResponseTemplate::new(200).set_body_bytes(SAMPLE_PDF_BODY.to_vec())) + .mount(&aps) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let temp_root = camino::Utf8Path::from_path(td.path()) + .expect("tempdir is utf-8") + .to_path_buf(); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_STORE_ROOT", temp_root.join("papers").as_str()); + env.set("DOIGET_LOG_PATH", temp_root.join("log.jsonl").as_str()); + env.set("DOIGET_CROSSREF_BASE", &server.uri()); + env.set("DOIGET_UNPAYWALL_BASE", &format!("{}/v2", server.uri())); + env.set("DOIGET_OA_PUBLISHER_BASE", &server.uri()); + env.set("DOIGET_APS_BASE", &aps.uri()); + env.set("DOIGET_KEY_APS", KEY); + env.set("DOIGET_AGREE_TDM_APS", "1"); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert("ref".to_string(), serde_json::json!(APS_DOI)); + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_fetch_paper").with_arguments(args)) + .await?; + let structured = result + .structured_content + .as_ref() + .expect("doiget_fetch_paper uses CallToolResult::structured"); + + assert_eq!( + structured["pdf"]["status"], + serde_json::json!("tdm_fetched"), + "the route the whole Tier-3 feature exists for, and the one that had no assertion anywhere: {structured:?}" + ); + assert_eq!( + structured["source"], + serde_json::json!("tdm-aps"), + "and it must name WHICH agreement was drawn on, because that one has terms attached: {structured:?}" + ); + assert!( + structured["size_bytes"].as_u64().unwrap_or(0) > 0, + "bytes, not a metadata-only stand-in: {structured:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} diff --git a/crates/doiget-mcp/tests/paper_search_e2e.rs b/crates/doiget-mcp/tests/paper_search_e2e.rs index b0837f933..0d32d84c9 100644 --- a/crates/doiget-mcp/tests/paper_search_e2e.rs +++ b/crates/doiget-mcp/tests/paper_search_e2e.rs @@ -78,6 +78,127 @@ const SAMPLE_SEARCH: &str = r#"{ ] }"#; +/// #534: an over-long query that matches nothing must not read as "this work +/// is not indexed". OpenAlex free-text matching degrades sharply past roughly +/// eight terms and returns nothing rather than a partial match, so the +/// envelope has to say which of the two happened -- an agent reading +/// `ok: true` with an empty array cannot tell, and in the session behind #534 +/// it concluded absence eleven times and abandoned a paper a shorter query +/// then found immediately. +#[tokio::test] +#[serial_test::serial] +async fn an_over_long_zero_result_query_carries_a_hint() -> anyhow::Result<()> { + let server = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/works")) + .respond_with( + ResponseTemplate::new(200) + .set_body_string(r#"{ "meta": { "count": 0 }, "results": [] }"#), + ) + .mount(&server) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let log_path = camino::Utf8Path::from_path(td.path()) + .expect("utf-8 tempdir") + .join("mcp-search-zero.jsonl"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_OPENALEX_BASE", &server.uri()); + env.set("DOIGET_LOG_PATH", log_path.as_str()); + + let (client, server_handle) = boot_in_memory_server().await?; + + // The exact query from the report: ten terms, zero results, real paper. + let mut args = serde_json::Map::new(); + args.insert( + "query".to_string(), + serde_json::json!( + "lithium refractoriness after discontinuation kindling sensitization course of illness Post" + ), + ); + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_paper_search").with_arguments(args)) + .await?; + let s = result + .structured_content + .as_ref() + .expect("doiget_paper_search uses CallToolResult::structured"); + + assert_eq!(s["ok"], serde_json::json!(true), "still a success: {s:?}"); + assert_eq!(s["count"], serde_json::json!(0)); + let hint = s["hint"].as_str().unwrap_or_default(); + assert!( + hint.contains("10 terms"), + "the hint names the term count so the reader can act on it: {s:?}" + ); + assert!( + hint.contains("3-5"), + "the hint says what to do instead, not only what went wrong: {s:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} + +/// The counterpart: a short query with no results gets no hint. A two-term +/// search really may mean the work is not indexed, and a hint on every empty +/// result would teach readers to skip it. +#[tokio::test] +#[serial_test::serial] +async fn a_short_zero_result_query_carries_no_hint() -> anyhow::Result<()> { + let server = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/works")) + .respond_with( + ResponseTemplate::new(200) + .set_body_string(r#"{ "meta": { "count": 0 }, "results": [] }"#), + ) + .mount(&server) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let log_path = camino::Utf8Path::from_path(td.path()) + .expect("utf-8 tempdir") + .join("mcp-search-short.jsonl"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_OPENALEX_BASE", &server.uri()); + env.set("DOIGET_LOG_PATH", log_path.as_str()); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert( + "query".to_string(), + serde_json::json!("nonexistent gibberish"), + ); + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_paper_search").with_arguments(args)) + .await?; + let s = result + .structured_content + .as_ref() + .expect("doiget_paper_search uses CallToolResult::structured"); + + assert_eq!(s["count"], serde_json::json!(0)); + assert!( + s.get("hint").is_none() || s["hint"].is_null(), + "a short query gets no hint: {s:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} + #[tokio::test] #[serial_test::serial] async fn paper_search_returns_external_envelope() -> anyhow::Result<()> { diff --git a/crates/doiget-mcp/tests/paper_text_e2e.rs b/crates/doiget-mcp/tests/paper_text_e2e.rs index 654ebc39d..8d378e8c6 100644 --- a/crates/doiget-mcp/tests/paper_text_e2e.rs +++ b/crates/doiget-mcp/tests/paper_text_e2e.rs @@ -180,9 +180,15 @@ async fn paper_text_max_chars_truncates() -> anyhow::Result<()> { #[tokio::test] #[serial_test::serial] -async fn paper_text_doi_maps_to_no_oa_available() -> anyhow::Result<()> { - // A DOI has no full-text source in this slice (ADR-0032 D5): the tool - // must return a structured NO_OA_AVAILABLE envelope, no network touched. +async fn paper_text_doi_is_terminal_not_a_config_problem() -> anyhow::Result<()> { + // A DOI has no full-text source in this slice (ADR-0032 D5). The code says + // WHICH kind of "no": `NO_OA_AVAILABLE` carries disposition `needs_config` + // -- "a named change makes it" -- and sends an agent looking for a config + // knob that does not exist, because the missing piece is DOI-to-arXiv + // linking (#281 item 5), not a grant. `NOT_IMPLEMENTED` is terminal and + // is the code `verify` / `batch_from_bibliography` already give the same + // situation for PMIDs (#500): valid input, absent support. + // No network touched either way. let env = EnvGuard::new(ENV_KEYS); let (client, server_handle) = boot_in_memory_server().await?; @@ -196,7 +202,10 @@ async fn paper_text_doi_maps_to_no_oa_available() -> anyhow::Result<()> { let s = result.structured_content.as_ref().expect("structured"); assert_eq!(s["ok"], serde_json::json!(false), "envelope: {s:?}"); - assert_eq!(s["error"]["code"], serde_json::json!("NO_OA_AVAILABLE")); + assert_eq!(s["error"]["code"], serde_json::json!("NOT_IMPLEMENTED")); + // The code is only half the answer; the disposition is what an agent + // branches on, and it is the half that was wrong. + assert_eq!(s["error"]["disposition"], serde_json::json!("terminal")); // MCP_TOOLS.md §5: an ok:false envelope echoes the input `ref`. assert_eq!(s["ref"], serde_json::json!("10.1234/example")); @@ -249,3 +258,32 @@ async fn paper_text_unconverted_paper_maps_to_text_unavailable() -> anyhow::Resu drop(td); Ok(()) } + +/// The sibling tool got the same re-code and had no test of its own, so the +/// two could have drifted the moment one was edited -- which is the failure +/// this release is about. +#[tokio::test] +#[serial_test::serial] +async fn paper_tex_source_doi_is_terminal_not_a_config_problem() -> anyhow::Result<()> { + let env = EnvGuard::new(ENV_KEYS); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert("ref".to_string(), serde_json::json!("10.1234/example")); + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_paper_tex_source").with_arguments(args)) + .await?; + let s = result.structured_content.as_ref().expect("structured"); + + assert_eq!(s["ok"], serde_json::json!(false), "envelope: {s:?}"); + assert_eq!(s["error"]["code"], serde_json::json!("NOT_IMPLEMENTED")); + assert_eq!(s["error"]["disposition"], serde_json::json!("terminal")); + assert_eq!(s["ref"], serde_json::json!("10.1234/example")); + + client.cancel().await?; + server_handle.await??; + drop(env); + Ok(()) +} diff --git a/crates/doiget-mcp/tests/resolve_paper_e2e.rs b/crates/doiget-mcp/tests/resolve_paper_e2e.rs index fefaaacec..9df8e1c53 100644 --- a/crates/doiget-mcp/tests/resolve_paper_e2e.rs +++ b/crates/doiget-mcp/tests/resolve_paper_e2e.rs @@ -165,6 +165,16 @@ async fn doiget_resolve_paper_invalid_ref_returns_invalid_ref_envelope() -> anyh .unwrap_or(false), "INVALID_REF message must mention 'invalid ref'; got: {structured:?}" ); + // #506: the retry decision an agent has to make was encoded in a markdown + // table it never reads, so its only signal was the code's NAME. Asserted + // through the real tool rather than on the helper, because "the envelope + // carries it" is the claim -- a helper that is right and unreached would + // pass a unit test and change nothing. + assert_eq!( + structured["error"]["disposition"], + serde_json::json!("terminal"), + "a malformed ref will not become well-formed by waiting: {structured:?}" + ); client.cancel().await?; server_handle.await??; @@ -356,6 +366,363 @@ async fn doiget_resolve_paper_doi_crossref_happy_path_returns_metadata_envelope( Ok(()) } +// --------------------------------------------------------------------------- +// 3b. #539 — the OA location is opt-in, and its absence is legible +// --------------------------------------------------------------------------- + +/// A REALISTIC Crossref response. `SAMPLE_CROSSREF_RESPONSE` above carries +/// `intended-application: "unspecified"`, which the #517 measurement found in +/// **zero** of twelve live `link[]` entries and zero of eight captured +/// fixtures -- it exercises the accept arm of the gate, but it is not what +/// Crossref actually returns. This one is: a Similarity Check link, correctly +/// refused, leaving `oa_url` null. That is the #539 condition. +const REALISTIC_CROSSREF_RESPONSE: &str = r#"{"status":"ok","message":{"title":["Example Paper"],"link":[{"URL":"https://publisher.example.org/similarity/10.1234/example.pdf","content-type":"application/pdf","intended-application":"similarity-checking"}]}}"#; + +const SAMPLE_UNPAYWALL_RESPONSE: &str = r#"{"doi":"10.1234/example","is_oa":true,"oa_status":"gold","best_oa_location":{"url_for_pdf":"https://repository.example.org/free.pdf","url":"https://repository.example.org/landing","license":"cc-by"}}"#; + +/// Default: one request, and `oa_url` is null because Crossref alone cannot +/// supply one. The Unpaywall mock is mounted with `.expect(0)`, so if the +/// default path ever starts paying for a second round-trip this test fails on +/// `MockServer` drop rather than silently doubling everyone's cost. +#[tokio::test] +#[serial_test::serial] +async fn resolve_paper_does_not_consult_unpaywall_by_default() -> anyhow::Result<()> { + use wiremock::matchers::{method, path}; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + let crossref = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/works/10.1234/example")) + .respond_with(ResponseTemplate::new(200).set_body_string(REALISTIC_CROSSREF_RESPONSE)) + .mount(&crossref) + .await; + + let unpaywall = MockServer::start().await; + Mock::given(method("GET")) + .respond_with(ResponseTemplate::new(200).set_body_string(SAMPLE_UNPAYWALL_RESPONSE)) + .expect(0) + .mount(&unpaywall) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let log_path = camino::Utf8Path::from_path(td.path()) + .expect("tempdir is utf-8") + .join("mcp-resolve-default.jsonl"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_CROSSREF_BASE", &crossref.uri()); + env.set("DOIGET_UNPAYWALL_BASE", &format!("{}/v2", unpaywall.uri())); + env.set("DOIGET_LOG_PATH", log_path.as_str()); + env.set("DOIGET_CONTACT_EMAIL", "test@example.org"); + env.set( + "DOIGET_STORE_ROOT", + td.path().to_str().expect("utf-8 tempdir"), + ); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert("ref".to_string(), serde_json::json!("10.1234/example")); + + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_resolve_paper").with_arguments(args)) + .await?; + let structured = result + .structured_content + .as_ref() + .expect("doiget_resolve_paper uses CallToolResult::structured"); + + assert_eq!(structured["ok"], serde_json::json!(true), "{structured:?}"); + assert_eq!( + structured["oa_url"], + serde_json::Value::Null, + "a similarity-checking link is not an OA URL (#517): {structured:?}" + ); + assert_eq!( + structured["oa_status"], + serde_json::Value::Null, + "nothing was consulted, so nothing is known: {structured:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + // Verifies the `.expect(0)`. + drop(unpaywall); + drop(td); + Ok(()) +} + +/// Opt in and the field is filled from Unpaywall's `best_oa_location` -- the +/// URL, the status, and the license, which the Crossref-only path also leaves +/// null. +#[tokio::test] +#[serial_test::serial] +async fn resolve_paper_with_include_oa_location_returns_a_real_oa_url() -> anyhow::Result<()> { + use wiremock::matchers::{method, path}; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + let crossref = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/works/10.1234/example")) + .respond_with(ResponseTemplate::new(200).set_body_string(REALISTIC_CROSSREF_RESPONSE)) + .mount(&crossref) + .await; + + let unpaywall = MockServer::start().await; + Mock::given(method("GET")) + .respond_with(ResponseTemplate::new(200).set_body_string(SAMPLE_UNPAYWALL_RESPONSE)) + .mount(&unpaywall) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let log_path = camino::Utf8Path::from_path(td.path()) + .expect("tempdir is utf-8") + .join("mcp-resolve-opt-in.jsonl"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_CROSSREF_BASE", &crossref.uri()); + env.set("DOIGET_UNPAYWALL_BASE", &format!("{}/v2", unpaywall.uri())); + env.set("DOIGET_LOG_PATH", log_path.as_str()); + env.set("DOIGET_CONTACT_EMAIL", "test@example.org"); + env.set( + "DOIGET_STORE_ROOT", + td.path().to_str().expect("utf-8 tempdir"), + ); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert("ref".to_string(), serde_json::json!("10.1234/example")); + args.insert("include_oa_location".to_string(), serde_json::json!(true)); + + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_resolve_paper").with_arguments(args)) + .await?; + let structured = result + .structured_content + .as_ref() + .expect("doiget_resolve_paper uses CallToolResult::structured"); + + assert_eq!(structured["ok"], serde_json::json!(true), "{structured:?}"); + assert_eq!( + structured["oa_url"], + serde_json::json!("https://repository.example.org/free.pdf"), + "{structured:?}" + ); + assert_eq!(structured["oa_status"], serde_json::json!("gold")); + assert_eq!(structured["license"], serde_json::json!("cc-by")); + // The record is still Crossref's; the lookup adds fields, it does not + // change whose metadata this is. + assert_eq!(structured["source"], serde_json::json!("crossref")); + assert_eq!( + structured["metadata"]["title"], + serde_json::json!(["Example Paper"]) + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} + +/// The case #539 is really about. A caller that paid for a lookup must be +/// able to tell "this work has no OA location" from "I could not find out". +/// Unpaywall is down here, so BOTH fields stay null -- and a null `oa_status` +/// is precisely what says the lookup did not complete, because a lookup that +/// completes on a closed work reports `oa_status: "closed"`. +/// +/// The call still succeeds: the Crossref metadata the caller also asked for +/// is good, and failing the whole resolve over an optional extra would be a +/// worse answer than an honest partial one. +#[tokio::test] +#[serial_test::serial] +async fn resolve_paper_leaves_oa_status_null_when_the_lookup_fails() -> anyhow::Result<()> { + use wiremock::matchers::{method, path}; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + let crossref = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/works/10.1234/example")) + .respond_with(ResponseTemplate::new(200).set_body_string(REALISTIC_CROSSREF_RESPONSE)) + .mount(&crossref) + .await; + + let unpaywall = MockServer::start().await; + Mock::given(method("GET")) + .respond_with(ResponseTemplate::new(500)) + // Without this the test would also pass against a build that ignored + // `include_oa_location` and never called Unpaywall at all: "both + // fields null" is what that produces too. + // + // A RANGE, not `== 1`: a 5xx is retried by the transport (measured: + // four attempts), and the retry policy is not this feature's contract + // to freeze. + .expect(1..) + .mount(&unpaywall) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let log_path = camino::Utf8Path::from_path(td.path()) + .expect("tempdir is utf-8") + .join("mcp-resolve-degraded.jsonl"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_CROSSREF_BASE", &crossref.uri()); + env.set("DOIGET_UNPAYWALL_BASE", &format!("{}/v2", unpaywall.uri())); + env.set("DOIGET_LOG_PATH", log_path.as_str()); + env.set("DOIGET_CONTACT_EMAIL", "test@example.org"); + env.set( + "DOIGET_STORE_ROOT", + td.path().to_str().expect("utf-8 tempdir"), + ); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert("ref".to_string(), serde_json::json!("10.1234/example")); + args.insert("include_oa_location".to_string(), serde_json::json!(true)); + + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_resolve_paper").with_arguments(args)) + .await?; + let structured = result + .structured_content + .as_ref() + .expect("doiget_resolve_paper uses CallToolResult::structured"); + + assert_eq!( + structured["ok"], + serde_json::json!(true), + "an optional extra failing must not sink the resolve: {structured:?}" + ); + assert_eq!( + structured["metadata"]["title"], + serde_json::json!(["Example Paper"]), + "the metadata the caller also asked for is still there: {structured:?}" + ); + assert_eq!(structured["oa_url"], serde_json::Value::Null); + assert_eq!( + structured["oa_status"], + serde_json::Value::Null, + "a null oa_status is the signal that the lookup did not complete; \ + inventing 'closed' here would assert the work is paywalled on the \ + strength of a 500: {structured:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + // Verifies the `.expect(1)`. + drop(unpaywall); + drop(td); + Ok(()) +} + +// --------------------------------------------------------------------------- +// 3c. #507 — the bookend records WHAT the call failed with +// --------------------------------------------------------------------------- + +/// #507: the provenance log could say a call failed but not what it failed +/// with, because every `SessionEnd` row carried `error_code: None`. +/// +/// That is a gap in the audit trail on its own terms -- the log cannot answer +/// "what did this session tell the caller about this ref?" -- and it is also +/// what blocks the repeat suppression #507 asks for: the rule is "do not +/// re-fetch a prior `terminal` or `needs_config` answer", and a disposition +/// cannot be recovered from a row with no code. +#[tokio::test] +#[serial_test::serial] +async fn a_failed_call_records_its_terminal_code_on_the_bookend() -> anyhow::Result<()> { + use wiremock::matchers::method; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + // Crossref and Unpaywall both answer 404, so the DOI resolves nowhere and + // the call ends as NOT_FOUND. + let upstream = MockServer::start().await; + Mock::given(method("GET")) + .respond_with(ResponseTemplate::new(404)) + .mount(&upstream) + .await; + + let td = tempfile::TempDir::new().expect("tempdir"); + let log_path = camino::Utf8Path::from_path(td.path()) + .expect("tempdir is utf-8") + .join("mcp-bookend.jsonl"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_CROSSREF_BASE", &upstream.uri()); + env.set("DOIGET_UNPAYWALL_BASE", &format!("{}/v2", upstream.uri())); + env.set("DOIGET_LOG_PATH", log_path.as_str()); + env.set("DOIGET_CONTACT_EMAIL", "test@example.org"); + env.set( + "DOIGET_STORE_ROOT", + td.path().to_str().expect("utf-8 tempdir"), + ); + + let (client, server_handle) = boot_in_memory_server().await?; + + let mut args = serde_json::Map::new(); + args.insert("ref".to_string(), serde_json::json!("10.1234/absent")); + let result = client + .peer() + .call_tool(CallToolRequestParams::new("doiget_resolve_paper").with_arguments(args)) + .await?; + let structured = result + .structured_content + .as_ref() + .expect("doiget_resolve_paper uses CallToolResult::structured"); + assert_eq!( + structured["ok"], + serde_json::json!(false), + "premise: the call fails: {structured:?}" + ); + let reported = structured["error"]["code"] + .as_str() + .unwrap_or_default() + .to_string(); + + client.cancel().await?; + server_handle.await??; + + let raw = std::fs::read_to_string(&log_path).expect("read the provenance log"); + let bookend = raw + .lines() + .filter_map(|l| serde_json::from_str::(l).ok()) + .find(|r| r["event"] == "session_end") + .expect("a SessionEnd row was written"); + + assert_eq!( + bookend["result"], + serde_json::json!("err"), + "row: {bookend}" + ); + assert_eq!( + bookend["error_code"].as_str().unwrap_or_default(), + reported, + "the row must record the code the CALLER was given, not null: {bookend}" + ); + assert_eq!(bookend["ref"], serde_json::json!("10.1234/absent")); + + // #506: this envelope is assembled field-by-field rather than through + // `error_object`, so the first pass at the disposition missed it -- and it + // is the shape an agent sees for the most ordinary failure there is. + // NOT_FOUND is terminal: an id does not become correct by waiting. + assert_eq!( + structured["error"]["disposition"], + serde_json::json!("terminal"), + "the hand-built failure envelope must carry it too: {structured:?}" + ); + + drop(env); + drop(td); + Ok(()) +} + // --------------------------------------------------------------------------- // 4. dry_run field rejection // --------------------------------------------------------------------------- diff --git a/crates/doiget-mcp/tests/route_coverage_e2e.rs b/crates/doiget-mcp/tests/route_coverage_e2e.rs new file mode 100644 index 000000000..8ee8d9898 --- /dev/null +++ b/crates/doiget-mcp/tests/route_coverage_e2e.rs @@ -0,0 +1,319 @@ +//! #462: which route produced the outcome, asserted per route. +//! +//! Four "unreachable source" bugs shipped with green unit tests -- #413, #442, +//! #454, #458 -- and they share one shape: a source was implemented, gated, +//! allowlisted and unit-tested, and was never reached. The unit test drove the +//! `Source` impl directly, or asserted a builder returned the right value, and +//! the production entry point was never in the picture. #454's own guard +//! carried a doc comment describing the failure it could not catch. +//! +//! The thing a unit test cannot do is notice that a correct component is never +//! reached. Only an assertion about WHICH ROUTE ran can do that. +//! +//! Measured before writing this file: of the five `PdfLegStatus` routes, +//! exactly **one** (`blocked`) was asserted anywhere in the e2e suites. +//! `tdm_fetched` had none -- which is why #458, "the Tier-3 chain is skipped +//! whenever Crossref answers", could ship. +//! +//! ## What this file is +//! +//! Not more tests. A **registry**: every route, and either the test that +//! asserts it or a stated reason it is not asserted yet. A gap recorded with a +//! reason is a gap someone can close; a gap that is merely absent is the one +//! that ships. +//! +//! Two checks keep it honest: +//! +//! * the named covering test must exist AND contain the route string, so a +//! claim of coverage cannot outlive the assertion it names; +//! * a posture-lint step compares this list against the `PdfLegStatus` +//! variants in `doiget-core`, so a new route cannot be added without landing +//! here first. + +#![allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)] + +use camino::Utf8PathBuf; + +/// How a route is covered. +#[derive(Debug)] +#[allow( + dead_code, + reason = "no route is uncovered today; the variant is how the next one gets recorded instead of silently omitted" +)] +enum Coverage { + /// Asserted by `fn` in the given test file. + By { + file: &'static str, + test_fn: &'static str, + /// The Cargo feature the covering test is `#[cfg]`-gated behind, if + /// any. `None` means it compiles in the default `oa-only` surface. + /// + /// This exists because the checks below read the test file as TEXT. + /// Text cannot tell "this function is compiled into the binary" from + /// "this function's source is present in the repository", so a + /// feature-gated test was accepted as unconditional coverage -- and + /// the two REQUIRED CI jobs run `--features oa-only`, where that + /// function does not exist. The registry vouched for a route with no + /// coverage in the builds that gate merges: this file's own stated + /// failure class, inside this file. + feature: Option<&'static str>, + }, + /// Not asserted yet, and why. Deliberately not `None`: the reason is the + /// difference between a known gap and an oversight. + Gap { why: &'static str }, +} + +/// Every `PdfLegStatus` route, by its wire name. +/// +/// Kept in the order the enum declares them so a reviewer can diff the two by +/// eye; the posture-lint does it mechanically. +const ROUTES: &[(&str, Coverage)] = &[ + ( + "fetched", + Coverage::By { + file: "fetch_paper_e2e.rs", + test_fn: "fetch_paper_arxiv_happy_path_writes_pdf_and_returns_envelope", + feature: None, + }, + ), + ( + "no_oa_url", + Coverage::By { + file: "fetch_paper_e2e.rs", + test_fn: "fetch_paper_doi_with_no_oa_anywhere_reports_the_no_oa_url_route", + feature: None, + }, + ), + ( + "blocked", + Coverage::By { + file: "fetch_paper_e2e.rs", + test_fn: "fetch_paper_doi_blocked_pdf_includes_suggested_arxiv_id", + feature: None, + }, + ), + ( + "preprint_fallback", + Coverage::By { + file: "fetch_paper_e2e.rs", + test_fn: "fetch_paper_doi_falls_back_to_the_arxiv_preprint", + feature: None, + }, + ), + ( + "tdm_fetched", + // Was the file's one `Gap`, on a reproduction that could not pass + // because the test harness had no way to register a Tier-3 allowlist + // entry -- not, as the reason here claimed, because production had + // regressed to #454's shape. + Coverage::By { + file: "fetch_paper_e2e.rs", + test_fn: "fetch_paper_doi_served_by_the_publisher_reports_the_tdm_fetched_route", + // Only the `test (tdm features)` CI job compiles this. + feature: Some("tdm-aps"), + }, + ), +]; + +fn tests_dir() -> Utf8PathBuf { + Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("tests") +} + +/// A claim of coverage must name a test that exists, is NOT `#[ignore]`d, and +/// asserts the route INSIDE ITS OWN BODY. +/// +/// The first version was two whole-file string searches: does the file +/// contain the test's name, and does the file contain the route string +/// anywhere. Review pointed out that both are satisfiable without the claim +/// being true -- a comment mentioning the function, plus some unrelated test +/// in the same file containing the literal -- and, worse, that it had no +/// concept of `#[ignore]`. A `By` entry pointing at an ignored test would +/// have passed while CI ran nothing. +/// +/// That is the same "correct component that nothing reaches" shape this file +/// exists to catch, reproduced inside the mechanism meant to catch it. So it +/// now extracts the named function's body and looks only there, and refuses a +/// covering test that CI skips. +#[test] +fn every_claimed_covering_test_exists_and_asserts_its_route() { + let mut problems = Vec::new(); + for (route, cov) in ROUTES { + let Coverage::By { + file, + test_fn, + feature, + } = cov + else { + continue; + }; + let path = tests_dir().join(file); + let Ok(src) = std::fs::read_to_string(&path) else { + problems.push(format!("{route}: {file} does not exist")); + continue; + }; + + // The definition, not a mention of the name in prose. + let Some(def) = src.find(&format!("fn {test_fn}(")) else { + problems.push(format!("{route}: {file} defines no `fn {test_fn}`")); + continue; + }; + + // Attributes sit immediately above the fn line. Walk back over the + // contiguous attribute block and refuse `#[ignore]`: a test CI does + // not run cannot be evidence that a route is covered. + let line_start = src[..def].rfind('\n').map_or(0, |i| i + 1); + let attrs_from = src[..line_start].rfind("\n\n").map_or(0, |i| i + 2); + // Line-by-line, and only lines that ARE an attribute. A substring search + // over the whole block also matches a doc comment that DISCUSSES + // `#[ignore]` -- which the corrected write-up of the TDM route now does, + // and it made this checker report the very test it was reading about as + // skipped. Prose is not an attribute; a checker that cannot tell the + // difference is the defect it exists to catch. + let is_ignored = src[attrs_from..line_start] + .lines() + .map(str::trim_start) + .filter(|l| l.starts_with("#[")) + // `#[ignore` catches the plain attribute; `ignore)` catches + // `#[cfg_attr(, ignore)]`, which cargo skips just as + // completely and which the first version of this check waved + // through -- it only compared the start of the line, so a + // conditionally-ignored test could be claimed as coverage. + // rustfmt leaves `cfg_attr` on one line, so `cargo fmt` does not + // rescue us here the way it does for `#[test] #[ignore]`. + .any(|l| l.starts_with("#[ignore") || l.contains("ignore)")); + // The registry's feature claim must match the source. A test the + // default build does not compile is not unconditional coverage, and + // saying so here is the only place a reader learns it: `cargo test` + // under `--features oa-only` cannot report a function it never built. + let attrs = &src[attrs_from..line_start]; + let src_feature = attrs.lines().map(str::trim_start).find_map(|l| { + let rest = l.strip_prefix("#[cfg(feature = \"")?; + rest.split('"').next() + }); + if src_feature != *feature { + problems.push(format!( + "{route}: registry says feature={feature:?} but `{test_fn}` is gated on {src_feature:?}. A feature-gated test is coverage only in a CI job that enables it" + )); + continue; + } + + if is_ignored { + problems.push(format!( + "{route}: `{test_fn}` is #[ignore]d, so CI never runs it -- that is a Gap with a reason, not coverage" + )); + continue; + } + + // The body only. A line that is exactly `}` ends a top-level fn. + let body_end = src[def..].find("\n}\n").map_or(src.len(), |i| def + i); + if !src[def..body_end].contains(&format!("\"{route}\"")) { + problems.push(format!( + "{route}: `{test_fn}` never asserts the string \"{route}\" in its own body" + )); + } + } + assert!( + problems.is_empty(), + "route coverage claims that are not backed by an assertion:\n {}", + problems.join("\n ") + ); +} + +/// The registry must name EVERY route, so a new one cannot be added without +/// deciding how it is covered -- the step all four bugs in the header skipped. +/// +/// Reads `PdfLegStatus` out of `doiget-core` and compares. Deliberately here +/// rather than as a shell step in posture-lint: the comparison needs +/// CamelCase-to-snake_case, that needs a regex backreference, and threading one +/// through YAML into bash is how the first attempt at this put a literal +/// control character into the workflow file. A test can just do it. +#[test] +fn the_registry_names_every_pdf_leg_route() { + let src = std::fs::read_to_string( + Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../doiget-core/src/orchestrator.rs"), + ) + .expect("read orchestrator.rs"); + + let body = src + .split_once("pub enum PdfLegStatus") + .and_then(|(_, rest)| { + rest.split_once( + " +}", + ) + }) + .map(|(body, _)| body) + .expect("PdfLegStatus enum not found -- did it move or get renamed?"); + + let mut variants: Vec = Vec::new(); + for line in body.lines() { + // A variant is ` Name,` or ` Name {`; anything more indented is + // a field, and anything less is not in the enum. + let Some(rest) = line.strip_prefix(" ") else { + continue; + }; + if rest.starts_with(' ') || !rest.starts_with(char::is_uppercase) { + continue; + } + let name: String = rest.chars().take_while(|c| c.is_alphanumeric()).collect(); + if name.is_empty() { + continue; + } + let mut snake = String::new(); + for (i, c) in name.chars().enumerate() { + if c.is_uppercase() && i > 0 { + snake.push('_'); + } + snake.extend(c.to_lowercase()); + } + variants.push(snake); + } + variants.sort(); + variants.dedup(); + + // The guard the assertion below cannot be: a parser that silently matched + // nothing would agree with an empty registry. + assert!( + variants.len() >= 5, + "parsed {} variants from PdfLegStatus; the enum has at least five, so this parser has stopped seeing them: {variants:?}", + variants.len() + ); + + let mut named: Vec = ROUTES.iter().map(|(r, _)| (*r).to_string()).collect(); + named.sort(); + + assert_eq!( + named, variants, + "the coverage registry and PdfLegStatus disagree. A route with no entry is a route nobody decided how to test, which is exactly how #413, #442, #454 and #458 shipped." + ); +} + +/// The gaps are the point of the file, so they are printed rather than hidden. +/// This test does not fail on a gap -- it fails if a gap has no reason, because +/// an unexplained gap is indistinguishable from an oversight, which is the +/// thing #462 is about. +#[test] +fn every_route_is_either_covered_or_has_a_stated_reason() { + let mut uncovered = Vec::new(); + for (route, cov) in ROUTES { + match cov { + Coverage::By { .. } => {} + Coverage::Gap { why } => { + assert!( + why.len() > 30, + "{route}: a gap needs a reason someone can act on, got {why:?}" + ); + uncovered.push(*route); + } + } + } + // `eprintln!` is banned in this crate -- stdout is the JSON-RPC channel and + // the lint does not distinguish the two streams. Asserting the count is + // better than printing it anyway: it turns "2 of 5" from a line someone + // might read into a number that has to be updated when it changes. + assert_eq!( + uncovered.len(), + 0, + "known gaps changed: {uncovered:?}. If a route just gained an assertion, move it from `Gap` to `By` and drop this count by one -- the point is that closing a gap is a visible edit, not a silent improvement." + ); +} diff --git a/crates/doiget-mcp/tests/tag_annotate_e2e.rs b/crates/doiget-mcp/tests/tag_annotate_e2e.rs new file mode 100644 index 000000000..2e8dcb31e --- /dev/null +++ b/crates/doiget-mcp/tests/tag_annotate_e2e.rs @@ -0,0 +1,255 @@ +//! E2E coverage for `doiget_tag` and `doiget_annotate`. +//! +//! These two tools had **no test of any kind** — not an error path, not even a +//! success path. That is how a wrong error code got into them and out again: +//! this release gave every one of their failure arms a structured `error` +//! object for the first time, and picked `NOT_FOUND` for "the ref is not in +//! the store yet". `docs/ERRORS.md` defines that code as *a metadata source +//! authoritatively reported the id does not exist*, which `doiget verify` +//! treats as a definite dead reference — so the tools would have told an agent +//! a perfectly good DOI was retracted. Nothing in the suite could notice. +//! +//! Every assertion below drives the real MCP tool over a real transport. + +#![allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)] + +use doiget_core::CapabilityProfile; +use doiget_mcp::Server; +use rmcp::{model::CallToolRequestParams, ServiceExt}; + +struct EnvGuard { + keys: Vec<&'static str>, +} + +impl EnvGuard { + fn new(keys: &[&'static str]) -> Self { + for k in keys { + std::env::remove_var(k); + } + Self { + keys: keys.to_vec(), + } + } + fn set(&self, key: &str, val: &str) { + std::env::set_var(key, val); + } +} + +impl Drop for EnvGuard { + fn drop(&mut self) { + for k in &self.keys { + std::env::remove_var(k); + } + } +} + +const ENV_KEYS: &[&str] = &["DOIGET_STORE_ROOT", "DOIGET_LOG_PATH"]; + +async fn boot_in_memory_server() -> anyhow::Result<( + rmcp::service::RunningService, + tokio::task::JoinHandle>, +)> { + let profile = CapabilityProfile::from_env().expect("clean env never errors"); + let server = Server::new(profile); + let (server_transport, client_transport) = tokio::io::duplex(64 * 1024); + let server_handle = tokio::spawn(async move { + let service = server.serve(server_transport).await?; + service.waiting().await?; + anyhow::Ok(()) + }); + let client = ().serve(client_transport).await?; + Ok((client, server_handle)) +} + +fn args(pairs: &[(&str, serde_json::Value)]) -> serde_json::Map { + let mut m = serde_json::Map::new(); + for (k, v) in pairs { + m.insert((*k).to_string(), v.clone()); + } + m +} + +/// A ref nobody has fetched is not a dead reference. +/// +/// The code must not be `NOT_FOUND`: `docs/ERRORS.md` reserves that for "a +/// metadata source authoritatively reported the id does not exist", and an +/// agent acting on it would drop a citation that is fine. The remedy here is +/// an action the caller can take, which is what `needs_config` means. +#[tokio::test] +#[serial_test::serial] +async fn tag_on_an_unfetched_ref_does_not_call_it_a_dead_reference() -> anyhow::Result<()> { + let td = tempfile::TempDir::new().expect("tempdir"); + let root = camino::Utf8Path::from_path(td.path()).expect("utf-8 tempdir"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_STORE_ROOT", root.join("papers").as_str()); + env.set("DOIGET_LOG_PATH", root.join("log.jsonl").as_str()); + + let (client, server_handle) = boot_in_memory_server().await?; + let result = client + .peer() + .call_tool( + CallToolRequestParams::new("doiget_tag").with_arguments(args(&[ + ("ref", serde_json::json!("10.1234/never-fetched")), + ("add", serde_json::json!(["to-read"])), + ])), + ) + .await?; + let s = result.structured_content.as_ref().expect("structured"); + + assert_eq!(s["ok"], serde_json::json!(false), "envelope: {s:?}"); + assert_ne!( + s["error"]["code"], + serde_json::json!("NOT_FOUND"), + "a ref that has not been fetched is not a dead reference: {s:?}" + ); + assert_eq!( + s["error"]["code"], + serde_json::json!("STORE_ERROR"), + "{s:?}" + ); + assert_eq!( + s["error"]["disposition"], + serde_json::json!("needs_config"), + "the remedy is a named action (fetch it), not a retry: {s:?}" + ); + let message = s["error"]["message"].as_str().unwrap_or_default(); + assert!( + message.contains("fetch"), + "the message names the action: {message:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} + +/// Same rule on the sibling tool, which had the same code and the same +/// absence of tests. +#[tokio::test] +#[serial_test::serial] +async fn annotate_on_an_unfetched_ref_does_not_call_it_a_dead_reference() -> anyhow::Result<()> { + let td = tempfile::TempDir::new().expect("tempdir"); + let root = camino::Utf8Path::from_path(td.path()).expect("utf-8 tempdir"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_STORE_ROOT", root.join("papers").as_str()); + env.set("DOIGET_LOG_PATH", root.join("log.jsonl").as_str()); + + let (client, server_handle) = boot_in_memory_server().await?; + let result = client + .peer() + .call_tool( + CallToolRequestParams::new("doiget_annotate").with_arguments(args(&[ + ("ref", serde_json::json!("10.1234/never-fetched")), + ("text", serde_json::json!("read the appendix")), + ])), + ) + .await?; + let s = result.structured_content.as_ref().expect("structured"); + + assert_eq!(s["ok"], serde_json::json!(false), "envelope: {s:?}"); + assert_ne!(s["error"]["code"], serde_json::json!("NOT_FOUND"), "{s:?}"); + assert_eq!( + s["error"]["code"], + serde_json::json!("STORE_ERROR"), + "{s:?}" + ); + assert_eq!( + s["error"]["disposition"], + serde_json::json!("needs_config"), + "{s:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} + +/// A malformed ref is the caller's input problem, and it must stay +/// distinguishable from the store miss above — same tool, two situations, two +/// codes. ADR-0055 exists so an agent can branch here. +#[tokio::test] +#[serial_test::serial] +async fn tag_on_a_malformed_ref_is_invalid_ref_not_a_store_problem() -> anyhow::Result<()> { + let td = tempfile::TempDir::new().expect("tempdir"); + let root = camino::Utf8Path::from_path(td.path()).expect("utf-8 tempdir"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_STORE_ROOT", root.join("papers").as_str()); + env.set("DOIGET_LOG_PATH", root.join("log.jsonl").as_str()); + + let (client, server_handle) = boot_in_memory_server().await?; + let result = client + .peer() + .call_tool( + CallToolRequestParams::new("doiget_tag").with_arguments(args(&[ + ("ref", serde_json::json!("not-a-doi")), + ("add", serde_json::json!(["x"])), + ])), + ) + .await?; + let s = result.structured_content.as_ref().expect("structured"); + + assert_eq!(s["ok"], serde_json::json!(false), "envelope: {s:?}"); + assert_eq!( + s["error"]["code"], + serde_json::json!("INVALID_REF"), + "{s:?}" + ); + assert_eq!( + s["error"]["disposition"], + serde_json::json!("terminal"), + "{s:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} + +/// `doiget_annotate` with neither `text` nor `clear` is a request-shape +/// failure. Before this release it answered with a bare string in `error`, so +/// a caller had no code to branch on at all. +#[tokio::test] +#[serial_test::serial] +async fn annotate_with_no_text_and_no_clear_says_so_in_a_structured_error() -> anyhow::Result<()> { + let td = tempfile::TempDir::new().expect("tempdir"); + let root = camino::Utf8Path::from_path(td.path()).expect("utf-8 tempdir"); + + let env = EnvGuard::new(ENV_KEYS); + env.set("DOIGET_STORE_ROOT", root.join("papers").as_str()); + env.set("DOIGET_LOG_PATH", root.join("log.jsonl").as_str()); + + let (client, server_handle) = boot_in_memory_server().await?; + let result = client + .peer() + .call_tool( + CallToolRequestParams::new("doiget_annotate") + .with_arguments(args(&[("ref", serde_json::json!("10.1234/never-fetched"))])), + ) + .await?; + let s = result.structured_content.as_ref().expect("structured"); + + assert_eq!(s["ok"], serde_json::json!(false), "envelope: {s:?}"); + assert!( + s["error"].is_object(), + "an ok:false envelope carries an error OBJECT, not a bare string (ADR-0055): {s:?}" + ); + assert!( + s["error"]["disposition"].is_string(), + "every failure envelope carries a disposition: {s:?}" + ); + + client.cancel().await?; + server_handle.await??; + drop(env); + drop(td); + Ok(()) +} diff --git a/docs/DECISIONS/0053-doi-resolvers-are-addressing.md b/docs/DECISIONS/0053-doi-resolvers-are-addressing.md new file mode 100644 index 000000000..64171899b --- /dev/null +++ b/docs/DECISIONS/0053-doi-resolvers-are-addressing.md @@ -0,0 +1,102 @@ +# 0053 - A DOI resolver is addressing, not hosting + +- **Date:** 2026-08-30 +- **Status:** Accepted +- **Supersedes:** - +- **Complements:** [0027](0027-redirect-allowlist-society-hosts.md) — answers a layer question 0027 never had to ask, and [0039](0039-publisher-hosts-stay-off-allowlist.md), which decided the adjacent case the other way for the same reason +- **Source:** #533 (a gold-OA cc-by paper refused at `doi.org`, one hop before an already-allowlisted publisher) + +## Context + +`doiget fetch 10.1002/pcn5.205` — Harada & Kato 2024, **gold OA, cc-by** — was +refused: + +``` +redirect target doi.org not in allowlist for source oa-publisher +``` + +with a remediation offering `doi.org` and `*.doi.org` as hosts to add. + +This is not an incomplete list. It is a gate applied at the wrong layer. + +Unpaywall reports `best_oa_location.url` for this work as literally +`https://doi.org/10.1002/pcn5.205`, with no `url_for_pdf` (verified against the +live API, 2026-08-30). That is not unusual — it is the normal shape for +publisher-hosted gold OA. So the very first host doiget touches on the fetch leg +is the DOI resolver, and it was being adjudicated as though it were the place +the bytes come from. The chain was refused before it ever reached +`onlinelibrary.wiley.com`, whose `*.wiley.com` was already on the list. + +The offered remediation is worse than the refusal. ADR-0027 justifies the +built-in list as *bounded* registrable-domain wildcards for established +publishers and repositories, and the list contains exactly that — no resolvers. +Adding `doi.org` would not widen the trusted surface toward one publisher. It +would remove the bound entirely, because **every DOI in existence resolves +through it**. An agent following that advice gets its PDF and silently loses the +invariant the allowlist exists to hold. + +ADR-0039 refused to add IEEE/ACM/SIAM/AMS to `oa-publisher` because those are +content hosts and the ADR would not widen the content surface. The same +principle decides this case the opposite way, and for the same reason: a +resolver is not a content host at all, so making it followable widens nothing. + +Three recent bugs share this shape — #503 (europe-pmc refused on a flag before +consulting the URL list), #516 (a Tier-2 gate keyed on the wrong feature), and +this one. #462 names why they survive review: *"every 'unreachable source' bug +passed its unit tests."* The allowlist matcher here is correct in isolation; the +defect only appears once `unpaywall → doi.org → wiley` is actually walked. + +## Decision + +**A closed set of DOI resolver hosts is transparent to host adjudication.** They +are followed, but never allowlisted, never named as remediation, and never +counted as the source of the content. + +``` +doi.org the canonical DOI resolver +dx.doi.org its long-standing alias, still present in live metadata +hdl.handle.net the Handle System resolver doi.org proxies +``` + +Each was measured issuing a single 302 straight to the publisher (2026-08-30, +`10.1002/pcn5.205`). + +Three constraints on that set: + +1. **Exact hosts, no wildcards.** `*.doi.org` would sweep in `www.doi.org` — + the DOI Foundation's website, not a resolver — and anything else ever stood + up there. `evil-doi.org` and `doi.org.evil.test` are what an attacker + registers. +2. **The host that serves the bytes is adjudicated exactly as before.** This is + transparent to the ADR-0027 invariant, not an exception to it. +3. **One predicate, not five gates.** `SourceAllowlist::permits` is what every + adjudication site calls; `matches` remains the narrower "is this host on the + list" used to build and assert the lists themselves. A resolver is therefore + never reported inside any source's `expected_hosts`. + +### Rejected: adjudicate only the terminal host + +#533's first suggestion. It is a larger change than it appears: a chain could +traverse *any* host so long as it ended somewhere allowed, and every hop still +observes the request. A named, closed set keeps the bound while fixing the +reported class. + +### Rejected: allowlist `doi.org` + +The remediation the tool was emitting. Mechanically effective, semantically +wrong, and it deletes the invariant — see Context. + +## Consequences + +- Publisher-hosted gold OA whose Unpaywall location is a `doi.org` URL is + reachable. This is a whole class, not one paper. +- No new content host is trusted. The set contains no host that serves papers. +- A denial now names the host the user would actually have to trust, because a + resolver hop can no longer produce one. +- **This does not promise a PDF.** `10.1002/pcn5.205` reaches Wiley and then + meets Wiley's own cookie wall (`/action/cookieAbsent`). That is a publisher + posture question — ADR-0039's territory — and it fails honestly at the + publisher rather than dishonestly at the addressing layer. The fix removes a + wrong refusal; it does not manufacture access. +- A posture-lint step fails any adjudication site that calls `matches` on a host + variable, so the fifth gate cannot be added without the fourth's lesson. diff --git a/docs/DECISIONS/0054-access-refusal-is-a-type.md b/docs/DECISIONS/0054-access-refusal-is-a-type.md new file mode 100644 index 000000000..1d733b451 --- /dev/null +++ b/docs/DECISIONS/0054-access-refusal-is-a-type.md @@ -0,0 +1,96 @@ +# 0054 - An access refusal is a type, and it collapses to `NO_OA_AVAILABLE` + +- **Date:** 2026-08-30 +- **Status:** Accepted +- **Supersedes:** - +- **Complements:** [0014](0014-docs-class-system.md) — this is the ADR that document requires for the `docs/ERRORS.md` §2 change below +- **Source:** #538 (an access refusal was classified by substring, so rewording one silently reclassified the row) + +## Context + +A source signalling *"I found it and cannot give it to you"* returned +`FetchError::SourceSchema { hint }` with an explanatory string, and +`classify_attempt` decided what the trace row said by **reading that string +back**: + +```rust +fn is_access_refusal(hint: &str) -> bool { + hint.contains("not open access") + || hint.contains("openAccess") + || hint.contains("no retrievable PDF") +} +``` + +Match → `AttemptOutcome::NotOpenAccess`, *"consulted: found, not open access"*. +Miss → `AttemptOutcome::Failed`, *"consulted: failed"*. Those are different +claims to an operator: **the source has it and will not give it to us** versus +**the source broke**, and only the second is a bug to chase. + +It had already fired. #503 reworded Europe PMC's refusal from *"is indexed but +not open access"* to *"advertises no retrievable PDF"* — correctly, because the +gate moved from OA-subset membership to per-entry retrievability — the hint fell +out of the predicate, and every Europe PMC refusal became `Failed`. Nothing in +`europepmc.rs` said the wording was load-bearing. `hal` matched on the substring +`openAccess`, which comes from a JSON **field name**, not prose anyone chose. + +It was also latent for every future source: an author had no way to learn that +the phrasing of an error message decides how the row renders. + +There is a second defect underneath. `SourceSchema` collapses to +`ErrorCode::InternalError`, so whenever such a refusal reached the boundary it +was reported as *a bug in doiget* — for the ordinary situation of a paper not +being free at one repository. + +## Decision + +**1. The refusal is a variant.** + +```rust +FetchError::NotRetrievable { source_key: String, detail: String } +``` + +`classify_attempt` matches the variant. `is_access_refusal` is deleted. `detail` +is carried verbatim into the row, because the reason is for a reader — it is +just no longer *parsed*. + +**2. It collapses to the existing `NO_OA_AVAILABLE`, and the closed set does +not widen.** + +`docs/ERRORS.md` §3 defines a closed `ErrorCode` set, and #538 asked explicitly +whether a new code was needed. It is not: *"found it, no free copy"* is what +`NO_OA_AVAILABLE` already means. Adding a code would have split one situation +across two wire values for an internal refactor, and every consumer switching on +the code would have needed to learn the new one to keep behaving the same. + +`docs/ERRORS.md` §2's description of `NO_OA_AVAILABLE` is widened to match: it +said *"Tier 1 sources reported no OA URL"*, and it now also covers an optional +source that holds the record but no retrievable copy. That is the NORMATIVE +change ADR-0014 requires this ADR for. **The wire value and its recoverability +guidance are unchanged.** + +**3. It is not a `DenialContext`.** + +`From<&FetchError> for Option` returns `None`. ADR-0023's denial +channel is for policy refusals — a capability to grant, an allowlist to widen. +An access refusal is a fact about the work: there is no configuration change +that makes a closed paper open, and offering a `denial_context` would send a +reader after one that does not exist. + +## Consequences + +- Rewording a refusal can no longer reclassify a row. The compiler is the guard, + which is what #538 asked for. +- An access refusal that reaches the boundary stops being reported as + `INTERNAL_ERROR`. This is a **wire-visible behaviour change for the same + input**, and a strictly more accurate one; the code it now returns was already + in the closed set and already documented as recoverable-by-enabling-a-source. +- Adding a source no longer requires knowing a phrasing convention. The variant + is the contract. +- Three tests that asserted `matches!(err, FetchError::SourceSchema { .. })` now + assert the category and the collapsed `ErrorCode`, which is a stronger claim. + One test asserted the *phrase* `"no retrievable PDF"`; it now asserts the + criterion the detail names (`documentStyle = pdf`, `availabilityCode`), since + asserting the phrase would have rebuilt the prose-coupling inside the test. +- `error.disposition` (#506) is left for its own ADR. An access refusal is + `needs_config` only when another source could be enabled, and that judgement + belongs with the disposition design rather than smuggled in here. diff --git a/docs/DECISIONS/0055-error-disposition.md b/docs/DECISIONS/0055-error-disposition.md new file mode 100644 index 000000000..83f2c4cc4 --- /dev/null +++ b/docs/DECISIONS/0055-error-disposition.md @@ -0,0 +1,93 @@ +# 0055 - A failure says what to do about it, in three states + +- **Date:** 2026-08-30 +- **Status:** Accepted +- **Supersedes:** - +- **Complements:** [0014](0014-docs-class-system.md) — the ADR `docs/ERRORS.md` §2/§3 changes require; [0023](0023-denial-context-structured.md) and [0043](0043-machine-readable-diagnostics.md), which built the *content* of a fix but only on the success envelope +- **Source:** #506 (the MCP envelope says what happened but not what to do next) + +## Context + +`docs/ERRORS.md` §2 has carried per-code retry guidance since Phase 0, and it is +good guidance: `INVALID_REF` → *"No (user must correct input)"*, `NOT_IMPLEMENTED` +→ *"do not retry"*. This is not missing thinking. It is missing **plumbing**: + +``` +grep -rniE 'retryable|do_not_retry|permanent|transient' crates/doiget-mcp → 0 +grep -rniE 'retry' crates/doiget-mcp/src/*.rs (tool descriptions) → 0 +``` + +The failure envelope was `{ok:false, error:{code, message, denial_context?}}`, so +an agent's only signal was **the name of the code** — and several names point the +wrong way. `NO_OA_AVAILABLE` is the most common failure there is, and both its +name and its ERRORS.md row (*"Try later, or enable opt-in source"*) read to a +machine as *wait*, when it is nearly always *configure*. That invites an +unbounded retry loop over something that will not change on its own. + +## Decision + +**1. `error.disposition`, with three states.** + +| value | meaning | +|---|---| +| `terminal` | the answer will not change. Do not retry, do not wait. | +| `retry_after` | it may change on its own. Retry with backoff. | +| `needs_config` | it will not change by itself, but a named change makes it. Surface it; do not loop. | + +Two states cannot express the third, and the third is the one that matters most +here. Facing it an agent should neither loop nor give up silently. + +`terminal` also covers failures a caller can act on by issuing a *different* +request — `INVALID_REF`, `AMBIGUOUS`, `TEXT_UNAVAILABLE`. **This** call is +settled, which is what a disposition is about. + +`STORE_ERROR` and `LOG_ERROR` are `needs_config`. A machine cannot name the fix +for a full disk, but it must not loop on one either, and "surface this to a +human" is exactly what the state means. + +**2. One source of truth.** `ErrorCode::disposition()` is an exhaustive `match` +with no wildcard, so a new code must decide. `docs/ERRORS.md` §2 gains a +Disposition column, and `errors_md_disposition_column_matches_the_code` parses +the shipped document and asserts every row against the function. #506 asked for +the table to be generated from the code or asserted against it, because +otherwise the doc and the wire drift — the #493 pattern. Generating it would +have cost the per-code prose, which is the useful part; asserting keeps both. + +The test also guards its own parser: it asserts it saw exactly 15 rows, because +a parser that silently matches nothing passes every time. + +**3. One builder.** Every failure envelope goes through `error_object`. A field +present on some failures and absent on others is worse than no field: it teaches +the reader to fall back to guessing from the code's name, which is the habit +this exists to replace. + +**4. Stated where an agent that never read ERRORS.md still meets it** — the MCP +server `instructions`, delivered on `initialize` to every client. One place +rather than twenty-two tool descriptions, and it names the specific trap: +`NO_OA_AVAILABLE` is `needs_config`, not something to wait out. + +## Consequences + +- The wire gains a field. Additive: `{ok:false}` consumers that ignore it are + unaffected, and `code` / `message` / `denial_context` are unchanged. +- `docs/ERRORS.md` §2 gains a column and §3's envelope shape gains the field. + That is the NORMATIVE change ADR-0014 requires this ADR for. No code's + *meaning* or recoverability guidance changed — only that the guidance is now + machine-readable. +- A new `ErrorCode` will not compile until its disposition is decided, and will + not pass tests until `docs/ERRORS.md` records the same answer. + +## Not in scope + +- **`error.retry_after_ms`.** #506 asks for it and it is not here. `Retry-After` + is parsed today (`http::parse_retry_after`) but consumed **inside** the retry + loop and discarded; by the time an error surfaces, the retries are exhausted + and no honest number remains. Emitting a plausible default would be a number + the caller could not distinguish from a measured one. Plumbing the header out + of the loop is its own change. +- **`remediation` on `ok:false`.** ADR-0043's channel keys on `DenialContext`, + which is present on only some failures; carrying it for those alone would + reproduce the sometimes-present problem this ADR rejects for `disposition`. +- **`rate_limit_budget` on live responses.** Independent of the retry contract. + +#506 stays open for those three. diff --git a/docs/DECISIONS/0056-not-determined-is-not-an-answer.md b/docs/DECISIONS/0056-not-determined-is-not-an-answer.md new file mode 100644 index 000000000..9dca40144 --- /dev/null +++ b/docs/DECISIONS/0056-not-determined-is-not-an-answer.md @@ -0,0 +1,88 @@ +# 0056 - A not-determined marker never overwrites a determination + +- **Date:** 2026-08-31 +- **Status:** Accepted +- **Supersedes:** - +- **Amends:** [`docs/STORE.md`](../STORE.md) §6 — the re-fetch downgrade note +- **Complements:** [0014](0014-docs-class-system.md) — the ADR a NORMATIVE doc change requires; [0055](0055-error-disposition.md), which drew the same distinction on the wire +- **Source:** #583 (a default `metadata_only` re-write silently downgraded `[doiget].oa_status` and `.license`) + +## Context + +`docs/STORE.md` §6 lets doiget rewrite the `[doiget]` table on a re-fetch, and +says why: + +> A doiget re-fetch of an entry that previously had a PDF but is now +> metadata-only ... rewrites the `[doiget]` table (`source`, `size_bytes`, …) +> in place. **This is intentional, not silent:** as of issue #118 the +> blocked-PDF reason is surfaced to the caller ... so the operator always +> learns the entry was downgraded and why. + +The permission is conditional on the report. Since #539 there is a path where +the report does not exist: `metadata_only` takes `include_oa_location`, and when +it is omitted — the ordinary call shape — the OA lookup never runs. The record +built from that outcome carries `oa_status: None` and `license: "unknown"`, the +merge let them win, and a caller got no `note:` line, no `pdf.status`, and no log +row saying a known `gold` / `CC-BY-4.0` had been replaced with *not determined*. + +Measured before deciding, because the issue as filed claimed the wrong field: + +| field | existing | after a default re-write | +|---|---|---| +| `url` (where `oa_url` lands) | `Some("https://…/paper.pdf")` | `Some(…)` — preserved by `merge_opt!` | +| `[doiget].oa_status` | `Some("gold")` | `None` | +| `[doiget].license` | `"CC-BY-4.0"` | `"unknown"` | + +So the reserved top-level fields were never at risk. The defect is confined to +the two `[doiget]` fields whose absent value is a **marker** rather than a +reading — and both are documented as such: `oa_status` is "omitted when not +determined (#281)", `license` is "an OA license string, or the literal +`unknown`". + +## Decision + +**A marker for "no answer" does not overwrite an answer.** + +In `merge_metadata`, the `[doiget]` arm keeps the existing `oa_status` when the +incoming one is `None`, and the existing `license` when the incoming one is +`LICENSE_UNDETERMINED`. Everything else in the table still follows §6: doiget +owns it and the re-write wins. + +This cannot suppress real news, which is the only reason it is safe: + +- a paper that stops being open access reports `oa_status: Some("closed")` +- a license that changes reports the new string + +Neither reports the marker. `None` and `"unknown"` are only ever produced by a +call that did not look, so preferring the stored value is not a guess about +which is newer — it is the observation that only one of the two is a value. + +`"unknown"` gained a name (`LICENSE_UNDETERMINED`) so the merge does not depend +on a string literal matching the others scattered through the orchestrator. + +## Consequences + +`docs/STORE.md` §6's note says a downgrade guard "is deferred (post-MVP) — it is +a policy choice, not a correctness bug." That remains true of the case it was +written about (#123: an entry loses its PDF because the OA host went +off-allowlist) — there the downgrade is real, and it is reported. It is not true +of a field the caller never asked about, so the note is amended to scope its +claim to determinations rather than markers. + +Nothing BiblioFetch.jl reads changes shape. No field is added, removed or +renamed, `schema_version` does not move, and the reserved top-level fields keep +the `merge_opt!` behaviour they already had. This is a change to which of two +`[doiget]` values doiget itself keeps. + +### Not done + +**Warning on every preserve.** The reserved-field arms `warn!` when they keep an +existing value, because there it means two tools disagree and a human may need +to look. Here it is the ordinary outcome of the ordinary call shape; a log line +per default `metadata_only` would be noise, and the thing worth reporting — +losing the value — no longer happens. + +**Extending this to `source` or `size_bytes`.** They have no marker value. +`size_bytes: 0` is a legitimate reading for a metadata-only entry, and `source` +always names the resolver that actually ran. Treating either as absence would +invent exactly the call-history dependence #583 warned against. diff --git a/docs/DECISIONS/INDEX.md b/docs/DECISIONS/INDEX.md index 6af163bf7..70b1c9c00 100644 --- a/docs/DECISIONS/INDEX.md +++ b/docs/DECISIONS/INDEX.md @@ -69,6 +69,10 @@ Status column reconciled 2026-05-17 against `CHANGELOG.md` slices (issue #150). | 0050 | credentials.toml carries the key; the agreement stays in the environment | Accepted | `feat/509-credentials-file` | #509 | | 0051 | Contributions carry a relicensable grant, recorded as a commit sign-off | Accepted | `chore/cla-relicensing-grant` | - | | 0052 | Crossref `link[]` is programme-scoped, so the fetch path does not carry it | Accepted | `feat/517-publisher-candidate` | #517 | +| 0053 | A DOI resolver is addressing, not hosting | Accepted | `fix/533-resolver-hop-is-addressing` | #533 | +| 0054 | An access refusal is a type, and it collapses to `NO_OA_AVAILABLE` | Accepted | `fix/538-typed-access-refusal` | #538 | +| 0055 | A failure says what to do about it, in three states | Accepted | `feat/506-error-disposition` | #506 | +| 0056 | A not-determined marker never overwrites a determination | Accepted | `fix/583-store-downgrade` | #583 | ## Conventions diff --git a/docs/ERRORS.md b/docs/ERRORS.md index c25bcea14..39ed6c3fb 100644 --- a/docs/ERRORS.md +++ b/docs/ERRORS.md @@ -33,29 +33,29 @@ Wire form (JSON / MCP): `"INVALID_REF"`, `"NO_OA_AVAILABLE"`, etc. ## 2. Code semantics -| Code | Meaning | Recoverable? | -|---|---|---| -| `INVALID_REF` | DOI / arXiv id failed validation. | No (user must correct input). | -| `NO_OA_AVAILABLE` | Tier 1 sources reported no OA URL. | Try later, or enable opt-in source. | -| `RATE_LIMITED` | Internal rate cap hit, OR 429 from source. | Retry after `Retry-After` (or 1 s). | -| `NETWORK_ERROR` | Transport / DNS / TLS failure. **Does NOT cover a deliberate supply-chain policy block** — see §6.1: an off-allowlist / redirect-denied / insecure-scheme OA-PDF leg is `CAPABILITY_DENIED`, not `NETWORK_ERROR`. | Retry usually fine. | -| `NOT_FOUND` | Metadata source authoritatively reported the id does not exist: HTTP `404` / `410` / `451`, or a source-specific absence (arXiv returns HTTP 200 with an empty `` for an unknown id). Network-independent and reproducible — distinct from the transient `NETWORK_ERROR` / `RATE_LIMITED`. `doiget verify` treats it as a definite dead reference (`absent`). For a DOI it is emitted only when all configured sources (Crossref, Unpaywall) fail to resolve it; a DataCite-only DOI may thus be reported `NOT_FOUND`. | No (the id is wrong or retracted). | -| `AMBIGUOUS` | A name filter (`--author` / `--venue` / `--publisher`) matched several entities with no clear winner; the error lists the candidates. Distinct from `NOT_FOUND` ("matched nothing"). CLI exit `2`. | Yes — narrow the name (add a first name / fuller title) or pass an exact id. | -| `STORE_ERROR` | Filesystem write failed (disk, permission, etc.). | Depends on cause. | -| `LOG_ERROR` | Provenance log write failed. **Fetch is aborted.** | Free disk / fix perms. | -| `CAPABILITY_DENIED` | Source not in `CapabilityProfile`. | User opts in, or pick different source. | -| `FETCH_TIMEOUT` | Per-request timeout exceeded. | Retry. | -| `SCHEMA_TOO_NEW` | Store entry's `schema_version` is ahead. | Upgrade doiget. | -| `LOCK_TIMEOUT` | Could not acquire `flock` within 5 s. | Retry; another process holds it. | -| `INTERNAL_ERROR` | Bug. | Report at . | -| `NOT_IMPLEMENTED` | Feature is spec'd but not yet wired in this Phase. | Wait for next minor release; do not retry. | -| `TEXT_UNAVAILABLE` | The id is valid and resolvable, but the **requested representation** is missing: `doiget text` got a 200 from ar5iv with no extractable prose (the paper was never converted to HTML). Distinct from `NOT_FOUND` (the id *does* exist) and `NO_OA_AVAILABLE` (the paper may still be OA — only the HTML render is missing). Issue #302. | Yes — fetch the PDF instead (`doiget fetch `); do not "fix" the identifier. | +| Code | Meaning | Disposition | Recoverable? | +|---|---|---|---| +| `INVALID_REF` | DOI / arXiv id failed validation. | `terminal` | No (user must correct input). | +| `NO_OA_AVAILABLE` | No source could supply a free copy: Tier 1 reported no OA URL, **or** a source holds the record and has no retrievable copy (`FetchError::NotRetrievable`, ADR-0054). | `needs_config` | Try later, or enable opt-in source. | +| `RATE_LIMITED` | Internal rate cap hit, OR 429 from source. | `retry_after` | Retry after `Retry-After` (or 1 s). | +| `NETWORK_ERROR` | Transport / DNS / TLS failure. **Does NOT cover a deliberate supply-chain policy block** — see §6.1: an off-allowlist / redirect-denied / insecure-scheme OA-PDF leg is `CAPABILITY_DENIED`, not `NETWORK_ERROR`. | `retry_after` | Retry usually fine. | +| `NOT_FOUND` | Metadata source authoritatively reported the id does not exist: HTTP `404` / `410` / `451`, or a source-specific absence (arXiv returns HTTP 200 with an empty `` for an unknown id). Network-independent and reproducible — distinct from the transient `NETWORK_ERROR` / `RATE_LIMITED`. `doiget verify` treats it as a definite dead reference (`absent`). For a DOI it is emitted only when all configured sources (Crossref, Unpaywall) fail to resolve it; a DataCite-only DOI may thus be reported `NOT_FOUND`. | `terminal` | No (the id is wrong or retracted). | +| `AMBIGUOUS` | A name filter (`--author` / `--venue` / `--publisher`) matched several entities with no clear winner; the error lists the candidates. Distinct from `NOT_FOUND` ("matched nothing"). CLI exit `2`. | `terminal` | Yes — narrow the name (add a first name / fuller title) or pass an exact id. | +| `STORE_ERROR` | The local store could not serve the request: a filesystem write failed (disk, permission, etc.), or the entry a mutating tool needs is not there yet (`doiget_tag` / `doiget_annotate` on a ref nobody has fetched). Deliberately NOT `NOT_FOUND` in that second case: the id is fine, and `NOT_FOUND` would tell a caller the reference is dead. The read-only tools (`doiget_info`, `doiget_paper_pdf_path`, `doiget_search_local`) report a store miss as `ok: true` with a null payload instead, which is preferable where the operation has nothing to mutate. | `needs_config` | Depends on cause; for a store miss, fetch the paper first. | +| `LOG_ERROR` | Provenance log write failed. **Fetch is aborted.** | `needs_config` | Free disk / fix perms. | +| `CAPABILITY_DENIED` | Source not in `CapabilityProfile`. | `needs_config` | User opts in, or pick different source. | +| `FETCH_TIMEOUT` | Per-request timeout exceeded. | `retry_after` | Retry. | +| `SCHEMA_TOO_NEW` | Store entry's `schema_version` is ahead. | `needs_config` | Upgrade doiget. | +| `LOCK_TIMEOUT` | Could not acquire `flock` within 5 s. | `retry_after` | Retry; another process holds it. | +| `INTERNAL_ERROR` | Bug. | `terminal` | Report at . | +| `NOT_IMPLEMENTED` | Feature is spec'd but not yet wired in this Phase. | `terminal` | Wait for next minor release; do not retry. | +| `TEXT_UNAVAILABLE` | The id is valid and resolvable, but the **requested representation** is missing: `doiget text` got a 200 from ar5iv with no extractable prose (the paper was never converted to HTML). Distinct from `NOT_FOUND` (the id *does* exist) and `NO_OA_AVAILABLE` (the paper may still be OA — only the HTML render is missing). Issue #302. | `terminal` | Yes — fetch the PDF instead (`doiget fetch `); do not "fix" the identifier. | ## 3. Persona × error matrix | Persona | Surface | |---|---| -| Agent (MCP) | Structured, never throws. On failure: `{ ok: false, error: { code, message, denial_context? } }`. `remediation` and `attempts` are **not** carried here — they belong to the `{ ok: true, … }` envelope, as `pdf.remediation` (present when the PDF leg was blocked) and top-level `attempts`. A blocked PDF leg is an `ok: true` result with a failed leg, not an `ok: false` call. | +| Agent (MCP) | Structured, never throws. On failure: `{ ok: false, error: { code, message, disposition, retry_after_ms?, remediation?, denial_context? } }`. `disposition` is present on every failure that carries a structured `error` OBJECT, and is the field to branch a retry on — see §2 and ADR-0055. `disposition` is now present on **every** `ok: false` envelope. Four tools -- `doiget_resolve_citation`, `doiget_batch_resolve_citations`, `doiget_tag` and `doiget_annotate` -- used to put a bare string in `error` instead, so a caller had nothing to branch on; they build the object like every other tool as of 0.8.13 **(breaking for anyone reading `error` as a string on those four)**. The set is pinned EMPTY by `every_bare_string_error_site_is_a_known_one`, and `the_exemption_list_and_the_document_agree` fails if this paragraph and that list ever disagree. Both are new: this document already named the guard as the reason the set "can shrink but not grow", and the guard had not been written -- a claim about the world resting on code that did not exist, which is exactly what this document defines. It is derived from `code` by `ErrorCode::disposition`, never hand-written per call site. `remediation` IS carried here when the failure has a `denial_context`, from the same `remediation::for_denial` the blocked leg and the CLI `= help:` block use — this document previously said it was not, which meant the one field naming the fix was present when a call succeeded with a blocked leg and absent when the call actually failed (#506). It is omitted when there is no named fix rather than emitted empty. `attempts` remains an `ok: true` field. A blocked PDF leg is still an `ok: true` result with a failed leg, not an `ok: false` call, and it carries `pdf.remediation` plus `pdf.disposition`. `retry_after_ms` is present only when the SERVER sent a `Retry-After` on the response that ended the attempt; it is never backfilled from doiget's internal backoff, because a guess about the server wearing the name of a server-supplied value is the defect these fields exist to remove (#506). | | Researcher (CLI human) | `cargo`-style stderr: `error[E0007]: rate limited from unpaywall: retry after 1s`. Exit code 1. | | CI / Batch (CLI `--json`) | JSON Lines record per ref with `{"ok":false, "error":{"code":"...","message":"...","denial_context":{...}?,"remediation":[...]?,"attempts":[...]?}}`. Exit code = number of failures (capped at 255). **Records are emitted in completion order, not input order** — see below. | | Library (Rust) | `Err(FetchError)` (typed via `thiserror`). | diff --git a/docs/MCP_TOOLS.md b/docs/MCP_TOOLS.md index 352abfc53..5c4e14b2f 100644 --- a/docs/MCP_TOOLS.md +++ b/docs/MCP_TOOLS.md @@ -17,7 +17,7 @@ speaks **stdio only** ([ADR-0001](DECISIONS/), [`SCOPE.md`](SCOPE.md) §non-goal | `doiget_info` | Retrieve a store entry's metadata. | | `doiget_search_local` | Search store metadata (title / authors / venue). | | `doiget_paper_search` | External literature discovery over OpenAlex (`/works?search=`); abstract-bearing candidates for triage. Tier-1 OA metadata, always-on; **never fetches a PDF** (ADR-0031). | -| `doiget_paper_text` | Extract an **arXiv** paper's full text from ar5iv as sectioned plain text (`ref`, optional `max_chars`). Tier-1 OA, always-on; **never opens the PDF blob** (ADR-0032). A DOI → `NO_OA_AVAILABLE`. | +| `doiget_paper_text` | Extract an **arXiv** paper's full text from ar5iv as sectioned plain text (`ref`, optional `max_chars`). Tier-1 OA, always-on; **never opens the PDF blob** (ADR-0032). A DOI → `NOT_IMPLEMENTED` (terminal: the tool is arXiv-only and DOI→arXiv linking is #281 item 5, not a config knob). | | `doiget_link` | Resolve a **DOI** to its arXiv preprint + identity cluster (`{ doi, arxiv, openalex_id, title }`) over OpenAlex, for reading or dedup (#281 item 5). Tier-1 OA, always-on; **never fetches a PDF**. arXiv → DOI is a follow-up; a non-DOI ref → `INVALID_REF`. | | `doiget_list_recent` | Last N fetched entries. | | `doiget_paper_pdf_path` | Return the local path of a cached PDF. **Does not read, parse, or transmit content.** | @@ -33,6 +33,10 @@ Additional tools: | `doiget_csl_export` | CSL JSON for one or many entries. | | `doiget_resolve_citation` | Resolve a free-form bibliographic citation string to ranked DOI candidates. | | `doiget_batch_resolve_citations` | Batch resolve bibliographic citation strings to ranked DOI candidates. | +| `doiget_batch_from_bibliography` | Fetch every OA-resolvable entry in a Zotero / Mendeley CSL-JSON export. | +| `doiget_paper_tex_source` | Fetch an **arXiv** paper's raw LaTeX source. More reliable than `doiget_paper_text` for papers ar5iv has not processed through LaTeXML. A DOI is not a valid input. | +| `doiget_tag` | Add or remove tags and collection membership on a stored entry, for local knowledge-base organisation (#294). | +| `doiget_annotate` | Attach or clear a freeform note on a stored entry (#294). | ## 2. Naming and convention @@ -279,10 +283,50 @@ metadata. It **MUST NOT** trigger a publisher-side PDF fetch, even when the metadata source returns an OA URL. The OA URL, when known, is surfaced in the response as `oa_url` (string) for the caller to act on separately. +### `oa_url` is opt-in (#539) + +On the default path `oa_url` is `null` whenever Crossref answered -- which is +nearly every DOI -- and callers MUST NOT read that as "this work has no OA +location". + +It is **not** unconditionally null without the flag. `metadata_only_doi` keeps a +pre-existing fallback: when Crossref *fails*, Unpaywall is consulted regardless +of `include_oa_location`, and `oa_url` / `oa_status` come from that record. +`source` tells the two apart -- `crossref` means the default path answered, +`unpaywall` means the fallback did. + +The DOI path is Crossref-first, on the rationale that Crossref's +`message.link[]` supplied an OA URL without a second request. It does not: +measured across twelve live entries and eight captured fixtures, every one +carried an `intended-application` scoping it to a licensed programme +(Similarity Check, TDM, syndication) rather than being general-purpose, and +following one outside that programme would be taking a licensed route +without the licence. So the extractor refuses all of them, correctly, and the +field it was meant to fill stays empty (ADR-0052, #517). + +A caller that wants a real OA location passes `include_oa_location: true`, +which consults Unpaywall. That costs one extra metadata round-trip; it does +NOT weaken any guarantee above, because Unpaywall is a metadata source and +the URL is still reported and never followed. + +With the flag set, the `oa_status`/`oa_url` pair says which answer you got: + +| `oa_status` | `oa_url` | meaning | +|---|---|---| +| `"closed"` | `null` | the lookup completed; this work has no OA location | +| `"gold"`, `"green"`, `"hybrid"`, `"bronze"` | string | the lookup completed; here it is | +| `null` | `null` | the lookup did not complete | + +A failed Unpaywall call is NOT an error: the Crossref metadata the caller also +asked for is still returned, and the null `oa_status` is what distinguishes +"could not find out" from a completed lookup reporting `"closed"`. This +mirrors the `oa_status` + `pdf.status` pairing already used by +`doiget_fetch_paper` (§4). + ```jsonc { "name": "doiget_metadata_only", - "description": "WHEN TO USE: User wants metadata for a DOI / arXiv id without paying for or being noticed by a PDF download.\nINPUTS: ref (DOI or arXiv id), dry_run (optional bool).\nOUTPUTS: { ok: true, ref, source, license?, oa_url:string|null, metadata } or { ok:false, error }.\nCOSTS: 1-2 s metadata round-trip. No publisher fetch.\nSIDE EFFECTS: Appends a provenance row tagged 'metadata-only' (unless dry_run). Writes the metadata TOML to the store.\nLIMITS: Subject to the same rate cap as fetch_paper (5/sec). The OA URL is reported but never followed.", + "description": "WHEN TO USE: User wants metadata for a DOI / arXiv id without paying for or being noticed by a PDF download.\nINPUTS: ref (DOI or arXiv id), dry_run (optional bool), include_oa_location (optional bool).\nOUTPUTS: { ok: true, ref, source, license?, oa_url:string|null, oa_status:string|null, metadata } or { ok:false, error }.\nCOSTS: 1-2 s metadata round-trip (roughly doubled when include_oa_location). No publisher fetch.\nSIDE EFFECTS: Appends a provenance row tagged 'metadata-only' (unless dry_run). Writes the metadata TOML to the store.\nLIMITS: Subject to the same rate cap as fetch_paper (5/sec). The OA URL is reported but never followed. oa_url is null unless include_oa_location is set.", "inputSchema": { "type": "object", "required": ["ref"], @@ -293,7 +337,8 @@ response as `oa_url` (string) for the caller to act on separately. "maxLength": 256, "pattern": "^(10\\.\\d{4,9}/[A-Za-z0-9._/()-]+|arXiv:\\d{4}\\.\\d{4,5}|\\d{4}\\.\\d{4,5})$" }, - "dry_run": { "type": "boolean", "default": false } + "dry_run": { "type": "boolean", "default": false }, + "include_oa_location": { "type": "boolean", "default": false } }, "additionalProperties": false } @@ -312,7 +357,12 @@ type MetadataOnlyResult = // Currently equal to `source` verbatim. resolver_profile: string, license: string, + // Null whenever Crossref answered and `include_oa_location` was not set; + // see above for the Crossref-failure case, which fills it either way. A + // null here is NOT evidence that the work has no OA location. oa_url: string | null, + // gold / green / hybrid / bronze / closed, or null when not determined. + oa_status: string | null, metadata: object, schema_version: string, } @@ -332,3 +382,28 @@ failure. `doiget_batch_resolve_citations` batch-resolves multiple citation strings (up to 50). Both tools calculate token-based overlap similarity scores, filter out results with a score < 0.5, and sort candidates by score descending. No local store writes or provenance logs are created. + +### Branch on `confidence`, not `score` (#536) + +Each candidate carries `confidence` and `matched` alongside `score`. + +`score` alone is not enough to judge with. It is **token overlap against your +query string**, not semantic similarity, and 0.5 is the **floor** — so the +worst candidate these tools can emit still looks like a positive number. A +citation naming an author, a title, a journal, a volume and a year that comes +back at 0.5 means most of it did not match, which for a known-item lookup is a +negative result. + +| `confidence` | meaning | +|---|---| +| `exact` | every query token was found in the candidate's record | +| `probable` | at least four query tokens in five | +| `weak` | cleared the 0.5 floor and no more — a near-miss, not a match | + +`matched` lists which of your tokens were found, which is how you see whether +the author and the journal were among them. In the case that produced #536 they +were `quality`, `life`, `bipolar`, `2010` — and the returned paper was by a +different author in a different journal. + +These are bands over token overlap, not a semantic verdict: `exact` is a strong +signal and still not proof, so verify with `doiget_resolve_paper` before citing. diff --git a/docs/PROVENANCE_LOG.md b/docs/PROVENANCE_LOG.md index 7431310b1..ef3555bb6 100644 --- a/docs/PROVENANCE_LOG.md +++ b/docs/PROVENANCE_LOG.md @@ -52,6 +52,7 @@ in all timestamps. | `size_bytes` | `u64` | event=fetch ok | | | `store_path` | string | event=fetch ok | Relative to store root. | | `capability` | enum | yes | `oa` / `metadata` / `tdm-elsevier` / `tdm-aps` / `tdm-springer` | +| `error_code` | enum (`docs/ERRORS.md` §3) | `result=err` | The closed-set code for this row's failure. **Which code depends on the layer, deliberately:** a failed `fetch` leg records the *transport* mechanism (a policy-blocked OA leg is `NETWORK_ERROR` — see `ERRORS.md` §6.1), while a `session_end` row records the code the CALLER was given, after any reclassification. So the two rows for one blocked fetch legitimately differ, and each is true about its own layer. `null` on `result=ok` rows, and on a batch `session_end`, which spans many refs and has no single code (#507). | | `session_id` | ULID (26 chars) | yes | One per process invocation. | | `schema_version` | string | yes | Always the literal `"v2"` for rows written by current builds (ADR-0024). v1 rows (pre-Slice-4) lack this field; the migration tool in §"Schema migration" below brings them onto the v2 shape. | | `canonical_digest` | hex SHA-256 (64 lowercase chars) | event-dependent | ADR-0021 §1 canonical-digest of the `(source_type, source_id, resolver_profile, version)` tuple. Present on rows with a `ref` (`fetch` / `resolve` / `store_write`); `null` on session bookend rows. Two fetches of the same DOI through Crossref vs. Unpaywall produce two distinct digests. | diff --git a/docs/REDIRECT_ALLOWLIST.md b/docs/REDIRECT_ALLOWLIST.md index da3004dbc..09d5e2018 100644 --- a/docs/REDIRECT_ALLOWLIST.md +++ b/docs/REDIRECT_ALLOWLIST.md @@ -74,6 +74,42 @@ redirect_hosts = [ The exact list of entries is given in §3. +### 2.4 Transparent DOI resolvers (NORMATIVE) + +A small, closed set of hosts is **addressing, not hosting**, and is exempt from +the rule in §2.2: + +| host | what it is | +|---|---| +| `doi.org` | the canonical DOI resolver | +| `dx.doi.org` | its long-standing alias, still present in live metadata | +| `hdl.handle.net` | the Handle System resolver `doi.org` proxies | + +These are **followed** but MUST NOT be added to any source's entry, MUST NOT +appear in a denial's `expected` list, and MUST NOT be offered as remediation. +The host that actually serves the response is adjudicated exactly as it would be +otherwise, so this is transparent to §1's invariant rather than an exception to +it. + +Matching is **exact**, not the §2.2 suffix-glob. `*.doi.org` would sweep in +`www.doi.org`, which is the DOI Foundation's website and not a resolver; +`evil-doi.org` and `doi.org.evil.test` are what an attacker registers. + +**Why this rule exists.** Unpaywall routinely reports a `doi.org` URL as +`best_oa_location.url` for publisher-hosted gold OA — for `10.1002/pcn5.205` it +is the *only* location, with no `url_for_pdf`. Adjudicating that first hop as a +content host refused a cc-by paper one hop before the publisher that was already +on the list, and the denial then advised adding `doi.org`, which does not widen +the trusted surface toward one publisher: it removes the bound entirely, because +every DOI resolves through it. See [ADR-0053](DECISIONS/0053-doi-resolvers-are-addressing.md) +and #533. + +**Enforced by:** `http::is_transparent_resolver`, reached through +`SourceAllowlist::permits` — the predicate every adjudication site calls. +`SourceAllowlist::matches` remains the narrower "is this host on the list" +question and is used only to build and assert the lists. A posture-lint step +fails any gate that calls `matches` on a host variable. + ## 3. Tier 1 entries The entries below are the binding redirect-host allowlist for the Tier 1 diff --git a/docs/STORE.md b/docs/STORE.md index 30b5cd671..c45e67980 100644 --- a/docs/STORE.md +++ b/docs/STORE.md @@ -166,7 +166,19 @@ existing on-disk value WINS over whatever doiget carries (issue #123). > entry's recorded state changes. This is intentional, not silent: as of issue #118 the > blocked-PDF reason is surfaced to the caller (CLI `note:` line / MCP `pdf.status`), > so the operator always learns the entry was downgraded and why. A guard that refuses -> to downgrade is deferred (post-MVP) — it is a policy choice, not a correctness bug. +> to downgrade a **determination** is deferred (post-MVP) — it is a policy choice, not a +> correctness bug. +> +> This permission covers determinations only. Two `[doiget]` fields carry a +> not-determined *marker* rather than a reading — `oa_status` is omitted when not +> determined, and `license` falls back to `"unknown"` — and a re-write that carries +> the marker did not look, so it is not news. Per +> [ADR-0056](DECISIONS/0056-not-determined-is-not-an-answer.md) the merge keeps the +> stored value in that case. A paper that genuinely stops being open access reports +> `oa_status = "closed"`, and a changed license reports the new string; both still +> win. The distinction matters because the permission above is granted on the +> condition that the downgrade is reported, and a field the caller never asked about +> has nothing to report it against (#583). ## 7. TOML normalization diff --git a/justfile b/justfile index ad4a52e8a..55265dde8 100644 --- a/justfile +++ b/justfile @@ -25,6 +25,38 @@ lint: test: cargo test --workspace --all-targets --no-default-features --features oa-only +# Mirrors CI `test (slow)` — the `#[ignore]`d tests, which the job above skips. +# One of them obeys arXiv's published 3 s/request rate and takes ~10 minutes; +# that is the whole reason it is not in `test`. +test-slow: + cargo test --workspace --all-targets --no-default-features --features oa-only -- --ignored + +# Register THIS checkout's build as the MCP server for Claude Code, without +# touching `.mcp.json`. +# +# `.mcp.json` is the plugin's server declaration and pins the last published +# release, so opening this repo in Claude Code otherwise runs the RELEASE, not +# your working tree — you can edit `crates/doiget-mcp` all day and test the +# version you shipped last month. A local-scoped server wins over the +# project-scoped `.mcp.json`, so this shadows it for you alone and leaves the +# file (and everyone else's) unchanged. +# +# The name MUST match `.mcp.json`'s (`doiget`). Local scope only shadows a +# project-scoped server of the SAME name -- registering `doiget-dev` alongside +# it leaves BOTH connected, so the release-pinned tools stay callable and the +# problem this recipe exists for is not solved. Verified by review. +# +# `just mcp-dev-off` removes it again. +mcp-dev: + cargo build --no-default-features --features oa-only -p doiget-cli + claude mcp remove doiget --scope local || true + claude mcp add doiget --scope local -- cargo run --quiet --no-default-features --features oa-only -p doiget-cli -- serve + +# Drop the local-scoped dev server registered by `just mcp-dev`; the +# project-scoped `.mcp.json` entry becomes visible again. +mcp-dev-off: + claude mcp remove doiget --scope local + # Mirrors CI `msrv` job (build of the published surface). build: cargo build --workspace --no-default-features --features oa-only diff --git a/npm/doiget-cli/test/stage-npm.test.sh b/npm/doiget-cli/test/stage-npm.test.sh index 3ad59664b..6a0f2d8fa 100644 --- a/npm/doiget-cli/test/stage-npm.test.sh +++ b/npm/doiget-cli/test/stage-npm.test.sh @@ -142,12 +142,55 @@ fi if command -v npm > /dev/null 2>&1; then export GIT_TERMINAL_PROMPT=0 cp -r "$WORK/out" "$WORK/npm-stage" - globs="$(grep -oE '^ *for p in [^;]+' "$WORKFLOW" | sed 's/.*for p in //')" - direct="$(grep -oE 'npm publish [^ "$]+' "$WORKFLOW" | awk '{print $3}')" - if [ -z "$globs" ] || [ -z "$direct" ]; then + # A `npm-stage/doiget-*` glob is banned outright. It matches the WRAPPER -- + # `doiget-cli` starts with `doiget-` too -- so the wrapper is published + # inside the loop and again on the explicit line after it. v0.8.12's release + # job died on `You cannot publish over the previously published versions: + # 0.8.12`, after everything had already shipped: red job, complete release. + # It also sorted first, so the wrapper went out ahead of the packages its + # optionalDependencies pin. + # + # This is checked as a shape, not as a duplicate count, because the glob is + # the thing that is wrong. The same trap was spotted and excluded in + # posture-lint's `find -name doiget-*` in the very PR that renamed the + # wrapper, and missed here. + if grep -qE 'npm-stage/doiget-\*' "$WORKFLOW"; then + check no "the publish step does not glob npm-stage/doiget-*" + grep -nE 'npm-stage/doiget-\*' "$WORKFLOW" | sed 's/^/ /' + else + check ok "the publish step does not glob npm-stage/doiget-*" + fi + + # The platform list the loop actually iterates, from the same source it reads. + # + # `|| true` is load-bearing, not defensive noise. Under `set -euo pipefail` a + # bare assignment takes the exit status of its command substitution, so a + # `grep` that matches nothing kills the script HERE -- before the `check no` + # written two lines down to report exactly that. The diagnostic was + # unreachable in the one case it exists for, and CI would show a bare + # non-zero exit with none of the ok/FAIL lines around it. + looped="$(grep -oE '^doiget-[a-z0-9-]+:' "$ROOT/scripts/stage-npm.sh" | tr -d ':' | sort -u | sed 's#^#./npm-stage/#' || true)" + direct="$(grep -oE 'npm publish [^ "$]+' "$WORKFLOW" | awk '{print $3}' || true)" + if [ -z "$looped" ] || [ -z "$direct" ]; then check no "found the npm publish invocations in release-plz.yml" + else + check ok "found the npm publish invocations in release-plz.yml" fi - for spec in $globs $direct; do + + all_specs="$(printf '%s +%s +' "$looped" "$direct" | sed '/^$/d' | sed 's#^\./##')" + dupes="$(printf '%s +' "$all_specs" | sort | uniq -d)" + if [ -n "$dupes" ]; then + check no "no package is published twice" + printf ' duplicated: %s +' "$dupes" + else + check ok "no package is published twice" + fi + + for spec in $looped $direct; do # Expand the glob where the release job would expand it. entries="$(cd "$WORK" && printf '%s\n' $spec)" for d in $entries; do diff --git a/scripts/stage-npm.sh b/scripts/stage-npm.sh old mode 100644 new mode 100755 diff --git a/scripts/update-homebrew-formula.sh b/scripts/update-homebrew-formula.sh new file mode 100755 index 000000000..23c4464ba --- /dev/null +++ b/scripts/update-homebrew-formula.sh @@ -0,0 +1,105 @@ +#!/usr/bin/env bash +# Regenerate Formula/doiget.rb for a released version (#501). +# +# The formula is GENERATED, never hand-edited: #247 was closed as completed +# while four fifths of it had not shipped, and a hand-maintained formula is +# exactly the kind of thing that half-lands the same way. +# +# A maintainer runs this after a stable release, not the release workflow -- +# see CONTRIBUTING.md and #426. The generator cannot be run WRONG (posture-lint +# asserts the committed formula is its output, and refuses malformed +# checksums); it can only be FORGOTTEN, which the release-sync posture check +# turns red as soon as the other release-tracking files are bumped. +# +# Usage: +# scripts/update-homebrew-formula.sh +# scripts/update-homebrew-formula.sh +# +# With only a version, the checksums are read from that release's published +# `.sha256` assets -- the same files the curl installer verifies against, so +# the formula cannot disagree with the binary it installs. The four-argument +# form exists so the generator can be tested without a network. +set -euo pipefail + +VERSION="${1:?usage: update-homebrew-formula.sh [sha_arm sha_x86 sha_linux]}" +REPO="${DOIGET_REPO:-QAtlasHub/doiget}" +OUT="${DOIGET_FORMULA_OUT:-Formula/doiget.rb}" + +# A leading `v` is how the tag is written and NOT how the formula version is; +# accept either and normalise, so a caller passing the tag is not silently +# wrong. +VERSION="${VERSION#v}" + +fetch_sha() { + local asset="$1" + # `.sha256` files are ` `; take the first field. + curl --proto '=https' --tlsv1.2 -fsSL \ + "https://github.com/${REPO}/releases/download/v${VERSION}/${asset}.sha256" \ + | awk '{print $1; exit}' +} + +if [ "$#" -ge 4 ]; then + SHA_ARM="$2"; SHA_X86="$3"; SHA_LINUX="$4" +else + SHA_ARM="$(fetch_sha doiget-macos-aarch64)" + SHA_X86="$(fetch_sha doiget-macos-x86_64)" + SHA_LINUX="$(fetch_sha doiget-linux-x86_64)" +fi + +for pair in "SHA_ARM:$SHA_ARM" "SHA_X86:$SHA_X86" "SHA_LINUX:$SHA_LINUX"; do + name="${pair%%:*}"; val="${pair#*:}" + if ! printf '%s' "$val" | grep -qE '^[0-9a-f]{64}$'; then + echo "::error::update-homebrew-formula: $name is not a sha256: '$val'" >&2 + exit 1 + fi +done + +mkdir -p "$(dirname "$OUT")" +cat > "$OUT" <\` after a stable +# release has published its assets. That is a MAINTAINER step, not a workflow +# step -- committing the formula back from CI needs a token with write access +# to a protected branch, and #426 says that token is broken. So the formula +# lags the tag until someone runs the script, and CONTRIBUTING.md says so +# (#501). +class Doiget < Formula + desc "Open Access paper fetcher with an MCP server" + homepage "https://github.com/${REPO}" + version "${VERSION}" + license "MIT" + + # The published binaries are the release assets themselves, not archives, so + # each URL is fetched verbatim. They are statically linked (musl on Linux), + # which is why there are no dependencies to declare. + on_macos do + on_arm do + url "https://github.com/${REPO}/releases/download/v${VERSION}/doiget-macos-aarch64" + sha256 "${SHA_ARM}" + end + on_intel do + url "https://github.com/${REPO}/releases/download/v${VERSION}/doiget-macos-x86_64" + sha256 "${SHA_X86}" + end + end + + on_linux do + on_intel do + url "https://github.com/${REPO}/releases/download/v${VERSION}/doiget-linux-x86_64" + sha256 "${SHA_LINUX}" + end + end + + def install + # One asset per platform, whichever was downloaded. + bin.install Dir["doiget-*"].first => "doiget" + end + + test do + assert_match "doiget #{version}", shell_output("#{bin}/doiget --version") + end +end +RUBY + +echo "wrote $OUT for v${VERSION}" diff --git a/scripts/update-homebrew-formula.test.sh b/scripts/update-homebrew-formula.test.sh new file mode 100755 index 000000000..50a9d94ec --- /dev/null +++ b/scripts/update-homebrew-formula.test.sh @@ -0,0 +1,74 @@ +#!/usr/bin/env bash +# Offline tests for scripts/update-homebrew-formula.sh (#501). +# +# The generator is what stands between "the formula is right" and "somebody +# remembered to edit it", so the properties that would silently ship a broken +# install are asserted here rather than discovered by a `brew install`. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +GEN="$ROOT/scripts/update-homebrew-formula.sh" +TMP="$(mktemp -d)" +# Named files only, never a recursive delete of a variable path. +cleanup() { rm -f "$TMP/doiget.rb" "$TMP/doiget-tag.rb" "$TMP/bad.rb" "$TMP/regen.rb"; rmdir "$TMP" 2>/dev/null || true; } +trap cleanup EXIT + +A=$(printf 'a%.0s' $(seq 64)) +B=$(printf 'b%.0s' $(seq 64)) +C=$(printf 'c%.0s' $(seq 64)) + +fail=0 +check() { + if [ "$2" != "$3" ]; then + echo "FAIL: $1: expected '$3', got '$2'"; fail=1 + else + echo "ok: $1" + fi +} + +# ---------------------------------------------------------------- generation +OUT="$TMP/doiget.rb" +DOIGET_FORMULA_OUT="$OUT" bash "$GEN" 1.2.3 "$A" "$B" "$C" >/dev/null + +check "version is set" "$(grep -c '^ version "1.2.3"$' "$OUT")" "1" +check "all three platform urls" "$(grep -c 'releases/download/v1.2.3/doiget-' "$OUT")" "3" +check "all three checksums" "$(grep -c '^ sha256 "' "$OUT")" "3" +check "arm checksum placed under on_arm" \ + "$(awk '/on_arm do/{f=1} f&&/sha256/{print substr($2,2,3); exit}' "$OUT")" "aaa" + +# A leading `v` is how the TAG is written. Accepting it silently and emitting +# `version "v1.2.3"` would produce a formula that installs but compares wrong. +OUT2="$TMP/doiget-tag.rb" +DOIGET_FORMULA_OUT="$OUT2" bash "$GEN" v1.2.3 "$A" "$B" "$C" >/dev/null +check "a leading v is normalised" "$(cmp -s "$OUT" "$OUT2" && echo same || echo differs)" "same" + +# The `#{...}` in the `test do` block is Ruby interpolation and must survive +# the shell heredoc intact. It leaked out backslash-escaped once. +check "no leaked backslash before ruby interpolation" \ + "$(grep -c '[\]#{' "$OUT")" "0" +check "ruby interpolation present" "$(grep -c 'doiget #{version}' "$OUT")" "1" + +# ------------------------------------------------------------- sha validation +# A truncated or missing checksum must stop the generator, not produce a +# formula that fails at `brew install` on someone else's machine. +for bad in "" "notahex" "$(printf 'a%.0s' $(seq 63))" "$A$A"; do + if DOIGET_FORMULA_OUT="$TMP/bad.rb" bash "$GEN" 1.2.3 "$bad" "$B" "$C" >/dev/null 2>&1; then + echo "FAIL: generator accepted a bad sha256: '$bad'"; fail=1 + fi +done +echo "ok: bad checksums are refused" + +# ------------------------------------------------ the shipped formula is fresh +# The formula in the tree must be something the generator would produce, so a +# hand-edit is caught rather than merged. +STABLE="$(grep -oE '^ version "[^"]+"' "$ROOT/Formula/doiget.rb" | grep -oE '[0-9][^"]*')" +SHIPPED_SHAS=$(grep -oE '^ sha256 "[0-9a-f]{64}"' "$ROOT/Formula/doiget.rb" | grep -oE '[0-9a-f]{64}') +# shellcheck disable=SC2086 +DOIGET_FORMULA_OUT="$TMP/regen.rb" bash "$GEN" "$STABLE" $SHIPPED_SHAS >/dev/null +check "Formula/doiget.rb is generator output, not hand-edited" \ + "$(cmp -s "$ROOT/Formula/doiget.rb" "$TMP/regen.rb" && echo same || echo differs)" "same" + +if [ "$fail" -ne 0 ]; then + echo "update-homebrew-formula tests FAILED"; exit 1 +fi +echo "update-homebrew-formula tests passed" diff --git a/site/content/developer/errors.md b/site/content/developer/errors.md index e8b6195cb..3005514ed 100644 --- a/site/content/developer/errors.md +++ b/site/content/developer/errors.md @@ -39,29 +39,29 @@ Wire form (JSON / MCP): `"INVALID_REF"`, `"NO_OA_AVAILABLE"`, etc. ## 2. Code semantics -| Code | Meaning | Recoverable? | -|---|---|---| -| `INVALID_REF` | DOI / arXiv id failed validation. | No (user must correct input). | -| `NO_OA_AVAILABLE` | Tier 1 sources reported no OA URL. | Try later, or enable opt-in source. | -| `RATE_LIMITED` | Internal rate cap hit, OR 429 from source. | Retry after `Retry-After` (or 1 s). | -| `NETWORK_ERROR` | Transport / DNS / TLS failure. **Does NOT cover a deliberate supply-chain policy block** — see §6.1: an off-allowlist / redirect-denied / insecure-scheme OA-PDF leg is `CAPABILITY_DENIED`, not `NETWORK_ERROR`. | Retry usually fine. | -| `NOT_FOUND` | Metadata source authoritatively reported the id does not exist: HTTP `404` / `410` / `451`, or a source-specific absence (arXiv returns HTTP 200 with an empty `` for an unknown id). Network-independent and reproducible — distinct from the transient `NETWORK_ERROR` / `RATE_LIMITED`. `doiget verify` treats it as a definite dead reference (`absent`). For a DOI it is emitted only when all configured sources (Crossref, Unpaywall) fail to resolve it; a DataCite-only DOI may thus be reported `NOT_FOUND`. | No (the id is wrong or retracted). | -| `AMBIGUOUS` | A name filter (`--author` / `--venue` / `--publisher`) matched several entities with no clear winner; the error lists the candidates. Distinct from `NOT_FOUND` ("matched nothing"). CLI exit `2`. | Yes — narrow the name (add a first name / fuller title) or pass an exact id. | -| `STORE_ERROR` | Filesystem write failed (disk, permission, etc.). | Depends on cause. | -| `LOG_ERROR` | Provenance log write failed. **Fetch is aborted.** | Free disk / fix perms. | -| `CAPABILITY_DENIED` | Source not in `CapabilityProfile`. | User opts in, or pick different source. | -| `FETCH_TIMEOUT` | Per-request timeout exceeded. | Retry. | -| `SCHEMA_TOO_NEW` | Store entry's `schema_version` is ahead. | Upgrade doiget. | -| `LOCK_TIMEOUT` | Could not acquire `flock` within 5 s. | Retry; another process holds it. | -| `INTERNAL_ERROR` | Bug. | Report at . | -| `NOT_IMPLEMENTED` | Feature is spec'd but not yet wired in this Phase. | Wait for next minor release; do not retry. | -| `TEXT_UNAVAILABLE` | The id is valid and resolvable, but the **requested representation** is missing: `doiget text` got a 200 from ar5iv with no extractable prose (the paper was never converted to HTML). Distinct from `NOT_FOUND` (the id *does* exist) and `NO_OA_AVAILABLE` (the paper may still be OA — only the HTML render is missing). Issue #302. | Yes — fetch the PDF instead (`doiget fetch `); do not "fix" the identifier. | +| Code | Meaning | Disposition | Recoverable? | +|---|---|---|---| +| `INVALID_REF` | DOI / arXiv id failed validation. | `terminal` | No (user must correct input). | +| `NO_OA_AVAILABLE` | No source could supply a free copy: Tier 1 reported no OA URL, **or** a source holds the record and has no retrievable copy (`FetchError::NotRetrievable`, ADR-0054). | `needs_config` | Try later, or enable opt-in source. | +| `RATE_LIMITED` | Internal rate cap hit, OR 429 from source. | `retry_after` | Retry after `Retry-After` (or 1 s). | +| `NETWORK_ERROR` | Transport / DNS / TLS failure. **Does NOT cover a deliberate supply-chain policy block** — see §6.1: an off-allowlist / redirect-denied / insecure-scheme OA-PDF leg is `CAPABILITY_DENIED`, not `NETWORK_ERROR`. | `retry_after` | Retry usually fine. | +| `NOT_FOUND` | Metadata source authoritatively reported the id does not exist: HTTP `404` / `410` / `451`, or a source-specific absence (arXiv returns HTTP 200 with an empty `` for an unknown id). Network-independent and reproducible — distinct from the transient `NETWORK_ERROR` / `RATE_LIMITED`. `doiget verify` treats it as a definite dead reference (`absent`). For a DOI it is emitted only when all configured sources (Crossref, Unpaywall) fail to resolve it; a DataCite-only DOI may thus be reported `NOT_FOUND`. | `terminal` | No (the id is wrong or retracted). | +| `AMBIGUOUS` | A name filter (`--author` / `--venue` / `--publisher`) matched several entities with no clear winner; the error lists the candidates. Distinct from `NOT_FOUND` ("matched nothing"). CLI exit `2`. | `terminal` | Yes — narrow the name (add a first name / fuller title) or pass an exact id. | +| `STORE_ERROR` | The local store could not serve the request: a filesystem write failed (disk, permission, etc.), or the entry a mutating tool needs is not there yet (`doiget_tag` / `doiget_annotate` on a ref nobody has fetched). Deliberately NOT `NOT_FOUND` in that second case: the id is fine, and `NOT_FOUND` would tell a caller the reference is dead. The read-only tools (`doiget_info`, `doiget_paper_pdf_path`, `doiget_search_local`) report a store miss as `ok: true` with a null payload instead, which is preferable where the operation has nothing to mutate. | `needs_config` | Depends on cause; for a store miss, fetch the paper first. | +| `LOG_ERROR` | Provenance log write failed. **Fetch is aborted.** | `needs_config` | Free disk / fix perms. | +| `CAPABILITY_DENIED` | Source not in `CapabilityProfile`. | `needs_config` | User opts in, or pick different source. | +| `FETCH_TIMEOUT` | Per-request timeout exceeded. | `retry_after` | Retry. | +| `SCHEMA_TOO_NEW` | Store entry's `schema_version` is ahead. | `needs_config` | Upgrade doiget. | +| `LOCK_TIMEOUT` | Could not acquire `flock` within 5 s. | `retry_after` | Retry; another process holds it. | +| `INTERNAL_ERROR` | Bug. | `terminal` | Report at . | +| `NOT_IMPLEMENTED` | Feature is spec'd but not yet wired in this Phase. | `terminal` | Wait for next minor release; do not retry. | +| `TEXT_UNAVAILABLE` | The id is valid and resolvable, but the **requested representation** is missing: `doiget text` got a 200 from ar5iv with no extractable prose (the paper was never converted to HTML). Distinct from `NOT_FOUND` (the id *does* exist) and `NO_OA_AVAILABLE` (the paper may still be OA — only the HTML render is missing). Issue #302. | `terminal` | Yes — fetch the PDF instead (`doiget fetch `); do not "fix" the identifier. | ## 3. Persona × error matrix | Persona | Surface | |---|---| -| Agent (MCP) | Structured, never throws. On failure: `{ ok: false, error: { code, message, denial_context? } }`. `remediation` and `attempts` are **not** carried here — they belong to the `{ ok: true, … }` envelope, as `pdf.remediation` (present when the PDF leg was blocked) and top-level `attempts`. A blocked PDF leg is an `ok: true` result with a failed leg, not an `ok: false` call. | +| Agent (MCP) | Structured, never throws. On failure: `{ ok: false, error: { code, message, disposition, retry_after_ms?, remediation?, denial_context? } }`. `disposition` is present on every failure that carries a structured `error` OBJECT, and is the field to branch a retry on — see §2 and ADR-0055. `disposition` is now present on **every** `ok: false` envelope. Four tools -- `doiget_resolve_citation`, `doiget_batch_resolve_citations`, `doiget_tag` and `doiget_annotate` -- used to put a bare string in `error` instead, so a caller had nothing to branch on; they build the object like every other tool as of 0.8.13 **(breaking for anyone reading `error` as a string on those four)**. The set is pinned EMPTY by `every_bare_string_error_site_is_a_known_one`, and `the_exemption_list_and_the_document_agree` fails if this paragraph and that list ever disagree. Both are new: this document already named the guard as the reason the set "can shrink but not grow", and the guard had not been written -- a claim about the world resting on code that did not exist, which is exactly what this document defines. It is derived from `code` by `ErrorCode::disposition`, never hand-written per call site. `remediation` IS carried here when the failure has a `denial_context`, from the same `remediation::for_denial` the blocked leg and the CLI `= help:` block use — this document previously said it was not, which meant the one field naming the fix was present when a call succeeded with a blocked leg and absent when the call actually failed (#506). It is omitted when there is no named fix rather than emitted empty. `attempts` remains an `ok: true` field. A blocked PDF leg is still an `ok: true` result with a failed leg, not an `ok: false` call, and it carries `pdf.remediation` plus `pdf.disposition`. `retry_after_ms` is present only when the SERVER sent a `Retry-After` on the response that ended the attempt; it is never backfilled from doiget's internal backoff, because a guess about the server wearing the name of a server-supplied value is the defect these fields exist to remove (#506). | | Researcher (CLI human) | `cargo`-style stderr: `error[E0007]: rate limited from unpaywall: retry after 1s`. Exit code 1. | | CI / Batch (CLI `--json`) | JSON Lines record per ref with `{"ok":false, "error":{"code":"...","message":"...","denial_context":{...}?,"remediation":[...]?,"attempts":[...]?}}`. Exit code = number of failures (capped at 255). **Records are emitted in completion order, not input order** — see below. | | Library (Rust) | `Err(FetchError)` (typed via `thiserror`). | diff --git a/site/content/developer/mcp-tools.md b/site/content/developer/mcp-tools.md index 28653ace8..eb49ad8d4 100644 --- a/site/content/developer/mcp-tools.md +++ b/site/content/developer/mcp-tools.md @@ -23,7 +23,7 @@ speaks **stdio only** ([ADR-0001](DECISIONS/), [`SCOPE.md`](SCOPE.md) §non-goal | `doiget_info` | Retrieve a store entry's metadata. | | `doiget_search_local` | Search store metadata (title / authors / venue). | | `doiget_paper_search` | External literature discovery over OpenAlex (`/works?search=`); abstract-bearing candidates for triage. Tier-1 OA metadata, always-on; **never fetches a PDF** (ADR-0031). | -| `doiget_paper_text` | Extract an **arXiv** paper's full text from ar5iv as sectioned plain text (`ref`, optional `max_chars`). Tier-1 OA, always-on; **never opens the PDF blob** (ADR-0032). A DOI → `NO_OA_AVAILABLE`. | +| `doiget_paper_text` | Extract an **arXiv** paper's full text from ar5iv as sectioned plain text (`ref`, optional `max_chars`). Tier-1 OA, always-on; **never opens the PDF blob** (ADR-0032). A DOI → `NOT_IMPLEMENTED` (terminal: the tool is arXiv-only and DOI→arXiv linking is #281 item 5, not a config knob). | | `doiget_link` | Resolve a **DOI** to its arXiv preprint + identity cluster (`{ doi, arxiv, openalex_id, title }`) over OpenAlex, for reading or dedup (#281 item 5). Tier-1 OA, always-on; **never fetches a PDF**. arXiv → DOI is a follow-up; a non-DOI ref → `INVALID_REF`. | | `doiget_list_recent` | Last N fetched entries. | | `doiget_paper_pdf_path` | Return the local path of a cached PDF. **Does not read, parse, or transmit content.** | @@ -39,6 +39,10 @@ Additional tools: | `doiget_csl_export` | CSL JSON for one or many entries. | | `doiget_resolve_citation` | Resolve a free-form bibliographic citation string to ranked DOI candidates. | | `doiget_batch_resolve_citations` | Batch resolve bibliographic citation strings to ranked DOI candidates. | +| `doiget_batch_from_bibliography` | Fetch every OA-resolvable entry in a Zotero / Mendeley CSL-JSON export. | +| `doiget_paper_tex_source` | Fetch an **arXiv** paper's raw LaTeX source. More reliable than `doiget_paper_text` for papers ar5iv has not processed through LaTeXML. A DOI is not a valid input. | +| `doiget_tag` | Add or remove tags and collection membership on a stored entry, for local knowledge-base organisation (#294). | +| `doiget_annotate` | Attach or clear a freeform note on a stored entry (#294). | ## 2. Naming and convention @@ -285,10 +289,50 @@ metadata. It **MUST NOT** trigger a publisher-side PDF fetch, even when the metadata source returns an OA URL. The OA URL, when known, is surfaced in the response as `oa_url` (string) for the caller to act on separately. +### `oa_url` is opt-in (#539) + +On the default path `oa_url` is `null` whenever Crossref answered -- which is +nearly every DOI -- and callers MUST NOT read that as "this work has no OA +location". + +It is **not** unconditionally null without the flag. `metadata_only_doi` keeps a +pre-existing fallback: when Crossref *fails*, Unpaywall is consulted regardless +of `include_oa_location`, and `oa_url` / `oa_status` come from that record. +`source` tells the two apart -- `crossref` means the default path answered, +`unpaywall` means the fallback did. + +The DOI path is Crossref-first, on the rationale that Crossref's +`message.link[]` supplied an OA URL without a second request. It does not: +measured across twelve live entries and eight captured fixtures, every one +carried an `intended-application` scoping it to a licensed programme +(Similarity Check, TDM, syndication) rather than being general-purpose, and +following one outside that programme would be taking a licensed route +without the licence. So the extractor refuses all of them, correctly, and the +field it was meant to fill stays empty (ADR-0052, #517). + +A caller that wants a real OA location passes `include_oa_location: true`, +which consults Unpaywall. That costs one extra metadata round-trip; it does +NOT weaken any guarantee above, because Unpaywall is a metadata source and +the URL is still reported and never followed. + +With the flag set, the `oa_status`/`oa_url` pair says which answer you got: + +| `oa_status` | `oa_url` | meaning | +|---|---|---| +| `"closed"` | `null` | the lookup completed; this work has no OA location | +| `"gold"`, `"green"`, `"hybrid"`, `"bronze"` | string | the lookup completed; here it is | +| `null` | `null` | the lookup did not complete | + +A failed Unpaywall call is NOT an error: the Crossref metadata the caller also +asked for is still returned, and the null `oa_status` is what distinguishes +"could not find out" from a completed lookup reporting `"closed"`. This +mirrors the `oa_status` + `pdf.status` pairing already used by +`doiget_fetch_paper` (§4). + ```jsonc { "name": "doiget_metadata_only", - "description": "WHEN TO USE: User wants metadata for a DOI / arXiv id without paying for or being noticed by a PDF download.\nINPUTS: ref (DOI or arXiv id), dry_run (optional bool).\nOUTPUTS: { ok: true, ref, source, license?, oa_url:string|null, metadata } or { ok:false, error }.\nCOSTS: 1-2 s metadata round-trip. No publisher fetch.\nSIDE EFFECTS: Appends a provenance row tagged 'metadata-only' (unless dry_run). Writes the metadata TOML to the store.\nLIMITS: Subject to the same rate cap as fetch_paper (5/sec). The OA URL is reported but never followed.", + "description": "WHEN TO USE: User wants metadata for a DOI / arXiv id without paying for or being noticed by a PDF download.\nINPUTS: ref (DOI or arXiv id), dry_run (optional bool), include_oa_location (optional bool).\nOUTPUTS: { ok: true, ref, source, license?, oa_url:string|null, oa_status:string|null, metadata } or { ok:false, error }.\nCOSTS: 1-2 s metadata round-trip (roughly doubled when include_oa_location). No publisher fetch.\nSIDE EFFECTS: Appends a provenance row tagged 'metadata-only' (unless dry_run). Writes the metadata TOML to the store.\nLIMITS: Subject to the same rate cap as fetch_paper (5/sec). The OA URL is reported but never followed. oa_url is null unless include_oa_location is set.", "inputSchema": { "type": "object", "required": ["ref"], @@ -299,7 +343,8 @@ response as `oa_url` (string) for the caller to act on separately. "maxLength": 256, "pattern": "^(10\\.\\d{4,9}/[A-Za-z0-9._/()-]+|arXiv:\\d{4}\\.\\d{4,5}|\\d{4}\\.\\d{4,5})$" }, - "dry_run": { "type": "boolean", "default": false } + "dry_run": { "type": "boolean", "default": false }, + "include_oa_location": { "type": "boolean", "default": false } }, "additionalProperties": false } @@ -318,7 +363,12 @@ type MetadataOnlyResult = // Currently equal to `source` verbatim. resolver_profile: string, license: string, + // Null whenever Crossref answered and `include_oa_location` was not set; + // see above for the Crossref-failure case, which fills it either way. A + // null here is NOT evidence that the work has no OA location. oa_url: string | null, + // gold / green / hybrid / bronze / closed, or null when not determined. + oa_status: string | null, metadata: object, schema_version: string, } @@ -338,3 +388,28 @@ failure. `doiget_batch_resolve_citations` batch-resolves multiple citation strings (up to 50). Both tools calculate token-based overlap similarity scores, filter out results with a score < 0.5, and sort candidates by score descending. No local store writes or provenance logs are created. + +### Branch on `confidence`, not `score` (#536) + +Each candidate carries `confidence` and `matched` alongside `score`. + +`score` alone is not enough to judge with. It is **token overlap against your +query string**, not semantic similarity, and 0.5 is the **floor** — so the +worst candidate these tools can emit still looks like a positive number. A +citation naming an author, a title, a journal, a volume and a year that comes +back at 0.5 means most of it did not match, which for a known-item lookup is a +negative result. + +| `confidence` | meaning | +|---|---| +| `exact` | every query token was found in the candidate's record | +| `probable` | at least four query tokens in five | +| `weak` | cleared the 0.5 floor and no more — a near-miss, not a match | + +`matched` lists which of your tokens were found, which is how you see whether +the author and the journal were among them. In the case that produced #536 they +were `quality`, `life`, `bipolar`, `2010` — and the returned paper was by a +different author in a different journal. + +These are bands over token overlap, not a semantic verdict: `exact` is a strong +signal and still not proof, so verify with `doiget_resolve_paper` before citing. diff --git a/site/content/developer/provenance-log.md b/site/content/developer/provenance-log.md index 9ecf01579..9cdd4c143 100644 --- a/site/content/developer/provenance-log.md +++ b/site/content/developer/provenance-log.md @@ -58,6 +58,7 @@ in all timestamps. | `size_bytes` | `u64` | event=fetch ok | | | `store_path` | string | event=fetch ok | Relative to store root. | | `capability` | enum | yes | `oa` / `metadata` / `tdm-elsevier` / `tdm-aps` / `tdm-springer` | +| `error_code` | enum (`docs/ERRORS.md` §3) | `result=err` | The closed-set code for this row's failure. **Which code depends on the layer, deliberately:** a failed `fetch` leg records the *transport* mechanism (a policy-blocked OA leg is `NETWORK_ERROR` — see `ERRORS.md` §6.1), while a `session_end` row records the code the CALLER was given, after any reclassification. So the two rows for one blocked fetch legitimately differ, and each is true about its own layer. `null` on `result=ok` rows, and on a batch `session_end`, which spans many refs and has no single code (#507). | | `session_id` | ULID (26 chars) | yes | One per process invocation. | | `schema_version` | string | yes | Always the literal `"v2"` for rows written by current builds (ADR-0024). v1 rows (pre-Slice-4) lack this field; the migration tool in §"Schema migration" below brings them onto the v2 shape. | | `canonical_digest` | hex SHA-256 (64 lowercase chars) | event-dependent | ADR-0021 §1 canonical-digest of the `(source_type, source_id, resolver_profile, version)` tuple. Present on rows with a `ref` (`fetch` / `resolve` / `store_write`); `null` on session bookend rows. Two fetches of the same DOI through Crossref vs. Unpaywall produce two distinct digests. | diff --git a/site/content/developer/redirect-allowlist.md b/site/content/developer/redirect-allowlist.md index e987f6320..b3b076cfd 100644 --- a/site/content/developer/redirect-allowlist.md +++ b/site/content/developer/redirect-allowlist.md @@ -80,6 +80,42 @@ redirect_hosts = [ The exact list of entries is given in §3. +### 2.4 Transparent DOI resolvers (NORMATIVE) + +A small, closed set of hosts is **addressing, not hosting**, and is exempt from +the rule in §2.2: + +| host | what it is | +|---|---| +| `doi.org` | the canonical DOI resolver | +| `dx.doi.org` | its long-standing alias, still present in live metadata | +| `hdl.handle.net` | the Handle System resolver `doi.org` proxies | + +These are **followed** but MUST NOT be added to any source's entry, MUST NOT +appear in a denial's `expected` list, and MUST NOT be offered as remediation. +The host that actually serves the response is adjudicated exactly as it would be +otherwise, so this is transparent to §1's invariant rather than an exception to +it. + +Matching is **exact**, not the §2.2 suffix-glob. `*.doi.org` would sweep in +`www.doi.org`, which is the DOI Foundation's website and not a resolver; +`evil-doi.org` and `doi.org.evil.test` are what an attacker registers. + +**Why this rule exists.** Unpaywall routinely reports a `doi.org` URL as +`best_oa_location.url` for publisher-hosted gold OA — for `10.1002/pcn5.205` it +is the *only* location, with no `url_for_pdf`. Adjudicating that first hop as a +content host refused a cc-by paper one hop before the publisher that was already +on the list, and the denial then advised adding `doi.org`, which does not widen +the trusted surface toward one publisher: it removes the bound entirely, because +every DOI resolves through it. See [ADR-0053](DECISIONS/0053-doi-resolvers-are-addressing.md) +and #533. + +**Enforced by:** `http::is_transparent_resolver`, reached through +`SourceAllowlist::permits` — the predicate every adjudication site calls. +`SourceAllowlist::matches` remains the narrower "is this host on the list" +question and is used only to build and assert the lists. A posture-lint step +fails any gate that calls `matches` on a host variable. + ## 3. Tier 1 entries The entries below are the binding redirect-host allowlist for the Tier 1 diff --git a/site/content/developer/store.md b/site/content/developer/store.md index 0cccce7fd..c229f5e8b 100644 --- a/site/content/developer/store.md +++ b/site/content/developer/store.md @@ -172,7 +172,19 @@ existing on-disk value WINS over whatever doiget carries (issue #123). > entry's recorded state changes. This is intentional, not silent: as of issue #118 the > blocked-PDF reason is surfaced to the caller (CLI `note:` line / MCP `pdf.status`), > so the operator always learns the entry was downgraded and why. A guard that refuses -> to downgrade is deferred (post-MVP) — it is a policy choice, not a correctness bug. +> to downgrade a **determination** is deferred (post-MVP) — it is a policy choice, not a +> correctness bug. +> +> This permission covers determinations only. Two `[doiget]` fields carry a +> not-determined *marker* rather than a reading — `oa_status` is omitted when not +> determined, and `license` falls back to `"unknown"` — and a re-write that carries +> the marker did not look, so it is not news. Per +> [ADR-0056](DECISIONS/0056-not-determined-is-not-an-answer.md) the merge keeps the +> stored value in that case. A paper that genuinely stops being open access reports +> `oa_status = "closed"`, and a changed license reports the new string; both still +> win. The distinction matters because the permission above is granted on the +> condition that the downgrade is reported, and a field the caller never asked about +> has nothing to report it against (#583). ## 7. TOML normalization diff --git a/supply-chain/config.toml b/supply-chain/config.toml index 8089eacd7..94f5e366b 100644 --- a/supply-chain/config.toml +++ b/supply-chain/config.toml @@ -223,7 +223,7 @@ version = "0.1.9" criteria = "safe-to-deploy" [[exemptions.flate2]] -version = "1.1.9" +version = "1.1.10" criteria = "safe-to-deploy" [[exemptions.float-cmp]] @@ -434,6 +434,10 @@ criteria = "safe-to-deploy" version = "2.8.0" criteria = "safe-to-deploy" +[[exemptions.miniz_oxide]] +version = "0.9.1" +criteria = "safe-to-deploy" + [[exemptions.mio]] version = "1.2.0" criteria = "safe-to-deploy" @@ -495,7 +499,7 @@ version = "0.36.2" criteria = "safe-to-deploy" [[exemptions.quick-xml]] -version = "0.41.0" +version = "0.42.0" criteria = "safe-to-deploy" [[exemptions.quinn]] @@ -875,7 +879,7 @@ version = "2.5.8" criteria = "safe-to-deploy" [[exemptions.uuid]] -version = "1.25.0" +version = "1.26.0" criteria = "safe-to-deploy" [[exemptions.walkdir]] @@ -1094,6 +1098,10 @@ criteria = "safe-to-deploy" version = "0.11.3" criteria = "safe-to-deploy" +[[exemptions.zlib-rs]] +version = "0.6.7" +criteria = "safe-to-deploy" + [[exemptions.zmij]] version = "1.0.21" criteria = "safe-to-deploy"