diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 78f312cf..254c39dc 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -68,7 +68,7 @@ jobs: run: python scripts/check_oss_boundary.py --root . --allowlist config/oss_boundary_allowlist.json - name: Lint committed BENCHMARK.md fixtures for leak patterns if: ${{ needs.classify-changes.outputs.docs_only != 'true' }} - run: python scripts/ci/check_public_benchmarks.py --require-files tests/golden + run: uv run python scripts/ci/check_public_benchmarks.py --require-files tests/golden - name: Lint if: ${{ needs.classify-changes.outputs.docs_only != 'true' }} run: uv run ruff check . @@ -276,6 +276,7 @@ jobs: tests/test_cli.py tests/test_harbor_output_provenance.py tests/test_harbor_runtime_skill_isolation.py + tests/test_publication_identity.py tests/test_harbor_secure_copy.py::test_checked_windows_fallback_accepts_crt_descriptor_identity tests/test_harbor_secure_copy.py::test_checked_windows_fallback_verifies_portable_chmod_semantics tests/test_harbor_secure_copy.py::test_native_windows_fallback_copies_tree_and_file diff --git a/CHANGELOG.md b/CHANGELOG.md index f54c4e84..88f9088a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,28 @@ All notable changes to SkillEvaluator are documented in this file. conversion limit, preserves nonzero Wilson interval widths and paired-effect directions at large case counts, and documents exact-rational omission markers. +- `BENCHMARK.md` publication verdicts now require completed Tier 1 and Tier 2 + execution evidence by default, conservatively resolve conflicting peer + policy and Tier 3 result metadata, bind every tier and policy claim to one + versioned source-tree digest, preserve the Tier 3 run ID, persist an explicit + publication status in JSON and HTML, and reject + publication `PASS` cards whose decision evidence is missing, incomplete, + linked, or hidden in raw HTML. Custom Tier 3 result roots inside the skill are + rejected; use the canonical `evals/results` path or an external results root. +- Publication source identity now seals forward, reverse, and final source reads, + normalizes filesystem case aliases, aligns Tier 1/Tier 2 and Tier 3 runtime + projections with the v2 generated-artifact exclusions, and carries a bounded + source-change marker through rerendered Tier 3 evidence. +- Report output is required outside the publication target. A default + in-target `reports/` location relocates to an authenticated sibling + `-reports` (or `-reports`), refuses unowned collisions, and is + excluded from Tier 3 full-repository staging. +- Public benchmark provenance now rejects placeholder identities, malformed or + future calendar dates, missing duplicated run IDs, hostile benchmark-policy + metadata, and environment-label substitutions in required proof fields. +- Hardened Tier 3 HTML and Markdown reporting against recursive or malformed + display metadata, Markdown structure injection, and inline-JavaScript + injection through untrusted agent or skill names. ## 0.2.1 - 2026-08-24 diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index 790b89b9..66dd7530 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -10,3 +10,46 @@ records the exact resolved dependency set used for this release. | LLM | Anthropic (MIT), Boto3 (Apache-2.0), LiteLLM (MIT), OpenAI (Apache-2.0) | | Tier 3 | Harbor (Apache-2.0) | | Security | Bandit (Apache-2.0), pip-audit (Apache-2.0) | + +## Unicode security data + +`src/skillevaluator/publication_text.py` contains Unicode 15.1 general-category +ranges and a generated subset of the Unicode 17.0.0 `confusables.txt` data used +by Unicode Technical Standard #39. + +UNICODE LICENSE V3 + +COPYRIGHT AND PERMISSION NOTICE + +Copyright © 1991-2026 Unicode, Inc. + +NOTICE TO USER: Carefully read the following legal agreement. BY DOWNLOADING, +INSTALLING, COPYING OR OTHERWISE USING DATA FILES, AND/OR SOFTWARE, YOU +UNEQUIVOCALLY ACCEPT, AND AGREE TO BE BOUND BY, ALL OF THE TERMS AND CONDITIONS +OF THIS AGREEMENT. IF YOU DO NOT AGREE, DO NOT DOWNLOAD, INSTALL, COPY, +DISTRIBUTE OR USE THE DATA FILES OR SOFTWARE. + +Permission is hereby granted, free of charge, to any person obtaining a copy of +data files and any associated documentation (the "Data Files") or software and +any associated documentation (the "Software") to deal in the Data Files or +Software without restriction, including without limitation the rights to use, +copy, modify, merge, publish, distribute, and/or sell copies of the Data Files +or Software, and to permit persons to whom the Data Files or Software are +furnished to do so, provided that either (a) this copyright and permission +notice appear with all copies of the Data Files or Software, or (b) this +copyright and permission notice appear in associated Documentation. + +THE DATA FILES AND SOFTWARE ARE PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT OF THIRD +PARTY RIGHTS. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR HOLDERS INCLUDED IN THIS +NOTICE BE LIABLE FOR ANY CLAIM, OR ANY SPECIAL INDIRECT OR CONSEQUENTIAL +DAMAGES, OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, +WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING +OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THE DATA FILES OR +SOFTWARE. + +Except as contained in this notice, the name of a copyright holder shall not be +used in advertising or otherwise to promote the sale, use or other dealings in +these Data Files or Software without prior written authorization of the +copyright holder. diff --git a/docs/benchmark-rollout.mdx b/docs/benchmark-rollout.mdx index bba00384..de12c71e 100644 --- a/docs/benchmark-rollout.mdx +++ b/docs/benchmark-rollout.mdx @@ -56,13 +56,43 @@ skillevaluator validate ./skills/example-skill \ --output-dir ./benchmark-backfill/example-skill ``` -Tier 3 is publication-required by default. Omitting `--agent-eval` therefore -produces an `INCOMPLETE` card with Tier 3 marked `NOT RUN`, and never a -publication recommendation. If a catalog intentionally makes Tier 3 optional, -its orchestration must persist `benchmark_policy.tier3_required = false`; the -card then discloses that policy. Do not reuse a prior live score while changing -only its label or date. If a baseline was not run, keep the explicit “uplift -unavailable” state. +Tier 1 is always publication-required. Tier 2 and Tier 3 are +publication-required by default. Omitting Tier 2 with `--no-dedup` or +`--tiers 1,3`, or omitting Tier 3 by leaving out `--agent-eval`, therefore +produces an `INCOMPLETE` card with the omitted tier marked `NOT RUN` and never a +publication recommendation. `--no-block-on-dedup` changes the command exit +gate; it is not a publication waiver. + +If a catalog intentionally makes Tier 2 or Tier 3 optional, its orchestration +must persist `benchmark_policy.tier2_required = false` or +`benchmark_policy.tier3_required = false`. The card then discloses that policy. +An optional tier that produced incomplete evidence still makes the card +`INCOMPLETE`; only absent or cleanly skipped evidence can use the waiver. +Do not reuse a prior live score while changing only its label or date. If a +baseline was not run, keep the explicit “uplift unavailable” state. + +When an orchestrator combines tiers from separate jobs, merge the actual result +objects, including each result's `publication_target` and the built-in Tier 1 +or Tier 2 `publication_evidence` producer marker, and preserve the explicit +`benchmark_policy` in the combined artifact. Every contributing result must +carry the same exact NFC filesystem-entry name and +`skill-evaluator-source-tree/2` digest. Completed Tier 3 evidence +must also preserve its run-owned `run_id` in the payload and summary. Anonymous +legacy results, a changed source tree, or inconsistent run identity make the +combined publication status `INCOMPLETE`. +Do not infer a waiver from an unselected tier or a non-blocking job. A missing +result or policy key defaults to required, so an incomplete split-tier merge +fails closed. Peer result objects have equal precedence: conflicting required +flags resolve to `true` independent of merge order. Agent-evaluation payload +metadata and its summary are duplicated higher-precedence policy claims that +must agree. Those claims are used only when the containing result is bound to +the same exact source digest; a foreign or anonymous payload cannot create a +waiver. + +Use the combined JSON report's top-level `publication_status` or +`publication.status` for publication automation. Do not substitute +`overall_status`, which records the process gate and can remain `passed` when +publication evidence is incomplete. The candidate card is: @@ -87,7 +117,7 @@ Expected presentation changes include: - removal of the `Num` column; - explicit `NOT RUN`, `SKIPPED`, and `INCOMPLETE` tier states; - collapsible methodology and non-blocking observations; -- evaluator version, dataset digest, task composition, Tier 3 requirement, +- evaluator version, dataset digest, task composition, Tier 2 and Tier 3 requirements, isolation wording, and freshness copy. Investigate any numerical change. The canonical mapping is Security=`security`, @@ -105,7 +135,7 @@ and does not override that gate. Scan all candidates before copying them into skills: ```bash -python scripts/ci/check_public_benchmarks.py \ +uv run python scripts/ci/check_public_benchmarks.py \ --require-files \ ./benchmark-backfill ``` @@ -113,7 +143,8 @@ python scripts/ci/check_public_benchmarks.py \ The linter rejects configured known leak patterns such as retired product identities, any validation-profile metadata line, common absolute home paths, ambiguous legacy uplift cells, the old `Num` column, missing metadata/decision -sections, and a publication `PASS` without completed required Tier 3 evidence. +sections, and a publication `PASS` without completed required Tier 1, Tier 2, +or Tier 3 evidence. A clean result means none of those configured patterns matched; it is a fixture regression guard, not proof that a card is safe to publish. Keep human review and the repository's broader boundary and security checks in the promotion @@ -128,7 +159,7 @@ catalog: cp ./benchmark-backfill/example-skill/BENCHMARK.md \ ./skills/example-skill/BENCHMARK.md -python scripts/ci/check_public_benchmarks.py --require-files ./skills +uv run python scripts/ci/check_public_benchmarks.py --require-files ./skills git diff --check git diff -- ./skills/example-skill/BENCHMARK.md ``` @@ -149,6 +180,8 @@ Regenerate a card when any of these inputs changes: - attempt count/pass threshold; - execution environment or isolation mode. -The generated evaluation date must come from the live run artifact. Older -artifacts without an unambiguous timestamp remain “not recorded”; never replace -that state with the report-generation date. +The generated evaluation date must come from the live run artifact as a +timezone-aware ISO timestamp. Date-only, timezone-naive, malformed, or +materially future-dated values are not publication evidence. Older artifacts +without an unambiguous timestamp remain “not recorded”; never replace that state +with the report-generation date. diff --git a/docs/ci-integration.mdx b/docs/ci-integration.mdx index 229074ac..b15b9861 100644 --- a/docs/ci-integration.mdx +++ b/docs/ci-integration.mdx @@ -194,11 +194,22 @@ When the exit code is not enough — dashboards, custom thresholds, badge genera | --- | --- | --- | | `overall_passed` | boolean | `true` when every check passed | | `overall_status` | string | `passed`, `failed`, or `incomplete` (a required scanner produced no evidence) | +| `publication_status` | string | `pass`, `neutral`, `fail`, or `incomplete`; use this field for publication eligibility | +| `publication` | object | Publication decision, eligibility, reasons, and Tier 3 evidence status | +| `benchmark_policy` | object | Persisted `tier2_required` and `tier3_required` evidence policy used for the publication decision | +| `results[].publication_target` | object | Exact NFC filesystem-entry name, versioned source-tree digest, and digest algorithm used to bind split-tier evidence | +| `results[].publication_evidence` | object | Validated built-in Tier 1 or Tier 2 producer contract: schema version, producer, tier, and canonical check ID | +| `tier3.run_id` | string | Run-owned Tier 3 identity; duplicated in the Tier 3 summary for consistency checks | | `severity_counts` | object | Totals for `critical`, `high`, `medium`, `low` | | `total_errors`, `total_warnings` | number | Aggregate counts across all validators | | `skills` | array | Per-skill `{ name, passed, issue_count }` | | `quality_summary` | array | Quality details with `overall_score` (0–100) and `grade` (A–F); a folder target reports the collection average plus a `skill_count` | +`overall_status` and `overall_passed` describe the configured command gate. They +can be green when an advisory tier did not run, so do not use them as a +publication signal. Require `publication_status == "pass"` (or +`publication.eligible == true`) before publishing a benchmark. + A step that enforces a stricter score than the built-in gate: ```bash title="Gate on quality score with jq" diff --git a/docs/environment-variables.mdx b/docs/environment-variables.mdx index 951ec1bc..f7814f33 100644 --- a/docs/environment-variables.mdx +++ b/docs/environment-variables.mdx @@ -100,7 +100,7 @@ execution environments are covered in | Variable | Default | Effect | | --- | --- | --- | -| `SKILLEVALUATOR_RESULTS_DIR` | `/evals/results` | External root for run results. Precedence for writes: the `--results-dir` flag, then this variable, then the legacy in-skill location. Read commands (`view`, `compare`) honor the same order and also fall back to the legacy location so older runs stay visible. | +| `SKILLEVALUATOR_RESULTS_DIR` | `/evals/results` | External root for run results. Precedence for writes: the `--results-dir` flag, then this variable, then the canonical in-skill location. Configured roots must remain outside the skill; `evals/results` is the only supported in-skill results root so generated output cannot change the skill's publication identity. Read commands (`view`, `compare`) honor the same order and also fall back to the legacy location so older runs stay visible. | | `SKILLEVALUATOR_LOCAL_SANDBOX` | `require` | Local-mode sandbox policy. `require` fails closed when no OS sandbox backend (Bubblewrap on Linux, Seatbelt on macOS) is usable; `prefer` degrades to advisory-only guardrails with a loud warning; `off` skips sandbox probing entirely — for skills you fully trust. No value enables native Windows: local mode fails closed there before anything runs — use WSL2 or `--env-mode docker`. | | `SKILLEVALUATOR_LOCAL_ALLOW_NET` | `true` | Network egress for local-mode trials. Set to `0` to airgap a skill that must not reach the network. Incompatible with the `nv_build` provider — NVIDIA Build local agents require network access, so airgapped runs are rejected up front. | | `SKILLEVALUATOR_LOCAL_STRICT_READS` | `false` | Tightens the sandbox's read-only view of the host system to a stricter path set. | diff --git a/docs/reports.mdx b/docs/reports.mdx index e6466eb6..cfaffdb9 100644 --- a/docs/reports.mdx +++ b/docs/reports.mdx @@ -24,7 +24,18 @@ The flag is repeatable and accepts comma- or space-separated values the terminal shows a compact summary and the run writes `html` and `json` report files. An explicit `-r` is honored exactly, including `-r cli`, which prints the full terminal report and writes no files. Files land in `reports/` -unless you override it with `-o`. +unless you override it with `-o`. Report output must be outside the source tree +whose publication identity it describes. When the default `reports/` would be +inside the target (for example, after `cd my-skill`), SkillEvaluator safely +reserves the sibling directory `my-skill-reports/` instead. A catalog uses the +same rule and relocates to `-reports/`. An explicit in-target `-o` is +rejected; choose an external directory instead. Existing nonempty sibling +directories are reused only when their SkillEvaluator ownership marker is +valid, preventing the default from overwriting an unrelated directory. + +For compatibility, commands launched from a parent directory keep the familiar +layout: `skillevaluator validate ./my-skill` still writes to `./reports/` +because that directory is outside `./my-skill/`. For `validate`, that compact default is a pipeline view — a per-check ticker plus a summary. Pass `--verbose` to get the full per-check detail stream in @@ -76,16 +87,84 @@ dataset provenance, agent/model details, and a dimension-by-dimension profile names and absolute host paths. The card carries one overall verdict — `PASS`, `NEUTRAL`, `FAIL`, or -`INCOMPLETE`. It -fails closed: if a required scanner produced no trustworthy evidence, the -verdict is `INCOMPLETE` and the card explicitly says not to use it to recommend -publication. A Publication Recommendation section appears only on a clean -`PASS`; `NEUTRAL` means the evidence is complete but at least one required -dimension remains below the pass band. Tier 3 evidence is publication-required -by default, so a missing, skipped, or incomplete live evaluation produces -`INCOMPLETE` rather than a publication recommendation. A catalog may -explicitly persist `benchmark_policy.tier3_required = false`; the card -discloses that exception instead of implying that Tier 3 ran. +`INCOMPLETE`. It fails closed: if a required scanner produced no trustworthy +evidence, the verdict is `INCOMPLETE` and the card explicitly says not to use +it to recommend publication. A Publication Recommendation section appears only +on a clean `PASS`; `NEUTRAL` means the evidence is complete but at least one +required dimension remains below the pass band. + +Tier 1 is always required for publication. Tier 2 and Tier 3 are also required +by default, so a missing, skipped, or incomplete required tier produces +`INCOMPLETE` rather than a publication recommendation. Catalog orchestration +may explicitly persist `benchmark_policy.tier2_required = false` or +`benchmark_policy.tier3_required = false`; the card discloses each exception +instead of implying that the tier ran. An optional tier that is present but +incomplete still makes the card `INCOMPLETE` because malformed evidence cannot +certify publication. A default-passed result object is not sufficient. Tier 1 +and Tier 2 evidence must carry a valid `publication_evidence` marker written by +a built-in command wrapper, and every recognized result must record a success, +finding, or positive check count that proves it executed. Legacy and custom +results remain visible, and their failures still block publication, but they +cannot satisfy a required tier or create a waiver. Generic per-result `optional` +metadata is not a publication waiver; only the resolved `benchmark_policy` from +recognized evidence can make Tier 2 or Tier 3 optional. An unrecognized `true` +claim can conservatively require a tier, but an unrecognized `false` claim +cannot waive one. + +Agent-evaluation payload metadata and its summary are duplicated claims at the +same precedence level and must agree. A conflict between valid booleans at that +level resolves to required evidence. Invalid entries do not count as policy +values; when the level contains no valid boolean, resolution falls through to +the next lower-precedence source, and no valid value anywhere means required. +The consistent payload/summary claim takes precedence over recognized peer result +metadata. Conflicting booleans among peer results likewise resolve to required +evidence so aggregation order cannot create a publication waiver. +Every Tier 1, Tier 2, and Tier 3 result used for publication carries a +`publication_target` with the exact NFC filesystem-entry name, a SHA-256 source digest, +and the versioned `skill-evaluator-source-tree/2` recipe. The recipe binds +author-owned relative paths, node kinds, executable bits, and file contents +while excluding only known generated artifacts. In particular, authored +`evals/` inputs are included. The v2 exclusions are any-depth `.git`, `.venv`, +`__pycache__`, and `node_modules`; root `.evals`, `.results`, `.versions`, +`results`, and `versions`; `evals/results`; and the generated root files +`BENCHMARK.md`, `skill-card.md`, and `skill.oms.sig`. Filesystem case aliases +are treated according to the host filesystem, while a case-distinct authored +name on a case-sensitive filesystem remains covered. A same-named file below +another authored directory is still included. Tier 1, Tier 2, Tier 3 runtime +projection, and full `--copy-repo` staging omit the same generated inputs so +bytes outside the claimed identity cannot affect evaluation evidence. Split-job +results must match this identity exactly. Legacy anonymous results remain +readable, but their publication status is `INCOMPLETE`. + +Built-in Tier 1 and Tier 2 results also carry a versioned +`publication_evidence` object with exact `schema_version`, `producer`, `tier`, +and canonical `check_id` fields. Reporters accept only the documented built-in +producer/check combinations. This marker is provenance inside trusted local or +CI artifacts; it is not a cryptographic signature for accepting result files +from an untrusted party. + +Publication model, evaluator, environment, and skill identities reject reserved +placeholder text after Unicode confusable-skeleton comparison. This prevents a +visually spoofed placeholder such as one containing a Cyrillic or Greek letter +from becoming publication evidence while preserving ordinary multilingual +identities. + +Payload and summary policy claims are accepted only when their containing +evidence matches that exact publication target; evidence from another source +cannot waive a required tier. A skipped Tier 3 payload must satisfy the same +target binding before its policy can waive evidence. A completed Tier 3 payload +also duplicates its run-owned `run_id` in the payload and summary; missing or +contradictory run identity is incomplete publication evidence. +When multiple complete Tier 3 payloads disagree, reporters select the most +conservative effective verdict (`FAIL`, then `NEUTRAL`, then `PASS`) before +applying deterministic tie-breakers. + +Execution controls do not rewrite that publication policy. `--no-dedup` and +`--tiers 1,3` select which validators run, while `--no-block-on-dedup` changes +whether Tier 2 affects the command exit code. None of those flags persists a +publication waiver. Without explicit policy metadata, a run that omits Tier 2 +therefore exits according to its CLI gate but still produces an `INCOMPLETE` +publication card. Keep the file with the skill and refresh it whenever the skill, eval dataset, agent/model, evaluator version, scoring policy, attempt policy, or execution @@ -112,7 +191,10 @@ of: With an external root (the first two options), each skill gets its own subdirectory — runs land under `//`, not directly under -``. +``. Configured roots must be outside the skill. The canonical +`/evals/results/` path is the only supported in-skill result root; this +keeps generated runs outside the versioned source-tree digest used to combine +publication evidence. Each run creates a collision-safe directory named by its run ID (`YYYYMMDD_HHMMSS__<12-hex-nonce>`). Where symlinks are supported, a @@ -316,9 +398,17 @@ For CI and tooling, two JSON entry points matter: - **`validate --tier3` runs** — the combined `skillevaluator-output-.json` report embeds the Tier 3 payload alongside the Tier 1 and Tier 2 results, so one file covers all three - tiers. Each result carries finalized gating metadata. Tier 3 is advisory by - default and becomes blocking with `--block-on-agent-eval`; Tier 2 is blocking - by default and becomes advisory with `--no-block-on-dedup`. + tiers. Each built-in Tier 1 or Tier 2 result carries its validated + `publication_evidence` producer marker alongside finalized gating metadata, + and the top-level + `benchmark_policy` records whether Tier 2 and Tier 3 evidence is required for + publication. `publication_status` is the compact machine-readable verdict; + `publication` carries eligibility, reasons, and the Tier 3 evidence status. + This is intentionally separate from `overall_status`, which continues to + describe the command/process gate. Tier 3 is advisory for the CLI exit code by default and becomes + blocking with `--block-on-agent-eval`; Tier 2 is blocking for the exit code + by default and becomes advisory with `--no-block-on-dedup`. Those exit-code + controls are separate from `benchmark_policy`. Both paths speak the same dialect: standalone `tier3 evaluate` and `validate --tier3` share one HTML renderer for `report.html`, and both embed diff --git a/docs/superpowers/specs/2026-08-24-required-tier-publication-evidence-design.md b/docs/superpowers/specs/2026-08-24-required-tier-publication-evidence-design.md new file mode 100644 index 00000000..51b0c2a6 --- /dev/null +++ b/docs/superpowers/specs/2026-08-24-required-tier-publication-evidence-design.md @@ -0,0 +1,133 @@ +# Required-Tier Publication Evidence Design + +## Problem + +`BenchmarkReporter` can recommend publication when its Tier Status table says +that Tier 1 or Tier 2 was not run. The overall verdict checks the results it was +given and has a special completeness rule for Tier 3, but it does not model the +absence of other publication-required tiers. The public benchmark linter repeats +the gap by validating only Tier 3 consistency. + +## Publication contract + +- Tier 1 is always required. A card without non-skipped Tier 1 evidence is + `INCOMPLETE`. +- Tier 2 is required by default. Only an explicitly persisted + `benchmark_policy.tier2_required = false` makes absent or cleanly skipped Tier + 2 evidence optional. +- Tier 3 keeps its current default-required policy and + `benchmark_policy.tier3_required = false` escape hatch. +- Run-selection flags such as `--no-dedup` and `--tiers 1,3` are not publication + waivers. Split-tier CI uses the same selectors, so inferring a waiver from + them would preserve the original bug. +- `--no-block-on-dedup` changes the process exit gate only. It does not make + Tier 2 optional for publication. +- Invalid policy values cannot waive required evidence. Each policy key is + resolved independently by source precedence: the duplicated agent-evaluation + payload and summary claims first, then peer result metadata. The duplicated + claims must agree when both are valid booleans; a conflict resolves to + required evidence. Invalid entries do not count as values; a level with no + valid boolean falls through to the next lower-precedence source, and no valid + value anywhere means required. Peer results share one precedence level, so + conflicting booleans resolve to required evidence independent of aggregation + order. +- Agent-evaluation policy claims are target-bound. A payload or summary for a + different skill cannot waive evidence for the current report, including when + that foreign Tier 3 result is a clean advisory skip. +- Every publication-contributing result is bound to one exact source snapshot + through a versioned `publication_target`. The digest covers normalized + author-owned paths, node kinds, executable bits, and file contents, including + authored `evals/` inputs. The versioned recipe omits only its enumerated + generated paths: any-depth `.git`, `.venv`, `__pycache__`, and `node_modules`; + root generated state/result/version directories; `evals/results`; and the + three generated root publication files. Filesystem aliases follow actual + filesystem identity, and a same-named nested authored file remains covered. + Producer scans and Tier 3 agent-visible projections share this exclusion + contract. A missing, malformed, or different digest makes + publication `INCOMPLETE` even when the visible target names match. +- Completed Tier 3 evidence carries its run-owned `run_id` in both the payload + and summary. The runner persists the target identity at the execution + boundary; reporters never backfill it from the current live path. +- Generic per-result `optional` metadata cannot waive a publication-required + validator. Only the resolved `benchmark_policy` makes Tier 2 or Tier 3 + optional; Tier 1 is always required. +- Optional evidence that is present but malformed or incomplete cannot certify + publication. + +## Components and data flow + +1. A shared publication assessment resolves `tier2_required` and + `tier3_required`, validates Tier 1/Tier 2 execution evidence, classifies Tier + 3 completeness, and keeps publication status separate from process status. + `BenchmarkReporter` renders that assessment and both requirements in + Evaluation Metadata. +2. Tier classification uses one shared validator-name classifier so the + benchmark and HTML reporters agree that names such as `Similarity Check` and + `Context Optimization` belong to Tier 2. +3. `JSONReporter` preserves the resolved publication policy plus explicit + `publication_status` and structured `publication` details so an external + aggregator does not mistake a successful process gate for publication + completeness. Per-result source identities remain available for split-job + aggregation. +4. `check_public_benchmarks.py` parses all three tier rows. For PASS cards it + requires completed Tier 1, completed Tier 2 unless explicitly optional, and + completed Tier 3 unless explicitly optional. Missing Tier 2 policy metadata + defaults to required so older contradictory cards fail closed. +5. Tier 1 and Tier 2 producers compare source snapshots before and after their + checks. Tier 3 persists its source identity and run ID in `result.json` and + propagates them through payload, summary, JSON, HTML, and BENCHMARK metadata. + A source-change conflict propagates as a bounded generic marker instead of + leaking raw path/digest details or disappearing during rerender. +6. Documentation and the changelog explain that execution selection, CLI exit + gating, and publication completeness are separate contracts. + +The downstream CI aggregator that produced the public cards is outside this +repository. This change gives that aggregator a fail-closed source/run binding; +deployment-specific artifact transport remains a separate integration proof. + +## Alternatives considered + +### Always require Tier 2 with no override + +This is simple but removes the existing pattern of persisted publication-policy +exceptions and gives catalog owners no explicit migration path. + +### Infer Tier 2 optionality from CLI selection + +This preserves current partial-run PASS cards, but it also treats a split-tier +job as publication-complete. That is the failure described by issue #73. + +### Explicit persisted required-tier policy + +This is the selected design. It separates what ran, what affects the process +exit code, and what evidence publication requires. + +## Error handling and compatibility + +- Existing programmatic Tier 1 + Tier 3 calls change from PASS to INCOMPLETE + unless they explicitly persist `tier2_required = false`. +- Existing cards without a Tier 2 policy line are interpreted as requiring Tier + 2. Good cards with completed Tier 2 remain valid; contradictory PASS cards do + not. +- Existing `tier3_required` policy behavior remains unchanged. +- Legacy results without source identity and completed Tier 3 results without a + run ID remain readable but cannot certify publication. +- A first-class CLI or YAML publication-policy option is intentionally out of + scope. Metadata injection remains the orchestration escape hatch. + +## Verification + +- Prove the reporter regression red before production changes and green after. +- Cover missing Tier 1, required and optional Tier 2, skipped/incomplete Tier 2, + invalid policy values, and Tier 2 validator-name classification. +- Cover the publication linter against required, optional, spoofed, and legacy + cards. +- Exercise `validate` with `--no-dedup`, `--tiers 1,3`, and + `--no-block-on-dedup` to show that CLI execution and exit semantics are + preserved while the publication card fails closed. +- Verify generated BENCHMARK, JSON, and HTML outputs from the same result set. +- Merge split-tier results from one unchanged source as a positive control, then + change an author-owned file between jobs and verify publication becomes + `INCOMPLETE`. +- Run the focused suites, committed-card gate, Ruff, package build, and the full + default test suite. diff --git a/docs/tier3-live-evaluation.mdx b/docs/tier3-live-evaluation.mdx index 924ab2d5..8d3d2435 100644 --- a/docs/tier3-live-evaluation.mdx +++ b/docs/tier3-live-evaluation.mdx @@ -110,11 +110,14 @@ native-task identities. The snapshot is removed when the run returns and requires temporary space only for the selected compatibility scope. Tier 3 does not install evaluator-owned directories from the target or any -nested, reference, or workspace skill into agent-visible copies. `--copy-repo` -uses the same filtering and excludes generated result roots. Authenticated -historical result roots remain excluded after the configured output location is -changed; a present marker that is copied, stale, or cannot be authenticated -fails staging closed instead of copying that tree. Custom result roots created +nested, reference, or workspace skill into agent-visible copies. Runtime skill +projection and `--copy-repo` also omit every generated path excluded by the +versioned publication-source digest. The active report directory and the whole +catalog report root are explicitly excluded from full repo context, preventing +a repeated run from exposing prior reports or Tier 3 evidence to the agent. +Authenticated historical result roots remain excluded after the configured +output location is changed; a present marker that is copied, stale, or cannot +be authenticated fails staging closed instead of copying that tree. Custom result roots created outside `evals/` by a pre-marker version cannot be identified safely after the configured location changes. Move or delete that old content before `--copy-repo` or another full-context evaluation, then rerun with the current @@ -392,7 +395,7 @@ The high-signal flags, with defaults: | `--progress` | `auto` | Progress rendering: `auto`, `rich` (live TTY view with secret redaction), `plain`, or `off` | | `--autopilot` | off | When no evaluation source exists, generate exactly one eval case — with the configured provider, falling back to a deterministic keyless template — never overwriting an existing source | | `--grading-mode` | `default` | `default`, `default_plus_custom`, or `custom_only` — see [Custom Graders & Tasks](custom-graders.mdx) | -| `--results-dir` | `evals/results`, or `SKILLEVALUATOR_RESULTS_DIR` when set | Write results under an external root instead of the skill directory | +| `--results-dir` | `evals/results`, or `SKILLEVALUATOR_RESULTS_DIR` when set | Write results under an external root instead of the skill directory; custom in-skill roots are rejected | | `--harbor-keep-jobs` | off | Retain Harbor job directories for inspection | | `--timeout-multiplier` | `1.0` | Scale Harbor step timeouts | | `--copy-repo` | off | Copy the surrounding repository into the task environment | diff --git a/scripts/ci/check_public_benchmarks.py b/scripts/ci/check_public_benchmarks.py index eda94073..463d81b7 100644 --- a/scripts/ci/check_public_benchmarks.py +++ b/scripts/ci/check_public_benchmarks.py @@ -14,24 +14,42 @@ import argparse import re import sys +import unicodedata from dataclasses import dataclass +from datetime import UTC, date, datetime, timedelta +from functools import lru_cache +from html import unescape +from html.parser import HTMLParser from pathlib import Path +from typing import TYPE_CHECKING +from urllib.parse import unquote as url_unquote + +from markdown_it import MarkdownIt + +from skillevaluator.publication_text import publication_identity_present + +if TYPE_CHECKING: + from markdown_it.token import Token REQUIRED_MARKERS = ( - "# Skill Benchmark:", - "Overall verdict:", - "## Evaluation Metadata", "- Evaluation date:", "- Evaluator version:", "- Agents:", "- Tasks:", + "- Source digest:", "- Dataset digest:", + "- Tier 3 run ID:", "- Attempts per task:", "- Environment:", "- Tier 3 evidence:", - "## Results at a Glance", - "## Tier Status", - "## Freshness", +) + +_REQUIRED_HEADINGS = ( + (1, "Skill Benchmark:", True, "# Skill Benchmark:"), + (2, "Evaluation Metadata", False, "## Evaluation Metadata"), + (2, "Results at a Glance", False, "## Results at a Glance"), + (2, "Tier Status", False, "## Tier Status"), + (2, "Freshness", False, "## Freshness"), ) LINE_RULES = ( @@ -76,15 +94,34 @@ ) _AGENT_MODEL_STATE = re.compile( - r"^[^,]+ \((?:`[^`,]+`|model not recorded)\)$", + r"^(?P[^,]+) \((?P`[^`]+`|model not recorded)\)$", flags=re.IGNORECASE, ) -_RECORDED_AGENT_MODEL_STATE = re.compile(r"^[^,]+ \(`[^`,]+`\)$") -_OVERALL_PASS = re.compile( - r"^\s*>\s*.*Overall verdict:\s*PASS\b", - flags=re.IGNORECASE | re.MULTILINE, +_OVERALL_VERDICT_FIELD = re.compile( + r"^\s*(?:(?:✅|❌|⚠\ufe0f?)\s*)?Overall verdict:\s*(?P.*)$", + flags=re.IGNORECASE, +) +_INVISIBLE_IDENTITY_CHARACTERS = frozenset( + { + "\u115f", + "\u1160", + "\u2800", + "\u3164", + "\uffa0", + "\U00013441", + "\U00013442", + "\U0001d159", + } ) +_SECURITY_CONFUSABLES = {"\u0406": "I", "\u0456": "i"} _METADATA_FIELD_RULES = ( + ( + "Source digest", + re.compile( + r"(?:`sha256:[0-9a-f]{64}`\s+\(skill-evaluator-source-tree/2\)|not recorded\b.*)", + flags=re.IGNORECASE, + ), + ), ( "Evaluation date", re.compile(r"(?:\d{4}-\d{2}-\d{2}|not recorded\b.*)", flags=re.IGNORECASE), @@ -101,6 +138,10 @@ "Dataset digest", re.compile(r"(?:`[^`\s][^`]*`(?:\s+\([^)]*\))?|not recorded\b.*)", flags=re.IGNORECASE), ), + ( + "Tier 3 run ID", + re.compile(r"(?:`[A-Za-z0-9][A-Za-z0-9._-]{0,159}`|not recorded\b.*)"), + ), ( "Attempts per task", re.compile(r"(?:[1-9]\d*|not recorded\b.*)", flags=re.IGNORECASE), @@ -111,7 +152,13 @@ ), ( "Tier 3 evidence", - re.compile(r"(?:required for publication|optional by policy)", flags=re.IGNORECASE), + re.compile(r"(?:required for publication|optional by policy)"), + ), +) +_PASS_SOURCE_METADATA_FIELD_RULES = ( + ( + "Source digest", + re.compile(r"`sha256:[0-9a-f]{64}`\s+\(skill-evaluator-source-tree/2\)"), ), ) _PASS_METADATA_FIELD_RULES = ( @@ -125,9 +172,133 @@ flags=re.IGNORECASE, ), ), + ("Tier 3 run ID", re.compile(r"`[A-Za-z0-9][A-Za-z0-9._-]{0,159}`")), ("Attempts per task", re.compile(r"[1-9]\d*")), ("Environment", re.compile(r"`[^`\s][^`]*`")), ) +_TIER_COMPLETION_STATUSES = { + 1: frozenset({"PASSED", "PASSED WITH OBSERVATIONS"}), + 2: frozenset({"PASSED", "PASSED WITH OBSERVATIONS"}), + 3: frozenset({"PASS"}), +} +_OPTIONAL_TIER_ABSENCE_STATUSES = frozenset({"NOT RUN", "SKIPPED (ADVISORY)"}) +_TIER_POLICY_VALUES = frozenset({"required for publication", "optional by policy"}) +_OVERALL_VERDICT_STATUSES = frozenset({"PASS", "FAIL", "NEUTRAL", "INCOMPLETE"}) +_KNOWN_TIER_STATUSES = { + 1: frozenset({"PASSED", "PASSED WITH OBSERVATIONS", "FAILED", "INCOMPLETE", "NOT RUN", "SKIPPED (ADVISORY)"}), + 2: frozenset({"PASSED", "PASSED WITH OBSERVATIONS", "FAILED", "INCOMPLETE", "NOT RUN", "SKIPPED (ADVISORY)"}), + 3: frozenset({"PASS", "FAIL", "FAILED", "NEUTRAL", "INCOMPLETE", "NOT RUN", "SKIPPED (ADVISORY)"}), +} +_MAX_BENCHMARK_BYTES = 128 * 1024 +_MAX_BENCHMARK_LINE_CHARACTERS = 32 * 1024 +_MARKDOWN = MarkdownIt("commonmark").enable("table") +_NON_RENDERED_HTML_TAGS = frozenset({"noscript", "script", "style", "template"}) +_VOID_HTML_TAGS = frozenset( + {"area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr"} +) + + +class _VisibleHTMLParser(HTMLParser): + """Collect rendered text while suppressing comments and control elements.""" + + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self._hidden_depth = 0 + self._line_parts: dict[int, list[str]] = {} + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + del attrs + if tag.casefold() in _NON_RENDERED_HTML_TAGS: + self._hidden_depth += 1 + + def handle_endtag(self, tag: str) -> None: + if tag.casefold() in _NON_RENDERED_HTML_TAGS and self._hidden_depth: + self._hidden_depth -= 1 + + def handle_data(self, data: str) -> None: + if self._hidden_depth: + return + start_line, _column = self.getpos() + for offset, part in enumerate(data.split("\n")): + self._line_parts.setdefault(start_line + offset, []).append(part) + + def visible_lines(self) -> tuple[tuple[int, str], ...]: + return tuple( + (line, normalized) + for line, parts in sorted(self._line_parts.items()) + if (normalized := " ".join("".join(parts).split())) + ) + + +class _HTMLAttributeParser(HTMLParser): + """Collect decoded HTML attribute values for leak-pattern scanning.""" + + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.values: list[str] = [] + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + del tag + self.values.extend(value for _name, value in attrs if value is not None) + + def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + self.handle_starttag(tag, attrs) + + +class _HTMLStructureParser(HTMLParser): + """Collect real tag events/headings without regex-parsing quoted markup.""" + + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.events: list[tuple[str, str]] = [] + self.headings: list[tuple[int, str]] = [] + self._heading_level: int | None = None + self._heading_parts: list[str] = [] + self._in_noscript = False + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + del attrs + normalized_tag = tag.casefold() + if self._in_noscript: + return + self.events.append(("start", normalized_tag)) + if normalized_tag == "noscript": + self._in_noscript = True + if normalized_tag in {"h1", "h2"} and self._heading_level is None: + self._heading_level = int(normalized_tag[1]) + self._heading_parts = [] + + def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + # HTML slash syntax does not self-close non-void elements in browsers. + self.handle_starttag(tag, attrs) + + def handle_endtag(self, tag: str) -> None: + normalized_tag = tag.casefold() + if self._in_noscript: + if normalized_tag != "noscript": + return + self._in_noscript = False + self.events.append(("end", normalized_tag)) + if self._heading_level is not None and normalized_tag == f"h{self._heading_level}": + title = " ".join("".join(self._heading_parts).split()) + self.headings.append((self._heading_level, title)) + self._heading_level = None + self._heading_parts = [] + + def handle_data(self, data: str) -> None: + if not self._in_noscript and self._heading_level is not None: + self._heading_parts.append(data) + + +@lru_cache(maxsize=256) +def _html_structure(content: str) -> tuple[tuple[tuple[str, str], ...], tuple[tuple[int, str], ...]]: + parser = _HTMLStructureParser() + try: + parser.feed(content) + parser.close() + except (AssertionError, ValueError): + return (), () + return tuple(parser.events), tuple(parser.headings) @dataclass(frozen=True) @@ -156,53 +327,133 @@ def benchmark_files(roots: list[Path]) -> list[Path]: def scan_file(path: Path) -> list[Offender]: try: + if path.stat().st_size > _MAX_BENCHMARK_BYTES: + return [Offender(path, 1, f"BENCHMARK.md exceeds {_MAX_BENCHMARK_BYTES:,} bytes")] text = path.read_text(encoding="utf-8") except (OSError, UnicodeError) as error: return [Offender(path, 1, f"unreadable file ({type(error).__name__})")] offenders: list[Offender] = [] + line_number = 1 + for character in text: + if character == "\n": + line_number += 1 + elif character not in {"\r", "\t"} and unicodedata.category(character) == "Cc": + offenders.append(Offender(path, line_number, "benchmark contains disallowed control character")) + break + + for line_number, line in enumerate(text.splitlines(), 1): + if len(line) > _MAX_BENCHMARK_LINE_CHARACTERS: + return [ + Offender( + path, + line_number, + f"benchmark line exceeds {_MAX_BENCHMARK_LINE_CHARACTERS:,} characters", + ) + ] for marker in REQUIRED_MARKERS: if marker not in text: offenders.append(Offender(path, 1, f"missing required section: {marker}")) + literal_code_lines = { + line_number + for token in _markdown_tokens(text) + if token.type in {"fence", "code_block"} and token.map is not None + for line_number in range(token.map[0] + 1, token.map[1] + 1) + } for line_number, line in enumerate(text.splitlines(), 1): + # Decode references in rendered text while preserving literal bytes in + # code spans, matching CommonMark rather than raw HTML unescaping. + rendered_surfaces = () if line_number in literal_code_lines else _rendered_inline_surfaces(line) + semantic_lines = tuple(_semantic_text(surface) for surface in (line, *rendered_surfaces)) for reason, pattern in LINE_RULES: - if pattern.search(line): + if any(pattern.search(semantic_line) for semantic_line in semantic_lines): offenders.append(Offender(path, line_number, reason)) + semantic_document = _semantic_text(text) + for reason, pattern in LINE_RULES: + if reason not in {"retired product identity", "internal environment identity"}: + continue + if pattern.search(semantic_document) and not any(offender.reason == reason for offender in offenders): + offenders.append(Offender(path, 1, reason)) + + _check_required_headings(path, text, offenders) _check_metadata_semantics(path, text, offenders) _check_verdict_tier_consistency(path, text, offenders) return offenders +def _check_required_headings(path: Path, text: str, offenders: list[Offender]) -> None: + """Require canonical headings in rendered Markdown structure.""" + headings = _heading_entries(text) + for required_level, required_title, is_prefix, marker in _REQUIRED_HEADINGS: + matching_headings = [ + (line, title, trusted) + for _index, level, line, title, trusted in headings + if level == required_level + and ( + _semantic_text(title).casefold().startswith(required_title.casefold()) + if is_prefix + else _semantic_text(title).casefold() == required_title.casefold() + ) + ] + trusted_headings = [heading for heading in matching_headings if heading[2]] + trusted_prefix_identity = ( + trusted_headings[0][1][len(required_title) :] + if trusted_headings and trusted_headings[0][1].casefold().startswith(required_title.casefold()) + else "" + ) + if not trusted_headings or (is_prefix and not _identity_present(trusted_prefix_identity)): + offenders.append(Offender(path, 1, f"missing required section: {marker}")) + if len(matching_headings) > 1 and required_title not in {"Evaluation Metadata", "Tier Status"}: + offenders.append(Offender(path, matching_headings[1][0], f"duplicate required section: {marker}")) + + def _check_metadata_semantics(path: Path, text: str, offenders: list[Offender]) -> None: - metadata_lines = _metadata_section_lines(text) - if not metadata_lines: + metadata_sections = _section_occurrences(text, "Evaluation Metadata") + if len(metadata_sections) > 1: + offenders.append(Offender(path, metadata_sections[1][0], "duplicate Evaluation Metadata section")) + if not metadata_sections: return + fallback_line, section_tokens = metadata_sections[0] + metadata_lines = _metadata_list_items(section_tokens) for field, pattern in _METADATA_FIELD_RULES: matches = _metadata_field_matches(metadata_lines, field) marker = f"- {field}:" if not matches: - offenders.append(Offender(path, metadata_lines[0][0], f"missing metadata field: {marker}")) + offenders.append(Offender(path, fallback_line, f"missing metadata field: {marker}")) continue line_number, value = matches[0] - if len(matches) > 1 or not pattern.fullmatch(value): + valid_value = pattern.fullmatch(value) is not None + if valid_value and not value.lower().startswith("not recorded"): + if field == "Evaluation date": + valid_value = _valid_evaluation_date(value) + elif field in {"Evaluator version", "Environment"}: + valid_value = _backticked_identity_present(value) + if len(matches) > 1 or not valid_value: offenders.append(Offender(path, line_number, f"invalid metadata field: {marker}")) + tier2_matches = _metadata_field_matches(metadata_lines, "Tier 2 evidence") + if tier2_matches: + line_number, value = tier2_matches[0] + if len(tier2_matches) > 1 or value not in _TIER_POLICY_VALUES: + offenders.append(Offender(path, line_number, "invalid metadata field: - Tier 2 evidence:")) + agent_matches = _metadata_field_matches(metadata_lines, "Agents") if not agent_matches: - offenders.append(Offender(path, metadata_lines[0][0], "missing metadata field: - Agents:")) + offenders.append(Offender(path, fallback_line, "missing metadata field: - Agents:")) return line_number, value = agent_matches[0] - lowered = value.lower() + lowered = value.casefold() if len(agent_matches) > 1: offenders.append(Offender(path, line_number, "agent model identity not recorded")) - elif lowered.startswith("not recorded"): + elif lowered == "not recorded (legacy or non-live result)": return - elif lowered.startswith("requested but not run"): - if "model not recorded" not in lowered: + elif lowered.startswith("requested but not run — "): + requested_states = value[len("requested but not run — ") :].strip() + if not requested_states or not _valid_unrecorded_agent_models(requested_states): offenders.append(Offender(path, line_number, "agent model identity not recorded")) elif not _valid_agent_model_states(value): offenders.append(Offender(path, line_number, "agent model identity not recorded")) @@ -210,31 +461,61 @@ def _check_metadata_semantics(path: Path, text: str, offenders: list[Offender]) def _valid_agent_model_states(value: str) -> bool: """Validate a comma-delimited agent list without a backtracking list regex.""" - agents = [agent.strip() for agent in value.split(",")] - return bool(agents) and all(_AGENT_MODEL_STATE.fullmatch(agent) for agent in agents) + agents = _split_agent_model_states(value) + if not agents: + return False + seen_agents: set[str] = set() + for agent in agents: + match = _AGENT_MODEL_STATE.fullmatch(agent) + rendered_agent = _rendered_identity(match.group("agent")) if match is not None else None + if rendered_agent is None: + return False + normalized_agent = _semantic_text(rendered_agent).casefold() + if normalized_agent in seen_agents: + return False + seen_agents.add(normalized_agent) + model = match.group("model") + if model.casefold() != "model not recorded" and not _identity_present(model[1:-1]): + return False + return True -def _metadata_section_lines(text: str) -> list[tuple[int, str]]: - """Return line-numbered content from the first Evaluation Metadata section.""" - lines = text.splitlines() - start = next( - ( - index - for index, line in enumerate(lines) - if re.fullmatch(r"\s*##\s+Evaluation Metadata\s*", line, flags=re.IGNORECASE) - ), - None, - ) - if start is None: - return [] +def _valid_unrecorded_agent_models(value: str) -> bool: + agents = _split_agent_model_states(value) + if not agents: + return False + seen_agents: set[str] = set() + for agent in agents: + match = _AGENT_MODEL_STATE.fullmatch(agent) + rendered_agent = _rendered_identity(match.group("agent")) if match is not None else None + if match is None or rendered_agent is None or match.group("model").casefold() != "model not recorded": + return False + normalized_agent = _semantic_text(rendered_agent).casefold() + if normalized_agent in seen_agents: + return False + seen_agents.add(normalized_agent) + return True - section: list[tuple[int, str]] = [] - for index in range(start + 1, len(lines)): - line = lines[index] - if re.match(r"^\s*##\s+", line): - break - section.append((index + 1, line)) - return section + +def _split_agent_model_states(value: str) -> list[str]: + """Split agent records on commas outside model code spans.""" + agents: list[str] = [] + start = 0 + in_code = False + for index, character in enumerate(value): + if character == "`": + in_code = not in_code + elif character == "," and not in_code: + agents.append(value[start:index].strip()) + start = index + 1 + agents.append(value[start:].strip()) + return agents + + +def _metadata_section_lines(text: str) -> list[tuple[int, str]]: + """Return canonical list items from the first Evaluation Metadata section.""" + sections = _section_occurrences(text, "Evaluation Metadata") + return _metadata_list_items(sections[0][1]) if sections else [] def _metadata_field_matches( @@ -254,33 +535,486 @@ def _metadata_field_value(text: str, field: str) -> str | None: return matches[0][1] if len(matches) == 1 else None -def _tier3_status(text: str) -> tuple[int, str] | None: - for line_number, line in enumerate(text.splitlines(), 1): - if not re.match(r"^\|\s*Tier\s*3\s*\|", line, flags=re.IGNORECASE): +def _overall_verdict_fields(text: str) -> list[tuple[int, str, str]]: + """Return canonical and visibly callout-shaped verdict fields.""" + tokens = _markdown_tokens(text) + fields: list[tuple[int, str, str]] = [] + + def add_field(line_number: int, line: str, *, trusted: bool) -> None: + semantic_line = _semantic_text(line) + match = _OVERALL_VERDICT_FIELD.fullmatch(semantic_line) + if match is None: + return + value = match.group("value").strip() + if value.casefold().startswith(("derived from", "pass only when every configured dimension passes")): + # This is the generated methodology definition, not a decision + # field. Exact late PASS/FAIL/NEUTRAL/INCOMPLETE fields are still + # counted so a second visible verdict cannot hide after metadata. + return + status_match = re.match(r"(?P[A-Za-z]+)\b", value) + fields.append( + ( + line_number, + status_match.group("status").upper() if trusted and status_match else "", + value, + ) + ) + + raw_container_stack: list[str] = [] + for token in tokens: + if token.type == "html_block" and token.map is not None: + for relative_line, line in _visible_html_lines(token.content): + add_field(token.map[0] + relative_line, line, trusted=False) + _update_raw_html_stack(token.content, raw_container_stack) + continue + if token.type != "inline" or token.map is None: continue - cells = [cell.strip() for cell in line.strip().strip("|").split("|")] - if len(cells) < 3: - return line_number, "" - status = re.sub(r"[*_`]", "", cells[2]).strip().upper() - return line_number, status - return None + unsafe_markup = bool(raw_container_stack) or any( + child.type in {"html_inline", "link_open", "image"} for child in token.children or [] + ) + for offset, line in enumerate(_inline_visible_lines(token, include_code=False)): + add_field(token.map[0] + offset + 1, line, trusted=not unsafe_markup) + return fields + + +def _section_occurrences(text: str, title: str) -> list[tuple[int, tuple[Token, ...]]]: + """Return structurally parsed level-two sections and their token bodies.""" + tokens = _markdown_tokens(text) + headings = _heading_entries(text) + sections: list[tuple[int, tuple[Token, ...]]] = [] + for token_index, level, start_line, heading_title, trusted in headings: + if not trusted or level != 2 or _semantic_text(heading_title).casefold() != title.casefold(): + continue + end_index = len(tokens) + # Even an untrusted linked/raw-markup heading ends the current section. + # Such a heading cannot establish a required section itself, but its + # body must not be credited to the preceding trusted section. + for next_index in range(token_index + 3, len(tokens)): + candidate = tokens[next_index] + if ( + candidate.type == "heading_open" and candidate.level == 0 and int(candidate.tag.removeprefix("h")) <= 2 + ) or ( + candidate.type == "html_block" and candidate.level == 0 and bool(_html_structure(candidate.content)[1]) + ): + end_index = next_index + break + sections.append((start_line, tokens[token_index + 3 : end_index])) + return sections + + +@lru_cache(maxsize=128) +def _markdown_tokens(text: str) -> tuple[Token, ...]: + """Parse Markdown once so structural checks use rendered block semantics.""" + return tuple(_MARKDOWN.parse(text)) + + +@lru_cache(maxsize=128) +def _heading_entries(text: str) -> tuple[tuple[int, int, int, str, bool], ...]: + """Return token index, level, line, title, and structural trust.""" + tokens = _markdown_tokens(text) + headings: list[tuple[int, int, int, str, bool]] = [] + raw_container_stack: list[str] = [] + for index, token in enumerate(tokens): + if token.type == "html_block": + if not raw_container_stack and token.map is not None: + headings.extend( + (index, level, token.map[0] + 1, title, False) for level, title in _html_structure(token.content)[1] + ) + _update_raw_html_stack(token.content, raw_container_stack) + continue + if raw_container_stack: + continue + if token.type != "heading_open" or token.level != 0 or token.map is None: + continue + inline = tokens[index + 1] if index + 1 < len(tokens) else None + if inline is None or inline.type != "inline": + continue + trusted = not any(child.type in {"html_inline", "link_open", "image"} for child in inline.children or []) + title = " ".join(_inline_visible_lines(inline, include_code=True)).strip() + headings.append((index, int(token.tag.removeprefix("h")), token.map[0] + 1, title, trusted)) + return tuple(headings) + + +def _inline_visible_lines(token: Token, *, include_code: bool) -> list[str]: + """Flatten inline Markdown into visible lines while excluding inert markup.""" + lines = [""] + hidden_html_depth = 0 + + def append_text(content: str) -> None: + parts = content.split("\n") + lines[-1] += parts[0] + lines.extend(parts[1:]) + + for child in token.children or []: + if child.type == "html_inline": + events, _headings = _html_structure(child.content) + for event, tag in events: + if tag not in _NON_RENDERED_HTML_TAGS: + continue + if event == "end": + hidden_html_depth = max(0, hidden_html_depth - 1) + elif tag not in _VOID_HTML_TAGS: + hidden_html_depth += 1 + elif hidden_html_depth: + continue + elif child.type == "text": + append_text(child.content) + elif child.type == "code_inline": + append_text(child.content if include_code else " ") + elif child.type == "image": + alt_parts: list[str] = [] + for alt_child in child.children or []: + if alt_child.type in {"text", "text_special", "code_inline"}: + alt_parts.append(alt_child.content) + elif alt_child.type in {"softbreak", "hardbreak"}: + alt_parts.append(" ") + append_text("".join(alt_parts) if child.children is not None else unescape(child.content)) + elif child.type in {"softbreak", "hardbreak"}: + lines.append("") + return [" ".join(line.split()) for line in lines] + + +def _rendered_inline_surfaces(value: str) -> tuple[str, ...]: + """Return visible text and decoded link/HTML attribute surfaces.""" + tokens = _MARKDOWN.parseInline(value) + if len(tokens) != 1 or tokens[0].type != "inline": + return () + inline = tokens[0] + surfaces = [" ".join(_inline_visible_lines(inline, include_code=True))] + for child in inline.children or []: + if child.type in {"link_open", "image"}: + for attribute in ("href", "title") if child.type == "link_open" else ("src", "title"): + if destination := child.attrGet(attribute): + surfaces.extend((destination, url_unquote(destination))) + elif child.type == "html_inline": + parser = _HTMLAttributeParser() + try: + parser.feed(child.content) + parser.close() + except (AssertionError, ValueError): + continue + for attribute_value in parser.values: + surfaces.extend((attribute_value, url_unquote(attribute_value))) + return tuple(surfaces) + + +def _rendered_inline_fragment(value: str) -> str | None: + """Return visible inline text while keeping code-span contents literal.""" + surfaces = _rendered_inline_surfaces(value) + return surfaces[0] if surfaces else None + + +def _rendered_identity(value: str) -> str | None: + tokens = _MARKDOWN.parseInline(value) + if len(tokens) != 1 or tokens[0].type != "inline": + return None + inline = tokens[0] + if any(child.type in {"html_inline", "link_open", "image", "code_inline"} for child in inline.children or []): + return None + rendered = " ".join(_inline_visible_lines(inline, include_code=False)) + return rendered if _identity_present(rendered) else None + + +def _rendered_identity_present(value: str) -> bool: + return _rendered_identity(value) is not None + + +def _update_raw_html_stack(content: str, stack: list[str]) -> None: + """Track Markdown nested inside raw HTML containers across block tokens.""" + events, _headings = _html_structure(content) + for event, tag in events: + if "noscript" in stack: + if event == "end" and stack[-1] == tag == "noscript": + stack.pop() + continue + if event == "end": + # Python's generic HTMLParser does not implement browser tree- + # builder insertion modes. Trust only properly nested closures; + # a mismatched close must not expose evidence that a browser keeps + # inside an outer container. + if stack and stack[-1] == tag: + stack.pop() + continue + # In text/html, a trailing slash does not self-close non-void elements + # such as
or
. Browsers keep those containers open. + if tag in _VOID_HTML_TAGS: + continue + stack.append(tag) + + +@lru_cache(maxsize=256) +def _visible_html_lines(content: str) -> tuple[tuple[int, str], ...]: + """Return line-relative text that a raw HTML block would visibly render.""" + parser = _VisibleHTMLParser() + try: + parser.feed(content) + parser.close() + except (AssertionError, ValueError): + # Malformed raw HTML is not trustworthy visible evidence. + return () + return parser.visible_lines() + + +def _metadata_list_items(section_tokens: tuple[Token, ...]) -> list[tuple[int, str]]: + """Return one-line items from top-level unordered metadata lists.""" + items: list[tuple[int, str]] = [] + in_root_bullet_list = False + raw_container_stack: list[str] = [] + for index, token in enumerate(section_tokens): + if token.type == "html_block": + _update_raw_html_stack(token.content, raw_container_stack) + continue + if raw_container_stack: + continue + if token.type == "bullet_list_open" and token.level == 0: + in_root_bullet_list = True + continue + if token.type == "bullet_list_close" and token.level == 0: + in_root_bullet_list = False + continue + if not in_root_bullet_list or token.type != "list_item_open" or token.level != 1: + continue + + item_end = next( + ( + candidate + for candidate in range(index + 1, len(section_tokens)) + if section_tokens[candidate].type == "list_item_close" + and section_tokens[candidate].level == token.level + ), + len(section_tokens), + ) + inline_tokens = [ + candidate + for candidate in section_tokens[index + 1 : item_end] + if candidate.type == "inline" and candidate.level == token.level + 2 and candidate.map is not None + ] + if len(inline_tokens) != 1: + continue + inline = inline_tokens[0] + if inline.map is None or inline.map[1] != inline.map[0] + 1: + continue + if any(child.type in {"html_inline", "link_open", "image"} for child in inline.children or []): + continue + # Preserve encoded commas because literal commas delimit agents. Each + # captured identity is rendered and decoded before it counts as proof. + items.append((inline.map[0] + 1, f"- {inline.content}")) + return items + + +def _tier_status_rows(section_tokens: tuple[Token, ...], tier: int) -> list[tuple[int, str]]: + """Return Tier rows from rendered root-level Markdown tables.""" + rows: list[tuple[int, str]] = [] + in_root_table = False + in_header = False + in_body = False + row_line = 1 + cells: list[str] | None = None + cell_parts: list[str] | None = None + tier_column: int | None = None + status_column: int | None = None + cell_has_unsafe_markup = False + cell_safety: list[bool] | None = None + raw_container_stack: list[str] = [] + + for token in section_tokens: + if token.type == "html_block": + _update_raw_html_stack(token.content, raw_container_stack) + continue + if raw_container_stack: + continue + if token.type == "table_open" and token.level == 0: + in_root_table = True + tier_column = None + status_column = None + continue + if token.type == "table_close" and token.level == 0: + in_root_table = False + in_header = False + in_body = False + continue + if not in_root_table: + continue + if token.type == "thead_open": + in_header = True + continue + if token.type == "thead_close": + in_header = False + continue + if token.type == "tbody_open": + in_body = True + continue + if token.type == "tbody_close": + in_body = False + continue + if not (in_header or in_body): + continue + if token.type == "tr_open": + row_line = token.map[0] + 1 if token.map is not None else 1 + cells = [] + cell_safety = [] + continue + if token.type in {"th_open", "td_open"} and cells is not None: + cell_parts = [] + cell_has_unsafe_markup = False + continue + if token.type == "inline" and cell_parts is not None: + cell_parts.extend(_inline_visible_lines(token, include_code=True)) + cell_has_unsafe_markup = cell_has_unsafe_markup or any( + child.type in {"html_inline", "link_open", "image"} for child in token.children or [] + ) + continue + if token.type in {"th_close", "td_close"} and cells is not None and cell_parts is not None: + cells.append(" ".join(cell_parts).strip()) + assert cell_safety is not None + cell_safety.append(not cell_has_unsafe_markup) + cell_parts = None + continue + if token.type != "tr_close" or cells is None: + continue + + if in_header: + normalized_headers = [" ".join(cell.split()).casefold() for cell in cells] + tier_indexes = [index for index, header in enumerate(normalized_headers) if header == "tier"] + status_indexes = [index for index, header in enumerate(normalized_headers) if header == "status"] + if len(tier_indexes) == len(status_indexes) == 1: + tier_column = tier_indexes[0] + status_column = status_indexes[0] + elif in_body and tier_column is not None and status_column is not None: + if max(tier_column, status_column) < len(cells): + tier_label = " ".join(cells[tier_column].split()) + if re.fullmatch(rf"Tier\s*{tier}", tier_label, flags=re.IGNORECASE): + assert cell_safety is not None + status = ( + " ".join(cells[status_column].split()).upper() + if cell_safety[tier_column] and cell_safety[status_column] + else "" + ) + rows.append((row_line, status)) + cells = None + cell_safety = None + cell_parts = None + return rows + + +def _has_publication_recommendation(text: str) -> bool: + """Return whether rendered content recommends publication.""" + needle = "recommended for publication" + for token in _markdown_tokens(text): + if token.type == "inline": + visible = " ".join(_inline_visible_lines(token, include_code=False)) + if needle in _semantic_text(visible).casefold(): + return True + elif token.type == "html_block": + visible = " ".join(line for _offset, line in _visible_html_lines(token.content)) + if needle in _semantic_text(visible).casefold(): + return True + return False + + +def _semantic_text(value: str) -> str: + """Normalize compatibility characters and remove invisible spoofing characters.""" + normalized = unicodedata.normalize("NFKD", value) + + def canonical_character(character: str) -> str: + if unicodedata.category(character) == "Pd" or character in {"\u2043", "\u2212"}: + return "-" + if character in {"\u2044", "\u2215", "\u29f8"}: + return "/" + if character in {"\u2216", "\u29f5"}: + return "\\" + return _SECURITY_CONFUSABLES.get(character, character) + + return "".join( + canonical_character(character) + for character in normalized + if unicodedata.category(character)[0] not in {"C", "M"} and character not in _INVISIBLE_IDENTITY_CHARACTERS + ) + + +def _identity_present(value: str) -> bool: + """Return whether public identity text contains recorded alphanumeric provenance.""" + return publication_identity_present(value) + + +def _valid_evaluation_date(value: str) -> bool: + """Return whether an ISO date exists on the calendar and is not future-dated.""" + try: + evaluated_on = date.fromisoformat(value) + except ValueError: + return False + return evaluated_on <= (datetime.now(UTC) + timedelta(minutes=5)).date() + + +def _backticked_identity_present(value: str) -> bool: + match = re.fullmatch(r"`(?P[^`]*)`", value) + return match is not None and _identity_present(match.group("identity")) def _check_verdict_tier_consistency(path: Path, text: str, offenders: list[Offender]) -> None: - tier3_row = _tier3_status(text) - if tier3_row is None: - offenders.append(Offender(path, 1, "missing Tier 3 status row")) + tier_rows: dict[int, tuple[int, str]] = {} + tier_sections = _section_occurrences(text, "Tier Status") + if len(tier_sections) > 1: + offenders.append(Offender(path, tier_sections[1][0], "duplicate Tier Status section")) + elif tier_sections: + section_tokens = tier_sections[0][1] + for tier in _TIER_COMPLETION_STATUSES: + rows = _tier_status_rows(section_tokens, tier) + if not rows: + offenders.append(Offender(path, 1, f"missing Tier {tier} status row")) + elif len(rows) > 1: + offenders.append(Offender(path, rows[1][0], f"duplicate Tier {tier} status row")) + else: + tier_rows[tier] = rows[0] + else: + for tier in _TIER_COMPLETION_STATUSES: + offenders.append(Offender(path, 1, f"missing Tier {tier} status row")) + + for tier, (line_number, status) in tier_rows.items(): + if status not in _KNOWN_TIER_STATUSES[tier]: + offenders.append(Offender(path, line_number, f"invalid Tier {tier} status")) + + verdict_fields = _overall_verdict_fields(text) + if not verdict_fields: + offenders.append(Offender(path, 1, "missing Overall verdict field")) + return + if len(verdict_fields) > 1: + offenders.append(Offender(path, verdict_fields[1][0], "duplicate Overall verdict field")) + return + verdict_line, verdict_status, _verdict_value = verdict_fields[0] + if verdict_status not in _OVERALL_VERDICT_STATUSES: + offenders.append(Offender(path, verdict_line, "invalid Overall verdict field")) return - if not _OVERALL_PASS.search(text): + recommendation_sections = _section_occurrences(text, "Publication Recommendation") + if len(recommendation_sections) > 1: + offenders.append(Offender(path, recommendation_sections[1][0], "duplicate Publication Recommendation section")) + if verdict_status != "PASS": + if _has_publication_recommendation(text) or recommendation_sections: + recommendation_line = recommendation_sections[0][0] if recommendation_sections else verdict_line + offenders.append(Offender(path, recommendation_line, "non-PASS verdict recommends publication")) return - line_number, tier3_status = tier3_row - tier3_complete = tier3_status == "PASS" - tier3_optional = (_metadata_field_value(text, "Tier 3 evidence") or "").lower() == "optional by policy" - if tier3_complete: - _check_pass_provenance(path, text, offenders) - if not tier3_complete and not tier3_optional: - offenders.append(Offender(path, line_number, "publication PASS without completed Tier 3 evidence")) + _check_pass_source_identity(path, text, offenders) + for tier, row in tier_rows.items(): + line_number, status = row + complete = status in _TIER_COMPLETION_STATUSES[tier] + optional = tier in {2, 3} and _metadata_field_value(text, f"Tier {tier} evidence") == "optional by policy" + + if tier == 3 and complete: + _check_pass_provenance(path, text, offenders) + if not complete and not (optional and status in _OPTIONAL_TIER_ABSENCE_STATUSES): + offenders.append(Offender(path, line_number, f"publication PASS without completed Tier {tier} evidence")) + + +def _check_pass_source_identity(path: Path, text: str, offenders: list[Offender]) -> None: + """Require every PASS card to identify the exact evaluated source tree.""" + metadata_lines = _metadata_section_lines(text) + fallback_line = metadata_lines[0][0] if metadata_lines else 1 + for field, pattern in _PASS_SOURCE_METADATA_FIELD_RULES: + matches = _metadata_field_matches(metadata_lines, field) + line_number = matches[0][0] if matches else fallback_line + if len(matches) != 1 or pattern.fullmatch(matches[0][1]) is None: + offenders.append(Offender(path, line_number, f"publication PASS without recorded {field.lower()}")) def _check_pass_provenance(path: Path, text: str, offenders: list[Offender]) -> None: @@ -290,23 +1024,41 @@ def _check_pass_provenance(path: Path, text: str, offenders: list[Offender]) -> for field, pattern in _PASS_METADATA_FIELD_RULES: matches = _metadata_field_matches(metadata_lines, field) line_number = matches[0][0] if matches else fallback_line - if len(matches) != 1 or not pattern.fullmatch(matches[0][1]): + valid_value = len(matches) == 1 and pattern.fullmatch(matches[0][1]) is not None + if valid_value: + if field == "Evaluation date": + valid_value = _valid_evaluation_date(matches[0][1]) + elif field in {"Evaluator version", "Environment"}: + valid_value = _backticked_identity_present(matches[0][1]) + if not valid_value: offenders.append(Offender(path, line_number, f"publication PASS without recorded {field.lower()}")) agent_matches = _metadata_field_matches(metadata_lines, "Agents") line_number = agent_matches[0][0] if agent_matches else fallback_line - if len(agent_matches) != 1: - return - value = agent_matches[0][1] - if (value.lower().startswith("not recorded") or _valid_agent_model_states(value)) and not ( - _valid_recorded_agent_models(value) - ): + if len(agent_matches) != 1 or not _valid_recorded_agent_models(agent_matches[0][1]): offenders.append(Offender(path, line_number, "publication PASS without recorded agent model identity")) def _valid_recorded_agent_models(value: str) -> bool: - agents = [agent.strip() for agent in value.split(",")] - return bool(agents) and all(_RECORDED_AGENT_MODEL_STATE.fullmatch(agent) for agent in agents) + agents = _split_agent_model_states(value) + if not agents: + return False + seen_agents: set[str] = set() + for agent in agents: + match = _AGENT_MODEL_STATE.fullmatch(agent) + rendered_agent = _rendered_identity(match.group("agent")) if match is not None else None + if ( + match is None + or rendered_agent is None + or not match.group("model").startswith("`") + or not _identity_present(match.group("model")[1:-1]) + ): + return False + normalized_agent = _semantic_text(rendered_agent).casefold() + if normalized_agent in seen_agents: + return False + seen_agents.add(normalized_agent) + return True def find_offenders(roots: list[Path]) -> tuple[list[Path], list[Offender]]: diff --git a/src/skillevaluator/cli.py b/src/skillevaluator/cli.py index 0ed29141..ced0fed7 100644 --- a/src/skillevaluator/cli.py +++ b/src/skillevaluator/cli.py @@ -324,6 +324,81 @@ def _report_options(func): )(func) +def _path_is_within(candidate: Path, root: Path) -> bool: + """Check containment using both path text and existing filesystem aliases.""" + for path, path_root in ( + (candidate.absolute(), root.absolute()), + (candidate.resolve(strict=False), root.resolve(strict=True)), + ): + try: + path.relative_to(path_root) + return True + except ValueError: + pass + + current = candidate.absolute() + while True: + try: + current.lstat() + break + except FileNotFoundError: + parent = current.parent + if parent == current: + return False + current = parent + except OSError: + return False + + while True: + try: + if current.samefile(root): + return True + except (OSError, ValueError): + return False + parent = current.parent + if parent == current: + return False + current = parent + + +def _resolve_report_output_location(target_path: Path, output_dir: Path) -> Path: + """Keep report writes outside the source tree whose identity they describe.""" + if not target_path.is_dir(): + return output_dir + try: + lexical_target = target_path.absolute() + resolved_target = target_path.resolve(strict=True) + lexical_output = output_dir.absolute() + resolved_output = output_dir.resolve(strict=False) + except (OSError, RuntimeError) as exc: + raise click.ClickException(f"Cannot resolve report output directory: {output_dir}") from exc + + output_is_in_target = _path_is_within(lexical_output, lexical_target) or _path_is_within( + resolved_output, + resolved_target, + ) + if not output_is_in_target: + return output_dir + + from click.core import ParameterSource + + context = click.get_current_context(silent=True) + parameter_source = context.get_parameter_source("output_dir") if context is not None else None + if parameter_source is ParameterSource.DEFAULT: + relocated = target_path.with_name(f"{target_path.name}-reports") + try: + if not _path_is_within(relocated, resolved_target): + from skillevaluator.tier3.output_provenance import mark_generated_output_root + + mark_generated_output_root(relocated) + return relocated + except (OSError, RuntimeError, ValueError) as exc: + raise click.ClickException(f"Cannot safely reserve default report output directory: {relocated}") from exc + raise click.ClickException( + f"Report output must be outside the publication target; use an external --output-dir: {output_dir}" + ) + + def _report_formats_explicit() -> bool: """True when the user passed ``-r``/``--report`` on the command line. @@ -351,12 +426,15 @@ def _run_dedup_or_skip(target_path: Path) -> list[ValidationResult]: import importlib.util def _skip(message: str) -> list[ValidationResult]: + from skillevaluator.publication_identity import stamp_publication_target + result = ValidationResult( validator_name="Tier 2 Deduplication", validator_description="Embedding-based duplicate detection", ) result.add_warning(message) result.metadata["skipped"] = True + stamp_publication_target([result], target_path) return [result] def _available(module: str) -> bool: @@ -453,6 +531,7 @@ def _run_agent_eval_or_skip( harbor_keep_jobs: bool = False, block_on_agent_eval: bool = False, validate_source: bool = True, + repo_context_exclude_paths: tuple[Path, ...] = (), progress_reporter=None, ) -> ValidationResult: """Run Tier 3 live agent evaluation and fold the result into the combined report. @@ -501,6 +580,7 @@ def _run_agent_eval_or_skip( copy_repo=copy_repo, timeout_multiplier=timeout_multiplier, harbor_keep_jobs=harbor_keep_jobs, + repo_context_exclude_paths=repo_context_exclude_paths, ) try: service = EvaluationService() @@ -711,20 +791,29 @@ def _validate_catalog( skill_dirs = sorted(marker.parent for marker in resolved_target.glob("*/SKILL.md")) failures: list[tuple[str, str]] = [] - for index, skill_dir in enumerate(skill_dirs, start=1): - _print_catalog_divider(index, len(skill_dirs), skill_dir.name) - overrides = { - **ctx.params, - "target_path": skill_dir, - "content_type": "skill", - "output_dir": output_dir / skill_dir.name, - } - try: - ctx.invoke(validate, **overrides) - except click.ClickException as exc: - failures.append((skill_dir.name, str(getattr(exc, "message", exc)))) - except Exception as exc: # unexpected: keep the catalog running, report it on the scoreboard - failures.append((skill_dir.name, f"unexpected error: {exc}")) + meta_key = "skillevaluator_catalog_report_root" + previous_catalog_root = ctx.meta.get(meta_key) + ctx.meta[meta_key] = output_dir + try: + for index, skill_dir in enumerate(skill_dirs, start=1): + _print_catalog_divider(index, len(skill_dirs), skill_dir.name) + overrides = { + **ctx.params, + "target_path": skill_dir, + "content_type": "skill", + "output_dir": output_dir / skill_dir.name, + } + try: + ctx.invoke(validate, **overrides) + except click.ClickException as exc: + failures.append((skill_dir.name, str(getattr(exc, "message", exc)))) + except Exception as exc: # unexpected: keep the catalog running, report it on the scoreboard + failures.append((skill_dir.name, f"unexpected error: {exc}")) + finally: + if previous_catalog_root is None: + ctx.meta.pop(meta_key, None) + else: + ctx.meta[meta_key] = previous_catalog_root _print_catalog_summary(len(skill_dirs), failures, output_dir) if failures: raise click.ClickException( @@ -1218,6 +1307,8 @@ def validate( from skillevaluator.constants import CONTENT_TYPE_UNKNOWN + output_dir = _resolve_report_output_location(target_path, output_dir) + # A directory of skills (no root SKILL.md) is a catalog: run the pipeline # once per skill, serially, each as its own job with its own reports. if ( @@ -1261,6 +1352,20 @@ def validate( _print_run_banner(target_path, resolved_type, getattr(policy, "profile", None)) _print_tier_banner(_TIER_BANNERS["tier1"]) + # Autopilot source generation must precede every tier's publication-target + # capture so one combined report cannot bind otherwise valid tier evidence + # to different source trees. Generation remains advisory: retain any error + # for the later Tier 3 skip result and continue through Tier 1/2. + autopilot_error: str | None = None + autopilot_dataset_note: str | None = None + if autopilot: + try: + autopilot_dataset_note = _ensure_autopilot_dataset(target_path, quiet=quiet) + except (Exception, SystemExit) as exc: + autopilot_error = f"autopilot dataset generation failed: {getattr(exc, 'message', exc)}" + if not quiet: + click.echo(f"Warning: {autopilot_error}", err=True) + view.start() view.tier_start(0) check_lineup = enabled_check_lineup(checks) @@ -1347,20 +1452,8 @@ def _on_check(name: str) -> None: detail_row("model", model_display), ] - # Autopilot: reuse the standalone evaluate command's dataset flow. - # Tier 3 is advisory, so a dataset-generation failure must not abort - # validate after Tier 1/2 already ran -- Tier 3 skips with the reason. - autopilot_error: str | None = None - if autopilot: - try: - dataset_note = _ensure_autopilot_dataset(target_path, quiet=quiet) - except (Exception, SystemExit) as exc: - autopilot_error = f"autopilot dataset generation failed: {getattr(exc, 'message', exc)}" - if not quiet: - click.echo(f"Warning: {autopilot_error}", err=True) - else: - if dataset_note: - tier3_config_rows.append(detail_row("dataset", dataset_note)) + if autopilot_dataset_note: + tier3_config_rows.append(detail_row("dataset", autopilot_dataset_note)) view.tier_progress( tier3_index, @@ -1371,6 +1464,13 @@ def _on_engine_tail(lines: list[str]) -> None: view.tier_progress(tier3_index, [*tier3_config_rows, *engine_feed_rows(lines)]) reporter = ViewProgressReporter(_on_engine_tail) if quiet else None + repo_context_exclude_paths = [output_dir] + catalog_report_root = click.get_current_context().meta.get("skillevaluator_catalog_report_root") + if isinstance(catalog_report_root, Path) and not paths_refer_to_same_location( + catalog_report_root, + output_dir, + ): + repo_context_exclude_paths.append(catalog_report_root) tier3_result = _run_agent_eval_or_skip( target_path, agents=agents, @@ -1392,6 +1492,7 @@ def _on_engine_tail(lines: list[str]) -> None: block_on_agent_eval=block_on_agent_eval_effective, validate_source=preflight_tier3_source, progress_reporter=reporter, + repo_context_exclude_paths=tuple(repo_context_exclude_paths), ) results.append(tier3_result) tier3_ran, tier3_ok, tier3_rows, tier3_skip = summarize_tier3(tier3_result) @@ -1460,6 +1561,7 @@ def _on_engine_tail(lines: list[str]) -> None: basename=report_basename_value, policy=policy, target_path=target_display, + expected_skill_name=target_path.name, content_label=content_label, announce_paths=not quiet, ) @@ -1481,7 +1583,9 @@ def _on_engine_tail(lines: list[str]) -> None: ) if tier3_result is not None and block_on_agent_eval_effective: effective_gate_results.append(tier3_result) - gate_failed = not all(result.passed for result in effective_gate_results) + from skillevaluator.reporting.base import passes_required_gate + + gate_failed = not all(passes_required_gate(result) for result in effective_gate_results) if quiet: _finish_pipeline_view( view, @@ -1513,6 +1617,7 @@ def _on_engine_tail(lines: list[str]) -> None: @_report_options def quality_check(target_path: Path, min_score: int, report_formats: tuple[str, ...], output_dir: Path) -> None: """Score skill quality across correctness, discoverability, reliability, and efficiency.""" + output_dir = _resolve_report_output_location(target_path.resolve(), output_dir) if not emit_reports( run_quality_check(target_path, min_score=min_score), report_formats=report_formats, @@ -1528,6 +1633,7 @@ def quality_check(target_path: Path, min_score: int, report_formats: tuple[str, @_report_options def rubric_eval(target_path: Path, min_score: int, report_formats: tuple[str, ...], output_dir: Path) -> None: """Run LLM-as-judge rubric evaluation for a skill.""" + output_dir = _resolve_report_output_location(target_path.resolve(), output_dir) if not emit_reports( run_rubric_eval(target_path, min_score=min_score), report_formats=report_formats, @@ -1546,6 +1652,7 @@ def security_scan( target_path: Path, llm: bool, llm_verify: bool, report_formats: tuple[str, ...], output_dir: Path ) -> None: """Scan for security vulnerabilities.""" + output_dir = _resolve_report_output_location(target_path.resolve(), output_dir) if not emit_reports( run_security_scan(target_path, use_llm=llm, llm_verify=llm_verify), report_formats=report_formats, @@ -1561,6 +1668,7 @@ def security_scan( @_report_options def pii_scan(target_path: Path, llm_verify: bool, report_formats: tuple[str, ...], output_dir: Path) -> None: """Scan for PII and local identifiers.""" + output_dir = _resolve_report_output_location(target_path.resolve(), output_dir) if not emit_reports( run_pii_scan(target_path, llm_verify=llm_verify), report_formats=report_formats, @@ -1575,6 +1683,7 @@ def pii_scan(target_path: Path, llm_verify: bool, report_formats: tuple[str, ... @_report_options def lint_scripts(target_path: Path, report_formats: tuple[str, ...], output_dir: Path) -> None: """Run advisory lint checks on skill scripts.""" + output_dir = _resolve_report_output_location(target_path.resolve(), output_dir) if not emit_reports( run_lint_scripts(target_path), report_formats=report_formats, @@ -1631,6 +1740,7 @@ def similarity_check( raise click.UsageError("--catalog and --save-catalog cannot be used together") _reject_linked_tier2_root(content_path) + output_dir = _resolve_report_output_location(content_path.resolve(), output_dir) similarity_basename = report_basename("similarity") _reject_catalog_report_collisions( resolved_catalog, @@ -1683,6 +1793,7 @@ def context_optimization_check( from skillevaluator.tier2.commands import run_context_optimization_check _reject_linked_tier2_root(skill_path) + output_dir = _resolve_report_output_location(skill_path.resolve(), output_dir) results = run_context_optimization_check(skill_path, threshold=threshold, model=model, llm_model=llm_model) sanitize_tier2_results(results, skill_path) if not emit_reports( @@ -1712,6 +1823,7 @@ def dedup_scan( from skillevaluator.tier2.commands import run_dedup_scan _reject_linked_tier2_root(skill_path) + output_dir = _resolve_report_output_location(skill_path.resolve(), output_dir) results = run_dedup_scan( skill_path, threshold=threshold, diff --git a/src/skillevaluator/constants.py b/src/skillevaluator/constants.py index 5926ba41..49c820b4 100644 --- a/src/skillevaluator/constants.py +++ b/src/skillevaluator/constants.py @@ -173,7 +173,7 @@ # Generated publishing/signing artifacts that may live in a skill root after # NVSkills CI runs. They are derived from the author-owned skill content, so # Tier 1 should not scan them as independent source files. -SCAN_EXCLUDED_FILES = frozenset({"skill-card.md", "benchmark.md", "skill.oms.sig"}) +SCAN_EXCLUDED_FILES = frozenset({"skill-card.md", "BENCHMARK.md", "skill.oms.sig"}) # Environment variables consulted (in order) to identify who is submitting a # skill, for the home-path PII check. A ``/home//`` path is only flagged @@ -289,12 +289,25 @@ # snapshots mirror the live skill and would dominate reports with self-matches. # Tier 1 has its own (broader) :data:`SCAN_EXCLUDED_DIRS`; keep these two in # sync for the artifact-dir entries. -CONTENT_DEDUP_EXCLUDED_DIRS = frozenset({"evals", ".evals", "results", ".results", "versions", ".versions"}) +CONTENT_DEDUP_EXCLUDED_DIRS = frozenset( + { + ".git", + ".venv", + "__pycache__", + "node_modules", + "evals", + ".evals", + "results", + ".results", + "versions", + ".versions", + } +) # Skipped by basename by Tier 2 dedup: generated reports/metadata are not # author-owned context, so comparing them against SKILL.md produces structural # self-matches. -CONTENT_DEDUP_EXCLUDED_FILES = frozenset({"skill-card.md", "benchmark.md", "skill.oms.sig"}) +CONTENT_DEDUP_EXCLUDED_FILES = frozenset({"skill-card.md", "BENCHMARK.md", "skill.oms.sig"}) # ============================================================================= diff --git a/src/skillevaluator/deduplication/utils/skill_collector.py b/src/skillevaluator/deduplication/utils/skill_collector.py index 56c6cde4..f5698d09 100644 --- a/src/skillevaluator/deduplication/utils/skill_collector.py +++ b/src/skillevaluator/deduplication/utils/skill_collector.py @@ -28,6 +28,7 @@ CONTENT_DEDUP_MAX_TOTAL_BYTES, CONTENT_DEDUP_SCANNABLE_EXTENSIONS, ) +from skillevaluator.utils.path_security import matches_filesystem_name from skillevaluator.utils.tier2_paths import is_contained_compatibility_alias from skillevaluator.validators.frontmatter_parser import FRONTMATTER_PATTERN @@ -454,7 +455,7 @@ def _iter_paths(root: Path) -> Iterator[Path]: for name in dirnames: directory = base / name rel_path = directory.relative_to(root).as_posix() - if name in excluded: + if matches_filesystem_name(directory, excluded): logger.debug("Skipping excluded path: %s", rel_path) continue # ``os.walk(..., followlinks=False)`` does not follow normal @@ -555,7 +556,7 @@ def _iter_paths(root: Path) -> Iterator[Path]: if ext not in CONTENT_DEDUP_SCANNABLE_EXTENSIONS: continue - if file_path.name.lower() in excluded_basenames: + if matches_filesystem_name(file_path, excluded_basenames): logger.debug("Skipping excluded file: %s", file_path) continue diff --git a/src/skillevaluator/evaluation/options.py b/src/skillevaluator/evaluation/options.py index a131944c..e0b5004a 100644 --- a/src/skillevaluator/evaluation/options.py +++ b/src/skillevaluator/evaluation/options.py @@ -46,6 +46,7 @@ class EvaluationOptions: override_cpus: int | None = None override_memory_mb: int | None = None override_storage_mb: int | None = None + repo_context_exclude_paths: tuple[Path, ...] = () def engine_kwargs(self) -> dict[str, Any]: """Return keyword arguments (excluding ``skill_path``) for the engine.""" diff --git a/src/skillevaluator/evaluation/tier3_report.py b/src/skillevaluator/evaluation/tier3_report.py index 0dc50134..584d6bce 100644 --- a/src/skillevaluator/evaluation/tier3_report.py +++ b/src/skillevaluator/evaluation/tier3_report.py @@ -47,6 +47,7 @@ _AGENT_EVAL_VALIDATOR = "AGENT_EVAL" _AGENT_EVAL_DESCRIPTION = "Tier 3: Live Agent Evaluation (Harbor)" +_PUBLICATION_TARGET_CONFLICT_MARKER = "source changed during evaluation" _DIMENSION_IDS = list(DIMENSION_MAPPING.keys()) @@ -441,8 +442,13 @@ def agent_eval_result_from_directory( run_truth = _run_truth_metadata(run_dir, engine_result, load_dataset_snapshot(run_dir)) dataset = run_truth.get("dataset") or load_staged_harbor_dataset(run_dir) + publication_target = run_truth.get("publication_target") + persisted_skill_name = publication_target.get("skill_name") if isinstance(publication_target, dict) else None + skill_name = ( + persisted_skill_name if isinstance(persisted_skill_name, str) and persisted_skill_name else skill_path.name + ) payload = build_agent_eval_payload( - skill_path.name, + skill_name, agents, dataset=dataset, attempt_policy=_read_attempt_policy(run_dir), @@ -458,6 +464,9 @@ def agent_eval_result_from_directory( persisted_dataset_summary=run_truth.get("dataset_summary"), dataset_digest=run_truth.get("dataset_digest"), dataset_digest_algorithm=run_truth.get("dataset_digest_algorithm"), + run_id=run_truth.get("run_id"), + publication_target=publication_target, + publication_target_conflict=run_truth.get("publication_target_conflict"), use_llm_judge=use_llm_judge, ) return _validation_result_from_payload(payload) @@ -473,6 +482,15 @@ def _validation_result_from_payload(payload: dict[str, Any] | None) -> Validatio validator_description=_AGENT_EVAL_DESCRIPTION, ) result.metadata["agent_eval"] = payload + publication_target = payload.get("publication_target") + if isinstance(publication_target, dict): + result.metadata["publication_target"] = dict(publication_target) + publication_target_conflict = payload.get("publication_target_conflict") + if isinstance(publication_target_conflict, str): + result.metadata["publication_target_conflict"] = publication_target_conflict + run_id = payload.get("run_id") + if isinstance(run_id, str): + result.metadata["run_id"] = run_id best = payload.get("best_agent") or "n/a" if payload.get("execution_status") == "succeeded" and _finite_float(payload.get("overall_score")) is not None: result.add_success( @@ -526,6 +544,7 @@ def render_agent_eval_html_report( target_path=str(skill_path), content_label="Skill", tabs=[{"id": "tier3", "label": "Tier 3: Live Agent Evaluation"}], + expected_skill_name=skill_path.name, ) reporter.save([result], target) return target @@ -549,6 +568,9 @@ def build_agent_eval_payload( persisted_dataset_summary: dict[str, Any] | None = None, dataset_digest: str | None = None, dataset_digest_algorithm: str | None = None, + run_id: str | None = None, + publication_target: dict[str, str] | None = None, + publication_target_conflict: str | None = None, use_llm_judge: bool = True, ) -> dict[str, Any] | None: """Assemble the canonical Tier 3 ``agent_eval`` payload from loaded agent data. @@ -654,6 +676,8 @@ def build_agent_eval_payload( "dataset_summary": dataset_summary, "dataset_digest": effective_dataset_digest, "dataset_digest_algorithm": effective_dataset_digest_algorithm, + "run_id": run_id, + "publication_target": dict(publication_target) if isinstance(publication_target, dict) else None, "verdict_policy": verdict_policy, "execution_status": execution_status, "execution_errors": execution_errors, @@ -662,6 +686,8 @@ def build_agent_eval_payload( ), "scored_attempts": sum(_as_nonnegative_int(agent.get("scored_attempts")) for agent in agent_payloads.values()), } + if publication_target_conflict == _PUBLICATION_TARGET_CONFLICT_MARKER: + summary["publication_target_conflict"] = _PUBLICATION_TARGET_CONFLICT_MARKER if harbor_summary: summary["harbor_viewer"] = { key: harbor_summary[key] for key in ("job_url", "analysis_url") if harbor_summary.get(key) @@ -713,6 +739,8 @@ def build_agent_eval_payload( "dataset_summary": dataset_summary, "dataset_digest": effective_dataset_digest, "dataset_digest_algorithm": effective_dataset_digest_algorithm, + "run_id": run_id, + "publication_target": dict(publication_target) if isinstance(publication_target, dict) else None, "verdict_policy": verdict_policy, "agents": agent_payloads, "dimensions": best_dimensions, @@ -741,6 +769,8 @@ def build_agent_eval_payload( detail_priority=detail_priority, ), } + if publication_target_conflict == _PUBLICATION_TARGET_CONFLICT_MARKER: + payload["publication_target_conflict"] = _PUBLICATION_TARGET_CONFLICT_MARKER if harbor_summary: payload["harbor_viewer"] = harbor_summary @@ -2576,7 +2606,8 @@ def _run_truth_metadata( """Read dataset/evaluator truth owned by the evaluated run, never live source.""" persisted_result: dict[str, Any] | None = None result_file = run_dir / "result.json" - if result_file.exists(): + result_file_present = result_file.exists() + if result_file_present: with contextlib.suppress(OSError, UnicodeError, ValueError): loaded = json.loads(result_file.read_text(encoding="utf-8")) if isinstance(loaded, dict): @@ -2603,6 +2634,20 @@ def _run_truth_metadata( value = candidate.get(field_name) if field_name not in truth and isinstance(value, str) and value.strip(): truth[field_name] = value.strip() + + # Publication identity is one run-owned pair. A completed ``result.json`` + # is authoritative for rerenders; the in-memory engine result is used only + # while the runner is rendering before that final artifact exists. + identity_source = persisted_result if result_file_present else engine_result + if isinstance(identity_source, dict): + run_id = identity_source.get("run_id") + publication_target = identity_source.get("publication_target") + if isinstance(run_id, str) and run_id == run_dir.name: + truth["run_id"] = run_id + if "publication_target_conflict" in identity_source: + truth["publication_target_conflict"] = _PUBLICATION_TARGET_CONFLICT_MARKER + elif isinstance(publication_target, dict): + truth["publication_target"] = dict(publication_target) return truth diff --git a/src/skillevaluator/publication_evidence.py b/src/skillevaluator/publication_evidence.py new file mode 100644 index 00000000..f8faab6f --- /dev/null +++ b/src/skillevaluator/publication_evidence.py @@ -0,0 +1,135 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Versioned producer identity for publication-certifying validator results. + +The marker records that a result passed through a built-in Tier 1 or Tier 2 +command wrapper. It is intentionally a provenance contract for trusted local +or CI artifacts, not a cryptographic signature for hostile result files. +""" + +from __future__ import annotations + +from collections.abc import Iterable +from dataclasses import asdict, dataclass +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from skillevaluator.models.result import ValidationResult + + +PUBLICATION_EVIDENCE_SCHEMA_VERSION = 1 +_PUBLICATION_EVIDENCE_FIELDS = frozenset({"schema_version", "producer", "tier", "check_id"}) +_RECOGNIZED_CHECKS_BY_TIER = { + 1: frozenset( + { + "schema", + "version", + "security", + "pii", + "license", + "code-integrity", + "dependency", + "unicode", + "quality", + "lint", + "rubric", + } + ), + 2: frozenset({"similarity", "context-optimization"}), +} +_PRODUCER_BY_TIER = {tier: f"skillevaluator.tier{tier}" for tier in _RECOGNIZED_CHECKS_BY_TIER} + + +@dataclass(frozen=True) +class PublicationEvidenceIdentity: + """Canonical identity of one built-in publication evidence producer.""" + + schema_version: int + producer: str + tier: int + check_id: str + + +def publication_evidence_identity(value: object) -> PublicationEvidenceIdentity | None: + """Parse a marker only when every field exactly matches the built-in contract.""" + if not isinstance(value, dict) or set(value) != _PUBLICATION_EVIDENCE_FIELDS: + return None + schema_version = value.get("schema_version") + producer = value.get("producer") + tier = value.get("tier") + check_id = value.get("check_id") + if ( + type(schema_version) is not int + or schema_version != PUBLICATION_EVIDENCE_SCHEMA_VERSION + or type(tier) is not int + or tier not in _RECOGNIZED_CHECKS_BY_TIER + or not isinstance(producer, str) + or producer != _PRODUCER_BY_TIER[tier] + or not isinstance(check_id, str) + or check_id not in _RECOGNIZED_CHECKS_BY_TIER[tier] + ): + return None + return PublicationEvidenceIdentity(schema_version, producer, tier, check_id) + + +def publication_evidence_dict(value: object) -> dict[str, object] | None: + """Project an untrusted marker to the four canonical fields safe for reports.""" + identity = publication_evidence_identity(value) + return asdict(identity) if identity is not None else None + + +def result_publication_evidence(result: ValidationResult) -> PublicationEvidenceIdentity | None: + """Return one result's recognized built-in producer identity, if present.""" + metadata = result.metadata if isinstance(result.metadata, dict) else {} + if result.validator_name == "AGENT_EVAL" or isinstance(metadata.get("agent_eval"), dict): + # A live Tier 3 result cannot double as static or semantic evidence, + # even if an imported artifact carries a syntactically valid marker. + return None + return publication_evidence_identity(metadata.get("publication_evidence")) + + +def result_publication_evidence_dict(result: ValidationResult) -> dict[str, object] | None: + """Return a fresh canonical producer marker safe for report serialization.""" + identity = result_publication_evidence(result) + return asdict(identity) if identity is not None else None + + +def result_has_publication_evidence(result: ValidationResult, *, tier: int | None = None) -> bool: + """Return whether a result came through a recognized built-in tier wrapper.""" + identity = result_publication_evidence(result) + return bool(identity is not None and (tier is None or identity.tier == tier)) + + +def stamp_publication_evidence( + results: Iterable[ValidationResult], + *, + tier: int, + check_id: str, +) -> None: + """Stamp results from one trusted built-in command wrapper.""" + candidate = { + "schema_version": PUBLICATION_EVIDENCE_SCHEMA_VERSION, + "producer": _PRODUCER_BY_TIER.get(tier), + "tier": tier, + "check_id": check_id, + } + marker = publication_evidence_dict(candidate) + if marker is None: + raise ValueError(f"tier {tier} check {check_id!r} is not a recognized publication check") + for result in results: + if not isinstance(result.metadata, dict): + result.metadata = {} + result.metadata["publication_evidence"] = dict(marker) + + +__all__ = [ + "PUBLICATION_EVIDENCE_SCHEMA_VERSION", + "PublicationEvidenceIdentity", + "publication_evidence_dict", + "publication_evidence_identity", + "result_has_publication_evidence", + "result_publication_evidence", + "result_publication_evidence_dict", + "stamp_publication_evidence", +] diff --git a/src/skillevaluator/publication_identity.py b/src/skillevaluator/publication_identity.py new file mode 100644 index 00000000..c7679cb5 --- /dev/null +++ b/src/skillevaluator/publication_identity.py @@ -0,0 +1,729 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Canonical target identity for cross-job publication evidence. + +The source-tree recipe is deliberately versioned independently from SHA-256. +It binds normalized relative paths, node kinds, executable bits, and regular +file contents while omitting only SkillEvaluator's generated artifact roots +and files. POSIX traversal is descriptor-anchored and no-follow; other +platforms use checked pre/post path metadata. Observed unsafe or unstable +snapshots do not receive an identity. Like the staging subsystem, this is not +a coherent filesystem snapshot or a guarantee against adversarial concurrent +mutation by another process with the same operating-system identity. +""" + +from __future__ import annotations + +import hashlib +import os +import stat +import unicodedata +from collections.abc import Iterable +from dataclasses import dataclass +from pathlib import Path +from typing import TYPE_CHECKING, Any + +from skillevaluator.utils.path_security import canonicalize_trusted_root_alias + +if TYPE_CHECKING: + from skillevaluator.models.result import ValidationResult + + +PUBLICATION_TARGET_DIGEST_ALGORITHM = "skill-evaluator-source-tree/2" + +# These path-aware exclusions are part of the version-2 digest contract. Do +# not extend or reinterpret them without introducing a new algorithm version. +# Authored Tier 3 inputs under ``evals/`` are deliberately included; only the +# generated ``evals/results/`` subtree is omitted. +_PUBLICATION_EXCLUDED_DIR_NAMES = frozenset({".git", ".venv", "__pycache__", "node_modules"}) +_PUBLICATION_EXCLUDED_ROOT_DIRS = frozenset({".evals", ".results", ".versions", "results", "versions"}) +_PUBLICATION_EXCLUDED_DIR_PATHS = frozenset({("evals", "results")}) +_PUBLICATION_EXCLUDED_ROOT_FILES = frozenset({"BENCHMARK.md", "skill-card.md", "skill.oms.sig"}) +_CHUNK_SIZE = 1024 * 1024 +_REPARSE_POINT = getattr(stat, "FILE_ATTRIBUTE_REPARSE_POINT", 0x400) +_BINARY_FLAG = getattr(os, "O_BINARY", 0) +_DIRECTORY_FLAGS = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0) +_FILE_FLAGS = ( + os.O_RDONLY + | _BINARY_FLAG + | getattr(os, "O_NOFOLLOW", 0) + | getattr(os, "O_NONBLOCK", 0) + | getattr(os, "O_NOCTTY", 0) +) +_DESCRIPTOR_BACKEND = ( + os.name == "posix" + and hasattr(os, "O_NOFOLLOW") + and os.open in os.supports_dir_fd + and os.stat in os.supports_dir_fd + and os.scandir in os.supports_fd +) +_PATH_DESCRIPTOR_IDENTITIES_COMPARABLE = os.name == "posix" + + +class PublicationTargetConflictError(ValueError): + """Raised when a producer encounters a different persisted target claim.""" + + +@dataclass(frozen=True, slots=True) +class _PublicationEntry: + parts: tuple[str, ...] + kind: str + mode: int + fingerprint: tuple[int, int, int, int, int, int, int] + digest: str | None = None + + +def _digest_field(digest: Any, value: bytes) -> None: + """Append one unambiguous, length-delimited field to a target digest.""" + digest.update(len(value).to_bytes(8, "big")) + digest.update(value) + + +def _absolute_lexical(path: Path) -> Path: + absolute = Path(os.path.abspath(os.fspath(path))) # noqa: PTH100 - lexical normalization is intentional + return canonicalize_trusted_root_alias(absolute) + + +def _is_link(metadata: os.stat_result) -> bool: + return bool(stat.S_ISLNK(metadata.st_mode) or getattr(metadata, "st_file_attributes", 0) & _REPARSE_POINT) + + +def _fingerprint(metadata: os.stat_result) -> tuple[int, int, int, int, int, int, int]: + return ( + metadata.st_mode, + metadata.st_dev, + metadata.st_ino, + metadata.st_nlink, + metadata.st_size, + metadata.st_mtime_ns, + metadata.st_ctime_ns, + ) + + +def _validate_entry(metadata: os.stat_result, path: Path, root_device: int | None) -> str: + if _is_link(metadata): + raise ValueError(f"publication source contains a symlink or reparse point: {path}") + if stat.S_ISDIR(metadata.st_mode): + kind = "directory" + elif stat.S_ISREG(metadata.st_mode): + kind = "file" + if metadata.st_nlink != 1: + raise ValueError(f"publication source contains a hard-linked file: {path}") + else: + raise ValueError(f"publication source contains a special file: {path}") + if root_device is not None and metadata.st_dev != root_device: + raise ValueError(f"publication source crosses a filesystem mount: {path}") + return kind + + +def _same_existing_path(left: Path, right: Path) -> bool: + try: + return left.samefile(right) + except (OSError, ValueError): + return False + + +def _matches_generated_name(path: Path, canonical_name: str) -> bool: + """Match an exclusion name using the filesystem's case semantics.""" + return path.name == canonical_name or _same_existing_path(path, path.with_name(canonical_name)) + + +def _is_generated_artifact(source: Path, parts: tuple[str, ...], kind: str) -> bool: + candidate = source.joinpath(*parts) + if kind == "directory": + return bool( + any(_matches_generated_name(candidate, excluded) for excluded in _PUBLICATION_EXCLUDED_DIR_NAMES) + or ( + len(parts) == 1 + and any(_matches_generated_name(candidate, excluded) for excluded in _PUBLICATION_EXCLUDED_ROOT_DIRS) + ) + or parts in _PUBLICATION_EXCLUDED_DIR_PATHS + or (len(parts) == 2 and _same_existing_path(candidate, source / "evals" / "results")) + ) + return bool( + kind == "file" + and ( + _matches_generated_name(candidate, ".git") + or ( + len(parts) == 1 + and any(_matches_generated_name(candidate, excluded) for excluded in _PUBLICATION_EXCLUDED_ROOT_FILES) + ) + ) + ) + + +def publication_source_entry_is_excluded(source: Path, candidate: Path) -> bool: + """Return whether one existing source entry is omitted by the v2 recipe. + + Callers traversing a tree must invoke this for each entry before descending + into directories. Unsafe nodes are not exclusions; the publication digest + and runtime staging layers reject them through their own no-follow checks. + """ + try: + root = _absolute_lexical(source.expanduser()) + path = _absolute_lexical(candidate.expanduser()) + parts = path.relative_to(root).parts + if not parts: + return False + metadata = path.lstat() + if _is_link(metadata): + return False + if stat.S_ISDIR(metadata.st_mode): + kind = "directory" + elif stat.S_ISREG(metadata.st_mode): + kind = "file" + else: + return False + return _is_generated_artifact(root, parts, kind) + except (OSError, RuntimeError, ValueError): + return False + + +def publication_source_path_is_excluded(source: Path, candidate: Path) -> bool: + """Return whether a path is at or below an entry excluded by v2.""" + try: + root = _absolute_lexical(source.expanduser()) + path = _absolute_lexical(candidate.expanduser()) + parts = path.relative_to(root).parts + except (OSError, RuntimeError, ValueError): + return False + return any( + publication_source_entry_is_excluded(root, root.joinpath(*parts[:index])) for index in range(1, len(parts) + 1) + ) + + +def _hash_descriptor(descriptor: int) -> str: + digest = hashlib.sha256() + os.lseek(descriptor, 0, os.SEEK_SET) + while chunk := os.read(descriptor, _CHUNK_SIZE): + digest.update(chunk) + return digest.hexdigest() + + +def _entry(parts: tuple[str, ...], kind: str, metadata: os.stat_result, digest: str | None = None) -> _PublicationEntry: + return _PublicationEntry( + parts=parts, + kind=kind, + mode=stat.S_IMODE(metadata.st_mode), + fingerprint=_fingerprint(metadata), + digest=digest, + ) + + +def _open_directory_no_follow(path: Path) -> int: + descriptor = os.open(path.anchor, _DIRECTORY_FLAGS) + try: + for component in path.parts[1:]: + before = os.stat(component, dir_fd=descriptor, follow_symlinks=False) + if _is_link(before) or not stat.S_ISDIR(before.st_mode): + raise ValueError(f"publication source path contains a symlink or non-directory: {path}") + child = os.open(component, _DIRECTORY_FLAGS, dir_fd=descriptor) + opened = os.fstat(child) + if _is_link(opened) or not stat.S_ISDIR(opened.st_mode) or not os.path.samestat(before, opened): + os.close(child) + raise ValueError(f"publication source directory changed while opening: {path}") + os.close(descriptor) + descriptor = child + return descriptor + except BaseException: + os.close(descriptor) + raise + + +def _scan_descriptor_tree( + source: Path, + descriptor: int, + parts: tuple[str, ...], + root_device: int, + entries: list[_PublicationEntry], + *, + reverse: bool, + known_digests: dict[tuple[str, ...], str] | None = None, +) -> None: + with os.scandir(descriptor) as iterator: + names = sorted((item.name for item in iterator), reverse=reverse) + for name in names: + child_parts = (*parts, name) + child_path = source.joinpath(*child_parts) + before = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + kind = _validate_entry(before, child_path, root_device) + if _is_generated_artifact(source, child_parts, kind): + continue + if kind == "directory": + child = os.open(name, _DIRECTORY_FLAGS, dir_fd=descriptor) + try: + opened = os.fstat(child) + if _fingerprint(opened) != _fingerprint(before): + raise ValueError(f"publication source directory was replaced: {child_path}") + entries.append(_entry(child_parts, kind, opened)) + _scan_descriptor_tree( + source, + child, + child_parts, + root_device, + entries, + reverse=reverse, + known_digests=known_digests, + ) + if _fingerprint(os.fstat(child)) != _fingerprint(opened): + raise ValueError(f"publication source directory changed while scanning: {child_path}") + finally: + os.close(child) + continue + child = os.open(name, _FILE_FLAGS, dir_fd=descriptor) + try: + opened = os.fstat(child) + _validate_entry(opened, child_path, root_device) + if _fingerprint(opened) != _fingerprint(before): + raise ValueError(f"publication source file was replaced: {child_path}") + if known_digests is None: + digest = _hash_descriptor(child) + else: + expected_digest = known_digests.get(child_parts) + if expected_digest is None: + raise ValueError(f"publication source file appeared while sealing: {child_path}") + digest = _hash_descriptor(child) + if digest != expected_digest: + raise ValueError(f"publication source file changed while sealing: {child_path}") + after = os.fstat(child) + named_after = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + if _fingerprint(after) != _fingerprint(opened) or _fingerprint(named_after) != _fingerprint(before): + raise ValueError(f"publication source file changed while scanning: {child_path}") + entries.append(_entry(child_parts, kind, after, digest)) + finally: + os.close(child) + + +def _manifests_match(first: Iterable[_PublicationEntry], second: Iterable[_PublicationEntry]) -> bool: + return sorted(first, key=lambda entry: entry.parts) == sorted(second, key=lambda entry: entry.parts) + + +def _descriptor_manifest(target: Path) -> tuple[_PublicationEntry, ...]: + if target.is_dir(): + descriptor = _open_directory_no_follow(target) + try: + before = os.fstat(descriptor) + root_kind = _validate_entry(before, target, before.st_dev) + entries = [_entry((), root_kind, before)] + _scan_descriptor_tree(target, descriptor, (), before.st_dev, entries, reverse=False) + if _fingerprint(os.fstat(descriptor)) != _fingerprint(before): + raise ValueError("publication source root changed while scanning") + revalidated_root = os.fstat(descriptor) + revalidated = [_entry((), root_kind, revalidated_root)] + _scan_descriptor_tree(target, descriptor, (), before.st_dev, revalidated, reverse=True) + if _fingerprint(os.fstat(descriptor)) != _fingerprint(revalidated_root) or not _manifests_match( + entries, revalidated + ): + raise ValueError("publication source changed while revalidating") + sealed_root = os.fstat(descriptor) + sealed = [_entry((), root_kind, sealed_root)] + known_digests = { + entry.parts: entry.digest for entry in revalidated if entry.kind == "file" and entry.digest is not None + } + _scan_descriptor_tree( + target, + descriptor, + (), + before.st_dev, + sealed, + reverse=False, + known_digests=known_digests, + ) + if _fingerprint(os.fstat(descriptor)) != _fingerprint(sealed_root) or not _manifests_match( + revalidated, sealed + ): + raise ValueError("publication source changed while sealing") + return tuple(entries) + finally: + os.close(descriptor) + + parent = _open_directory_no_follow(target.parent) + try: + before = os.stat(target.name, dir_fd=parent, follow_symlinks=False) + if _validate_entry(before, target, before.st_dev) != "file": + raise ValueError("publication source is not a regular file") + descriptor = os.open(target.name, _FILE_FLAGS, dir_fd=parent) + try: + opened = os.fstat(descriptor) + if _fingerprint(opened) != _fingerprint(before): + raise ValueError("publication source file was replaced") + digest = _hash_descriptor(descriptor) + after = os.fstat(descriptor) + named_after = os.stat(target.name, dir_fd=parent, follow_symlinks=False) + if _fingerprint(after) != _fingerprint(opened) or _fingerprint(named_after) != _fingerprint(before): + raise ValueError("publication source file changed while scanning") + revalidated_digest = _hash_descriptor(descriptor) + revalidated = os.fstat(descriptor) + named_revalidated = os.stat(target.name, dir_fd=parent, follow_symlinks=False) + if ( + digest != revalidated_digest + or _fingerprint(revalidated) != _fingerprint(after) + or _fingerprint(named_revalidated) != _fingerprint(named_after) + ): + raise ValueError("publication source file changed while revalidating") + return (_entry((), "file", revalidated, revalidated_digest),) + finally: + os.close(descriptor) + finally: + os.close(parent) + + +def _validate_fallback_components(path: Path) -> None: + current = Path(path.anchor) + for component in path.parts[1:]: + current /= component + metadata = current.lstat() + if _is_link(metadata) or not stat.S_ISDIR(metadata.st_mode): + raise ValueError(f"publication source path contains a symlink or non-directory: {current}") + + +def _fallback_opened_matches_named(opened: os.stat_result, named: os.stat_result) -> bool: + """Compare fallback metadata without assuming Windows CRT identities.""" + if _PATH_DESCRIPTOR_IDENTITIES_COMPARABLE: + return _fingerprint(opened) == _fingerprint(named) + return ( + stat.S_IFMT(opened.st_mode) == stat.S_IFMT(named.st_mode) + and opened.st_nlink == named.st_nlink + and opened.st_size == named.st_size + ) + + +def _hash_path_checked(path: Path, before: os.stat_result, root_device: int) -> str: + descriptor = os.open(path, _FILE_FLAGS) + try: + opened = os.fstat(descriptor) + _validate_entry( + opened, + path, + root_device if _PATH_DESCRIPTOR_IDENTITIES_COMPARABLE else None, + ) + if not _fallback_opened_matches_named(opened, before): + raise ValueError(f"publication source file was replaced: {path}") + digest = _hash_descriptor(descriptor) + after = os.fstat(descriptor) + finally: + os.close(descriptor) + named_after = path.lstat() + if _fingerprint(after) != _fingerprint(opened) or _fingerprint(named_after) != _fingerprint(before): + raise ValueError(f"publication source file changed while scanning: {path}") + return digest + + +def _scan_fallback_tree( + source: Path, + parts: tuple[str, ...], + root_device: int, + entries: list[_PublicationEntry], + *, + reverse: bool, + known_digests: dict[tuple[str, ...], str] | None = None, +) -> None: + directory = source.joinpath(*parts) + names = sorted((item.name for item in os.scandir(directory)), reverse=reverse) + for name in names: + child_parts = (*parts, name) + child = source.joinpath(*child_parts) + before = child.lstat() + kind = _validate_entry(before, child, root_device) + if _is_generated_artifact(source, child_parts, kind): + continue + if kind == "directory": + entries.append(_entry(child_parts, kind, before)) + _scan_fallback_tree( + source, + child_parts, + root_device, + entries, + reverse=reverse, + known_digests=known_digests, + ) + if _fingerprint(child.lstat()) != _fingerprint(before): + raise ValueError(f"publication source directory changed while scanning: {child}") + else: + if known_digests is None: + digest = _hash_path_checked(child, before, root_device) + metadata = before + else: + expected_digest = known_digests.get(child_parts) + if expected_digest is None: + raise ValueError(f"publication source file appeared while sealing: {child}") + digest = _hash_path_checked(child, before, root_device) + if digest != expected_digest: + raise ValueError(f"publication source file changed while sealing: {child}") + metadata = before + entries.append(_entry(child_parts, kind, metadata, digest)) + + +def _fallback_manifest(target: Path) -> tuple[_PublicationEntry, ...]: + if target.is_dir(): + _validate_fallback_components(target) + before = target.lstat() + root_kind = _validate_entry(before, target, before.st_dev) + entries = [_entry((), root_kind, before)] + _scan_fallback_tree(target, (), before.st_dev, entries, reverse=False) + if _fingerprint(target.lstat()) != _fingerprint(before): + raise ValueError("publication source root changed while scanning") + revalidated_root = target.lstat() + revalidated = [_entry((), root_kind, revalidated_root)] + _scan_fallback_tree(target, (), before.st_dev, revalidated, reverse=True) + if _fingerprint(target.lstat()) != _fingerprint(revalidated_root) or not _manifests_match(entries, revalidated): + raise ValueError("publication source changed while revalidating") + sealed_root = target.lstat() + sealed = [_entry((), root_kind, sealed_root)] + known_digests = { + entry.parts: entry.digest for entry in revalidated if entry.kind == "file" and entry.digest is not None + } + _scan_fallback_tree( + target, + (), + before.st_dev, + sealed, + reverse=False, + known_digests=known_digests, + ) + if _fingerprint(target.lstat()) != _fingerprint(sealed_root) or not _manifests_match( + revalidated, + sealed, + ): + raise ValueError("publication source changed while sealing") + return tuple(entries) + + _validate_fallback_components(target.parent) + before = target.lstat() + if _validate_entry(before, target, before.st_dev) != "file": + raise ValueError("publication source is not a regular file") + digest = _hash_path_checked(target, before, before.st_dev) + revalidated_before = target.lstat() + revalidated_digest = _hash_path_checked(target, revalidated_before, revalidated_before.st_dev) + if _fingerprint(revalidated_before) != _fingerprint(before) or revalidated_digest != digest: + raise ValueError("publication source file changed while revalidating") + return (_entry((), "file", revalidated_before, revalidated_digest),) + + +def _normalized_relative_path(parts: tuple[str, ...]) -> str: + """Return the platform-independent NFC path used by the recipe.""" + return unicodedata.normalize("NFC", "/".join(parts)) + + +def _manifest_digest( + entries: Iterable[Any], +) -> str: + """Digest securely captured manifest entries using recipe version 2.""" + captured_entries = tuple(entries) + normalized_entries: list[tuple[str, Any]] = [] + normalized_paths: set[str] = set() + for entry in captured_entries: + relative = _normalized_relative_path(entry.parts) + relative.encode("utf-8", errors="strict") + if relative in normalized_paths: + raise ValueError(f"normalized publication path collision: {relative}") + normalized_paths.add(relative) + normalized_entries.append((relative, entry)) + + digest = hashlib.sha256() + _digest_field(digest, PUBLICATION_TARGET_DIGEST_ALGORITHM.encode("ascii")) + for relative, entry in sorted(normalized_entries, key=lambda item: item[0].encode("utf-8")): + if entry.kind == "directory": + node_kind = b"D" + elif entry.kind == "file": + node_kind = b"F" + else: + raise ValueError(f"unsupported publication node kind: {entry.kind}") + if node_kind == b"D": + if entry.digest is not None: + raise ValueError("a publication directory cannot contain a file digest") + content_digest = b"" + else: + if not isinstance(entry.digest, str) or len(entry.digest) != 64: + raise ValueError("a publication file requires a SHA-256 digest") + content_digest = bytes.fromhex(entry.digest) + if content_digest.hex() != entry.digest: + raise ValueError("a publication file digest is not canonical lowercase hexadecimal") + for field in ( + relative.encode("utf-8"), + node_kind, + (entry.mode & 0o111).to_bytes(1, "big"), + content_digest, + ): + _digest_field(digest, field) + return digest.hexdigest() + + +def publication_source_digest(target_path: Path) -> str | None: + """Return the canonical digest of one stable, safe source snapshot. + + Directories use a descriptor-anchored manifest on supported POSIX systems + and checked pre/post metadata elsewhere. Both detect incidental mid-scan + mutation. Symlinks, reparse points, hard-linked regular files, special + files, mount crossings, invalid Unicode paths, and NFC path collisions fail + closed by returning ``None``. + """ + try: + target = _absolute_lexical(target_path.expanduser()) + metadata = target.lstat() + _validate_entry(metadata, target, None) + entries = _descriptor_manifest(target) if _DESCRIPTOR_BACKEND else _fallback_manifest(target) + return _manifest_digest(entries) + except ( + OSError, + OverflowError, + RuntimeError, + UnicodeError, + ValueError, + ): + return None + return None + + +def _canonical_directory_entry_name(path: Path) -> str: + """Return the spelling recorded by the parent directory for *path*.""" + metadata = path.lstat() + matches: list[str] = [] + with os.scandir(path.parent) as entries: + for entry in entries: + try: + # DirEntry.stat() exposes zero identity fields on Windows; + # re-stat the named path before comparing filesystem identity. + observed = path.parent.joinpath(entry.name).lstat() + except OSError: + continue + if os.path.samestat(metadata, observed): + matches.append(entry.name) + if path.name in matches: + return path.name + if len(matches) == 1: + return matches[0] + raise ValueError(f"publication source directory entry is ambiguous: {path}") + + +def publication_target_from_path(target_path: Path) -> dict[str, str] | None: + """Return the canonical publication target for one source snapshot.""" + try: + target = _absolute_lexical(target_path.expanduser()) + entry_name = _canonical_directory_entry_name(target) + skill_name = unicodedata.normalize("NFC", entry_name) + if not skill_name: + return None + skill_name.encode("utf-8", errors="strict") + except (OSError, RuntimeError, UnicodeError, ValueError): + return None + digest = publication_source_digest(target) + if digest is None: + return None + try: + if _canonical_directory_entry_name(target) != entry_name: + return None + except (OSError, RuntimeError, ValueError): + return None + return { + "skill_name": skill_name, + "skill_digest": f"sha256:{digest}", + "skill_digest_algorithm": PUBLICATION_TARGET_DIGEST_ALGORITHM, + } + + +def _identity_containers(result: ValidationResult) -> tuple[dict[str, Any], ...]: + """Return all persisted identity containers owned by one result.""" + metadata = result.metadata + if not isinstance(metadata, dict): + raise PublicationTargetConflictError("validation result metadata is not a mapping") + containers = [metadata] + payload = metadata.get("agent_eval") + if isinstance(payload, dict): + containers.append(payload) + summary = payload.get("summary") + if isinstance(summary, dict): + containers.append(summary) + return tuple(containers) + + +def stamp_publication_identity( + results: Iterable[ValidationResult], + target: dict[str, str], +) -> dict[str, str]: + """Atomically stamp an already captured, producer-owned target identity.""" + expected_keys = {"skill_name", "skill_digest", "skill_digest_algorithm"} + if ( + set(target) != expected_keys + or not isinstance(target.get("skill_name"), str) + or not target["skill_name"] + or unicodedata.normalize("NFC", target["skill_name"]) != target["skill_name"] + or not isinstance(target.get("skill_digest"), str) + or len(target["skill_digest"]) != 71 + or not target["skill_digest"].startswith("sha256:") + or any(character not in "0123456789abcdef" for character in target["skill_digest"][7:]) + or target.get("skill_digest_algorithm") != PUBLICATION_TARGET_DIGEST_ALGORITHM + ): + raise PublicationTargetConflictError("publication target identity is malformed") + + containers: list[dict[str, Any]] = [] + for result in tuple(results): + containers.extend(_identity_containers(result)) + for container in containers: + if "publication_target" in container and container["publication_target"] != target: + raise PublicationTargetConflictError("conflicting publication target already persisted") + for container in containers: + if "publication_target" not in container: + container["publication_target"] = dict(target) + return target + + +def _mark_publication_target_conflict( + results: Iterable[ValidationResult], + message: str, +) -> None: + """Remove contradictory target claims before persisting a conflict.""" + for result in results: + if not isinstance(result.metadata, dict): + continue + for container in _identity_containers(result): + container.pop("publication_target", None) + result.metadata["publication_target_conflict"] = message + + +def finalize_publication_target( + results: Iterable[ValidationResult], + target_path: Path, + initial_target: dict[str, str] | None, +) -> dict[str, str] | None: + """Stamp results only when the producer's source stayed unchanged.""" + result_list = tuple(results) + final_target = publication_target_from_path(target_path) + if initial_target is None or final_target != initial_target: + _mark_publication_target_conflict(result_list, "source changed during validation") + return None + try: + return stamp_publication_identity(result_list, final_target) + except PublicationTargetConflictError: + _mark_publication_target_conflict(result_list, "conflicting producer identity") + return None + + +def stamp_publication_target( + results: Iterable[ValidationResult], + target_path: Path, +) -> dict[str, str] | None: + """Atomically stamp newly generated results with their source identity. + + Every preexisting claim is checked before any result is changed. A claim + for another snapshot (including a malformed or explicit-null claim) raises + :class:`PublicationTargetConflictError` instead of being retained beside + newly stamped matching claims. + """ + target = publication_target_from_path(target_path) + if target is None: + return None + + return stamp_publication_identity(results, target) + + +__all__ = [ + "PUBLICATION_TARGET_DIGEST_ALGORITHM", + "PublicationTargetConflictError", + "finalize_publication_target", + "publication_source_digest", + "publication_target_from_path", + "stamp_publication_identity", + "stamp_publication_target", +] diff --git a/src/skillevaluator/publication_text.py b/src/skillevaluator/publication_text.py new file mode 100644 index 00000000..ccc1b8f9 --- /dev/null +++ b/src/skillevaluator/publication_text.py @@ -0,0 +1,397 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Shared Unicode security normalization for publication identities.""" + +from __future__ import annotations + +import math +import unicodedata + +UNICODE_CONFUSABLES_VERSION = "17.0.0" +UNICODE_CONFUSABLES_SOURCE_SHA256 = "091c7f82fc39ef208faf8f94d29c244de99254675e09de163160c810d13ef22a" +UNICODE_CONFUSABLES_GENERATOR_UCD_VERSION = "15.1.0" +UNICODE_CONFUSABLES_SUBSET_SOURCE_COUNT = 1954 + +_INVISIBLE_IDENTITY_CHARACTERS = frozenset( + { + "\u115f", + "\u1160", + "\u2800", + "\u3164", + "\uffa0", + "\U00013441", + "\U00013442", + "\U0001d159", + } +) +_RESERVED_IDENTITIES = frozenset( + { + "model not recorded", + "model unknown", + "n a", + "na", + "none", + "not applicable", + "not available", + "not recorded", + "null", + "tbd", + "unknown", + "unknown model", + "unset", + } +) +_SECURITY_TEXT_CONFUSABLES = {"\u0406": "I", "\u0456": "i"} + +# Generated from the Unicode 17.0.0 confusables data. Each source maps directly +# to its case-insensitive effective prototype, retaining separator boundaries +# and sources whose prototype disappears under the pinned filtering contract. +# This is the complete subset relevant to the reserved identities above; +# keeping that closed subset avoids carrying the full Unicode security table. +# Generation pins Python's Unicode Character Database 15.1.0 so future host +# category changes do not silently alter the committed map. +# Source: https://www.unicode.org/Public/17.0.0/security/confusables.txt +# Unicode Data Files and Software License: https://www.unicode.org/license.txt +_CONFUSABLE_PROTOTYPE_GROUPS = ( + ( + "", + "\u05ad\u05ae\u05a8\u05a4\u1ab4\u20db\u0619\u08f3\u0343\u0315\u064f\u065d\u059c\u059d\u0618\u0747\u0341" + "\u0954\u064e\u0340\u0953\u030c\ua67c\u0658\u065a\u036e\u0945\U00011b66\u06e8\u0310\u0901\u0981\u0a81\u0b01" + "\u0c00\u0c81\u0d01\U000114bf\u1cd0\u0311\u065b\u07ee\ua6f0\u05af\u06df\u17d3\u309a\u0652\u0b82\u1036\u17c6" + "\U00011300\u0e4d\u0ecd\u0366\u2dea\u08eb\u07f3\u064b\u08f0\u0342\u0653\u05c4\u06ec\u0740\u08ea\u0741\u0358" + "\u05b9\u05ba\u05c2\u05c1\u07ed\u0902\u0a02\u0a82\u0bcd\u0337\u1ab7\u0322\u0345\u1cd2\u0305\u0659\u07eb" + "\ua6f1\u1ae2\u1ae8\u1cda\u0657\u0357\u08ff\u08f8\u0900\u1ad9\U0001e6ee\u1ced\u1cdc\u0656\u1cd5\u0347\u08f9" + "\u08fa\u309b\u309c\u0336\u302c\u05c5\u08ed\u1cdd\u05b4\u065c\u093c\u09bc\u0a3c\u0abc\u0b3c\U000111ca" + "\U000114c3\U00010a3a\u08ee\u1cde\u0f37\u302d\u0327\u0321\u0339\u1cd9\u1cd8\u0952\u0320\u08f1\u08e8\u08e5" + "\u08f2\u061a\u0317\u065f\u030d\u0742\u0a03\u0c03\u0c83\u0d03\u0d83\u1038\U000114c1\u17cb\u0ec8\u0ec9\u0eca" + "\u0ecb\ua66f\u2df6\u2ded\u2df7\u2de8\u2def\u1dee\u0949\u093b\U000111cb\U00011b60\u0ac1\u0ac2\u0a4b\u0a48" + "\u0a4d\u0acd\U000114b0\u093f\u0a3f\U000114b1\U000114b9\U000114bc\U000114be\U000114c2\U000114bd\u1031\u0d3f" + "\u0d40\u0d46\u0d48\u0d47\u0cbf\u0cc1\u0c42\u0cc3\u0c44\u0d42\u0d43\U000115dc\U000115dd\u17b7\u17b8\u17b9" + "\u17ba\u0eb8\u0eb9\u0f77\u0f79\u0f7b\u0f7d\U00011cb2\u1734\u109e\u3164\u0cdc\u1de8\u2dee\u1ae7\u031a\u0295" + "\ua7cf\u0348\u0956\u0a41\u0957\u0a42\u0947\u0a47\u5152\U0001f40d\U0001f443\U0001f377\U0001f3e2\U0001f333" + "\U0001f34e\U0001f34f\U0001f352\U0001f353\u28ff\u29b5\u21c4\u21cc\u2657\u265d\U0001f514\u6138", + ), + ( + " ", + "\ufc5e\ufc5f\ufc60\ufc61\ufc62\ufc63\u2028\u2029\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2008" + "\u2009\u200a\u205f\xa0\u2007\u202f\u07fa\ufe4d\ufe4e\ufe4f\u2010\u2011\u2012\u2013\ufe58\u06d4\u2043\u02d7" + "\u2212\u2796\u2cbb\u2cba\u2a29\u2e1a\ufb29\u2238\u2cb3\u2cb2\u2a2a\ua4fe\uff5e\u060d\u066b\u201a\xb8\ua4f9" + "\u2e32\u066c\u037e\u2e35\u0903\u0a83\uff1a\u0589\u0703\u0704\u16ec\ufe30\u1803\u1809\u205a\u05c3\u02f8" + "\ua789\u2236\u02d0\ua4fd\U00011dd9\u2a74\u29f4\uff01\u01c3\u2d51\u203c\u2049\u0294\u0241\u097d\u13ae\ua6eb" + "\u2048\u2047\u2e2e\U0001d16d\u2024\u0701\u0702\ua60e\U00010a50\u0660\u06f0\ua4f8\ua4fb\u2025\ua4fa\u2026" + "\ua6f4\u30fb\uff65\u16eb\u0387\u2e31\U00010101\u2022\u2027\u2219\u22c5\ua78f\u1427\u22ef\u2d48\u1444\u22d7" + "\u1437\u1440\ua830\u0965\u1c3c\u104b\u1aa9\u1aab\u1b5f\U00010a57\U0001144c\U00011642\U00011c42\u1c7f\u055d" + "\uff07\u2018\u2019\u201b\u05f3\u2032\u2035\u055a`\u1fef\uff40\xb4\u0384\u1ffd\u1fbd\u1fbf\u1ffe\u02b9" + "\u0374\u02c8\u02ca\u02cb\u02f4\u02bb\u02bd\u02bc\u02be\ua78c\u05d9\u07f4\u07f5\u144a\u16cc\U00016f51" + '\U00016f52\u1cd3"\uff02\u201c\u201d\u201f\u05f4\u2033\u2036\u3003\u02dd\u02ba\u02f6\u02ee\u05f2\u2034' + "\u2037\u2057\uff3b\u2768\u2772\u3014\ufd3e\u2e28\uff3d\u2769\u2773\u3015\ufd3f\u2e29\u2774\U0001d114\u2775" + "\u301a\u301b\u27e8\u2329\u3008\u31db\u304f\U00021fe8\u27e9\u232a\u3009\uff3e\u2e3f\u204e\u066d\u2217" + "\U0001031f\u1735\u2041\u2215\u2044\u2571\u27cb\u29f8\U0001d23a\u31d3\u3033\u2cc7\u2cc6\u30ce\u4e3f\u2f03" + "\u29f6\u2afd\u2afb\uff3c\ufe68\u2216\u27cd\u29f5\u29f9\U0001d20f\U0001d23b\u31d4\u4e36\u2f02\u2cf9\u244a" + "\ua778\u0af0\U000110bb\U000111c7\u26ac\U000111db\u17d9\u17d5\u17da\u0f0c\u0f0e\u02c4\u02c6\u02fb\ua716" + "\ua714\u3002\u2e30\u02da\u2218\u25cb\u25e6\u235c\U00010ed0\u2364\u0bf5\u0f1b\u0f1f\u0fce\u0f1e\u24b8\u24c7" + "\u24c5\U0001d21b\u2bec\u2bed\u2bee\u2bef\u21b5\u2965\U0001d6db\U0001d715\U0001d74f\U0001d789\U0001d7c3" + "\U0001e8cc\U0001e8cd\xf0\u2300\U0001d6c1\U0001d6fb\U0001d735\U0001d76f\U0001d7a9\U000118a8\u2362\u236b" + "\u2588\u25a0\u2a3f\u16ed\u2795\U0001029b\U0001e6e9\u2a23\u2a22\u2a24\u2214\u2a25\u2a26\u2797\u2039\u276e" + "\u02c2\U0001d236\u1438\u16b2\u22d6\u2cb5\u2cb4\u1445\u226a\u22d8\u1400\u2e40\u30a0\ua4ff\u225a\u2259\u2257" + "\u2250\u2251\u2b96\u2a6e\u2a75\u2a76\u225e\u203a\u276f\u02c3\U0001d237\u1433\U00016f3f\u1441\u2aa5\u226b" + "\u2a20\u22d9\u2053\u02dc\u1fc0\u223c\u2368\u2e1e\u2a6a\u2e1f\U0001e8c8\u22c0\u222f\u2230\u2e2b\u2e2a\u2e2c" + "\U000111de\u264e\U0001f75e\u2263\u2cb7\u2a03\u2a04\U0001d238\U0001d239\u2a05\u2a06\u2a02\u235f\U0001f771" + "\U0001f755\u25c1\u25b7\u2363\ufe34\u25e0\u2a3d\u2325\u29c7\u25ce\u29be\u29c5\u29b0\u23c3\u23c2\u23c1\u23c6" + "\u2638\ufe35\ufe36\ufe37\ufe38\ufe39\ufe3a\u25b1\u23fc\ufe31\uff5c\u2503\u250f\u2523\u2590\u2597\u259d" + "\u2610\uffed\u25b8\u25ba\u2ce9\U0001f70a\U0001f312\U0001f319\u23fe\U0001f318\u29d9\U0001f73a\u2a3e\u2669" + "\u266a\u24ea\u21ba\U0001ccfb\U0001f10f\u20a4\u3012\u3036\ufb39", + ), + ( + "a", + "\u237a\uff41\U0001d41a\U0001d44e\U0001d482\U0001d4b6\U0001d4ea\U0001d51e\U0001d552\U0001d586\U0001d5ba" + "\U0001d5ee\U0001d622\U0001d656\U0001d68a\u0251\u03b1\U0001d6c2\U0001d6fc\U0001d736\U0001d770\U0001d7aa" + "\u0430\uff21\U0001ccd6\U0001d400\U0001d434\U0001d468\U0001d49c\U0001d4d0\U0001d504\U0001d538\U0001d56c" + "\U0001d5a0\U0001d5d4\U0001d608\U0001d63c\U0001d670\u0391\U0001d6a8\U0001d6e2\U0001d71c\U0001d756\U0001d790" + "\u0410\u13aa\u15c5\ua4ee\U00016f40\U000102a0\u2376\u01ce\u01cd\u0227\u0226\u1e9a", + ), + ( + "b", + "\U0001d41b\U0001d44f\U0001d483\U0001d4b7\U0001d4eb\U0001d51f\U0001d553\U0001d587\U0001d5bb\U0001d5ef" + "\U0001d623\U0001d657\U0001d68b\u0184\u042c\u13cf\u1472\u15af\U00016eb6\uff22\u212c\U0001ccd7\U0001d401" + "\U0001d435\U0001d469\U0001d4d1\U0001d505\U0001d539\U0001d56d\U0001d5a1\U0001d5d5\U0001d609\U0001d63d" + "\U0001d671\ua7b4\u0392\U0001d6a9\U0001d6e3\U0001d71d\U0001d757\U0001d791\u2c82\u0412\u13f4\u15f7\ua4d0" + "\U00010282\U000102a1\U00010301\u0253\u1473\u0183\u0182\u0411\u0180\u048d\u048c\u0463\u0462", + ), + ( + "c", + "\uff43\u217d\U0001d41c\U0001d450\U0001d484\U0001d4b8\U0001d4ec\U0001d520\U0001d554\U0001d588\U0001d5bc" + "\U0001d5f0\U0001d624\U0001d658\U0001d68c\u1d04\u03f2\u2ca5\u0441\u1004\u105a\uabaf\U0001043d\U0001f74c" + "\U000118e9\U000118f2\uff23\u216d\u2102\u212d\U0001ccd8\U0001d402\U0001d436\U0001d46a\U0001d49e\U0001d4d2" + "\U0001d56e\U0001d5a2\U0001d5d6\U0001d60a\U0001d63e\U0001d672\u03f9\u2ca4\u0421\u13df\ua4da\U000102a2" + "\U00010302\U00010415\U0001051c\xa2\u023c\u20a1\U0001f16e\xe7\u04ab\xc7\u04aa", + ), + ( + "d", + "\u217e\u2146\U0001d41d\U0001d451\U0001d485\U0001d4b9\U0001d4ed\U0001d521\U0001d555\U0001d589\U0001d5bd" + "\U0001d5f1\U0001d625\U0001d659\U0001d68d\u0501\u13e7\u146f\ua4d2\u216e\u2145\U0001ccd9\U0001d403\U0001d437" + "\U0001d46b\U0001d49f\U0001d4d3\U0001d507\U0001d53b\U0001d56f\U0001d5a3\U0001d5d7\U0001d60b\U0001d63f" + "\U0001d673\u13a0\u15de\u15ea\ua4d3\u0257\u0256\u018c\u0111\u0110\xd0\u0189\u20ab", + ), + ( + "e", + "\u212e\uff45\u212f\u2147\U0001d41e\U0001d452\U0001d486\U0001d4ee\U0001d522\U0001d556\U0001d58a\U0001d5be" + "\U0001d5f2\U0001d626\U0001d65a\U0001d68e\uab32\u0435\u04bd\u22ff\uff25\u2130\U0001ccda\U0001d404\U0001d438" + "\U0001d46c\U0001d4d4\U0001d508\U0001d53c\U0001d570\U0001d5a4\U0001d5d8\U0001d60c\U0001d640\U0001d674\u0395" + "\U0001d6ac\U0001d6e6\U0001d720\U0001d75a\U0001d794\u0415\u2d39\u13ac\ua4f0\U000118a6\U000118ae\U00010286" + "\u011b\u011a\u0247\u0246\u04bf\u04be", + ), + ( + "i", + "\u02db\u2373\uff49\u2170\u2139\u2148\U0001d422\U0001d456\U0001d48a\U0001d4be\U0001d4f2\U0001d526\U0001d55a" + "\U0001d58e\U0001d5c2\U0001d5f6\U0001d62a\U0001d65e\U0001d692\u0131\U0001d6a4\u026a\u0269\u03b9\u1fbe\u037a" + "\U0001d6ca\U0001d704\U0001d73e\U0001d778\U0001d7b2\u2c93\u0456\ua647\u0582\uab75\u13a5\U000118c3\u24db" + "\u2378\u01d0\u0268\u1d7b\u1d7c", + ), + ( + "k", + "\U0001d424\U0001d458\U0001d48c\U0001d4c0\U0001d4f4\U0001d528\U0001d55c\U0001d590\U0001d5c4\U0001d5f8" + "\U0001d62c\U0001d660\U0001d694\u212a\uff2b\U0001cce0\U0001d40a\U0001d43e\U0001d472\U0001d4a6\U0001d4da" + "\U0001d50e\U0001d542\U0001d576\U0001d5aa\U0001d5de\U0001d612\U0001d646\U0001d67a\u039a\U0001d6b1\U0001d6eb" + "\U0001d725\U0001d75f\U0001d799\u2c94\u041a\u13e6\u16d5\ua4d7\U00010518\u0199\u2c69\u049a\u20ad\ua740\u049e", + ), + ( + "l", + "\u01cf\u05c0|\u2223\u23fd\uffe81\u0661\u06f1\U00010320\U0001e8c7\U0001ccf1\U0001d7cf\U0001d7d9\U0001d7e3" + "\U0001d7ed\U0001d7f7\U0001fbf1I\uff29\u2160\u2110\u2111\U0001ccde\U0001d408\U0001d43c\U0001d470\U0001d4d8" + "\U0001d540\U0001d574\U0001d5a8\U0001d5dc\U0001d610\U0001d644\U0001d678\u0196\uff4c\u217c\u2113\U0001d425" + "\U0001d459\U0001d48d\U0001d4c1\U0001d4f5\U0001d529\U0001d55d\U0001d591\U0001d5c5\U0001d5f9\U0001d62d" + "\U0001d661\U0001d695\u01c0\u0399\U0001d6b0\U0001d6ea\U0001d724\U0001d75e\U0001d798\u2c92\u0406\u04cf\u04c0" + "\u05d5\u05df\u0627\U0001ee00\U0001ee80\ufe8e\ufe8d\u07ca\u2d4f\u16c1\ua4f2\U00016f28\U0001028a\U00010309" + "\U00011dda\U00011de1\U00016eaa\U0001d22a\u216c\u2112\U0001cce1\U0001d40b\U0001d43f\U0001d473\U0001d4db" + "\U0001d50f\U0001d543\U0001d577\U0001d5ab\U0001d5df\U0001d613\U0001d647\U0001d67b\u2cd0\u13de\u14aa\ua4e1" + "\U00016f16\U000118a3\U000118b2\U0001041b\U00010526\ufd3c\ufd3d\u0142\u0141\u026d\u0197\u019a\u026b\u0625" + "\ufe88\ufe87\u0673\u0623\ufe82\ufe81", + ), + ( + "n", + "\U0001d427\U0001d45b\U0001d48f\U0001d4c3\U0001d4f7\U0001d52b\U0001d55f\U0001d593\U0001d5c7\U0001d5fb" + "\U0001d62f\U0001d663\U0001d697\u0578\u057c\uff2e\u2115\U0001cce3\U0001d40d\U0001d441\U0001d475\U0001d4a9" + "\U0001d4dd\U0001d511\U0001d579\U0001d5ad\U0001d5e1\U0001d615\U0001d649\U0001d67d\u039d\U0001d6b4\U0001d6ee" + "\U0001d728\U0001d762\U0001d79c\u2c9a\ua4e0\U00010513\U0001018e\u0273\u019e\u014b\u03b7\U0001d6c8\U0001d702" + "\U0001d73c\U0001d776\U0001d7b0\u0572\u019d\u1d70\u0146\u2229\u22c2\U0001d245\u1260\u144e\ua4f5", + ), + ( + "o", + "\u0c02\u0c82\u0d02\u0d82\u0966\u09e6\u0a66\u0ae6\u0b66\u0be6\u0c66\u0d66\u0e50\u0ed0\u1040\u17e0\U000114d0" + "\u0665\u06f5\uff4f\u2134\U0001d428\U0001d45c\U0001d490\U0001d4f8\U0001d52c\U0001d560\U0001d594\U0001d5c8" + "\U0001d5fc\U0001d630\U0001d664\U0001d698\u1d0f\u1d11\uab3d\u03bf\U0001d6d0\U0001d70a\U0001d744\U0001d77e" + "\U0001d7b8\u03c3\U0001d6d4\U0001d70e\U0001d748\U0001d782\U0001d7bc\u2c9f\u03ed\u043e\u10ff\u0585\u05e1" + "\u0647\U0001ee24\U0001ee64\U0001ee84\ufeeb\ufeec\ufeea\ufee9\u06be\ufbac\ufbad\ufbab\ufbaa\u06c1\ufba8" + "\ufba9\ufba7\ufba6\u06d5\u0d20\u101d\U000104ea\U000118c8\U000118d7\U0001042c0\u07c0\u0ce6\u3007\U000118e0" + "\U0001ccf0\U0001d7ce\U0001d7d8\U0001d7e2\U0001d7ec\U0001d7f6\U0001fbf0\uff2f\U0001cce4\U0001d40e\U0001d442" + "\U0001d476\U0001d4aa\U0001d4de\U0001d512\U0001d546\U0001d57a\U0001d5ae\U0001d5e2\U0001d616\U0001d64a" + "\U0001d67e\u039f\U0001d6b6\U0001d6f0\U0001d72a\U0001d764\U0001d79e\u2c9e\u041e\u0555\u2d54\u12d0\u0b20" + "\U000104c2\ua4f3\U000118b5\U00010292\U000102ab\U00010404\U00010516\U00011de0\u2070\u1d52\u01d2\u01d1\u06ff" + "\u0150\xf8\uab3e\xd8\u2d41\u01fe\u0275\ua74b\u2c91\u04e9\u0473\uab8e\uabbb\u2296\u229d\u236c\U0001d21a" + "\U0001f714\u019f\ua74a\u03b8\u03d1\U0001d6c9\U0001d6dd\U0001d703\U0001d717\U0001d73d\U0001d751\U0001d777" + "\U0001d78b\U0001d7b1\U0001d7c5\u0398\u03f4\U0001d6af\U0001d6b9\U0001d6e9\U0001d6f3\U0001d723\U0001d72d" + "\U0001d75d\U0001d767\U0001d797\U0001d7a1\u2c90\u04e8\u0472\u2d31\u13be\u13eb\uab74\ufcd9\u01a1\u01a0\u10d7" + "\u1010\u03db\U0001d6d3\U0001d70d\U0001d747\U0001d781\U0001d7bb\u2c8b\xf6\u06c2\ufba5\ufba4", + ), + ( + "p", + "\u2374\uff50\U0001d429\U0001d45d\U0001d491\U0001d4c5\U0001d4f9\U0001d52d\U0001d561\U0001d595\U0001d5c9" + "\U0001d5fd\U0001d631\U0001d665\U0001d699\xfe\u01bf\u03c1\u03f1\U0001d6d2\U0001d6e0\U0001d70c\U0001d71a" + "\U0001d746\U0001d754\U0001d780\U0001d78e\U0001d7ba\U0001d7c8\u03f8\u2ca3\u2ccf\u0440\uff30\u2119\U0001cce5" + "\U0001d40f\U0001d443\U0001d477\U0001d4ab\U0001d4df\U0001d513\U0001d57b\U0001d5af\U0001d5e3\U0001d617" + "\U0001d64b\U0001d67f\u03a1\U0001d6b8\U0001d6f2\U0001d72c\U0001d766\U0001d7a0\u2ca2\u2cce\u0420\u13e2\u146d" + "\ua4d1\U00010295\u01a5\u1d7d\u03f7\U000104c4", + ), + ( + "r", + "\U0001d42b\U0001d45f\U0001d493\U0001d4c7\U0001d4fb\U0001d52f\U0001d563\U0001d597\U0001d5cb\U0001d5ff" + "\U0001d633\U0001d667\U0001d69b\uab47\uab48\u1d26\u2c85\u0433\uab81\U0001d216\u211b\u211c\u211d\U0001cce7" + "\U0001d411\U0001d445\U0001d479\U0001d4e1\U0001d57d\U0001d5b1\U0001d5e5\U0001d619\U0001d64d\U0001d681\u01a6" + "\u13a1\u13d2\U000104b4\u1587\ua4e3\U00016f35\u027d\u027c\u024d\u0493\u1d72", + ), + ( + "s", + "\uff53\U0001d42c\U0001d460\U0001d494\U0001d4c8\U0001d4fc\U0001d530\U0001d564\U0001d598\U0001d5cc\U0001d600" + "\U0001d634\U0001d668\U0001d69c\ua731\u01bd\u0455\u0d1f\uabaa\U000118c1\U00010448\uff33\U0001cce8\U0001d412" + "\U0001d446\U0001d47a\U0001d4ae\U0001d4e2\U0001d516\U0001d54a\U0001d57e\U0001d5b2\U0001d5e6\U0001d61a" + "\U0001d64e\U0001d682\u0405\u054f\u13d5\u13da\ua4e2\U00016f3a\U00010296\U00010420\u0282\u1d74", + ), + ( + "t", + "\U0001d42d\U0001d461\U0001d495\U0001d4c9\U0001d4fd\U0001d531\U0001d565\U0001d599\U0001d5cd\U0001d601" + "\U0001d635\U0001d669\U0001d69d\u22a4\u27d9\U0001f768\uff34\U0001cce9\U0001d413\U0001d447\U0001d47b" + "\U0001d4af\U0001d4e3\U0001d517\U0001d54b\U0001d57f\U0001d5b3\U0001d5e7\U0001d61b\U0001d64f\U0001d683\u03a4" + "\U0001d6bb\U0001d6f5\U0001d72f\U0001d769\U0001d7a3\u2ca6\u0422\u13a2\ua4d4\U00016f0a\U000118bc\U00010297" + "\U000102b1\U00010315\u01ad\u2361\u023e\u021a\u01ae\u04ac\u20ae\u0167\u0166\u1d75\u0163\u021b", + ), + ( + "u", + "\U0001d42e\U0001d462\U0001d496\U0001d4ca\U0001d4fe\U0001d532\U0001d566\U0001d59a\U0001d5ce\U0001d602" + "\U0001d636\U0001d66a\U0001d69e\ua79f\u1d1c\uab4e\uab52\u028b\u03c5\U0001d6d6\U0001d710\U0001d74a\U0001d784" + "\U0001d7be\u057d\U000104f6\U000118d8\u222a\u22c3\U0001ccea\U0001d414\U0001d448\U0001d47c\U0001d4b0" + "\U0001d4e4\U0001d518\U0001d54c\U0001d580\U0001d5b4\U0001d5e8\U0001d61c\U0001d650\U0001d684\u054d\u1200" + "\U000104ce\u144c\ua4f4\U00016f42\U000118b8\u01d4\u01d3\u045f\u1d7e\uab9c\u0244\u13cc", + ), + ( + "v", + "\u2228\u22c1\uff56\u2174\U0001d42f\U0001d463\U0001d497\U0001d4cb\U0001d4ff\U0001d533\U0001d567\U0001d59b" + "\U0001d5cf\U0001d603\U0001d637\U0001d66b\U0001d69f\u1d20\u03bd\U0001d6ce\U0001d708\U0001d742\U0001d77c" + "\U0001d7b6\u0475\u05d8\U00011706\uaba9\U000118c0\U0001d20d\u0667\u06f7\u2164\U0001cceb\U0001d415\U0001d449" + "\U0001d47d\U0001d4b1\U0001d4e5\U0001d519\U0001d54d\U0001d581\U0001d5b5\U0001d5e9\U0001d61d\U0001d651" + "\U0001d685\u0474\u2d38\u13d9\u142f\ua6df\ua4e6\U00016f08\U000118a0\U0001051d\U00010197\U0001f708", + ), + ( + "w", + "\u026f\U0001d430\U0001d464\U0001d498\U0001d4cc\U0001d500\U0001d534\U0001d568\U0001d59c\U0001d5d0\U0001d604" + "\U0001d638\U0001d66c\U0001d6a0\u1d21\u2cbd\u0461\u0448\u051d\u0561\U0001170a\U0001170e\U0001170f\uab83" + "\U000118e6\U000118ef\U0001ccec\U0001d416\U0001d44a\U0001d47e\U0001d4b2\U0001d4e6\U0001d51a\U0001d54e" + "\U0001d582\U0001d5b6\U0001d5ea\U0001d61e\U0001d652\U0001d686\u051c\u13b3\u13d4\ua4ea\u047d\U000114c5\u20a9" + "\ua761\U0001d222\u13c7\u15ef\u047c\u2cbc", + ), + (" b", "\u147e\u1480\u0181"), + (" c", "\u2103"), + (" d", "\u147a\u018a"), + (" l", "\u14b6"), + (" n", "\u1459\u0149"), + (" p", "\u1476\u01a4"), + (" t", "\u01ac"), + (" u", "\u1457"), + (" v", "\u143a"), + ("av", "\ua739\ua73b\ua738\ua73a"), + ("b ", "\u147f\u1481\u1488"), + ("bl", "\u042b"), + ("c ", "\u0187"), + ("d ", "\u147b\u1487"), + ("k ", "\u0198"), + ("l ", "\u0140\u013f\u14b7\U0001f102\u2488\u05f1"), + ("ll", "\u2016\u2225\u2161\u01c1\u05f0\U00010199"), + ("n ", "\u145a\u1468"), + ("no", "\u2116"), + ("o ", "\U0001f101\U0001f100\u13a4"), + ("p ", "\u1477\u1486"), + ("r ", "\u0491"), + ( + "rn", + "\uff2d\u216f\u2133\U0001cce2\U0001d40c\U0001d440\U0001d474\U0001d4dc\U0001d510\U0001d544\U0001d578" + "\U0001d5ac\U0001d5e0\U0001d614\U0001d648\U0001d67c\u039c\U0001d6b3\U0001d6ed\U0001d727\U0001d761\U0001d79b" + "\u03fa\u2c98\u041c\u13b7\u15f0\u16d6\ua4df\U000102b0\U00010311\u04cd\U000118e3m\u217f\U0001d426\U0001d45a" + "\U0001d48e\U0001d4c2\U0001d4f6\U0001d52a\U0001d55e\U0001d592\U0001d5c6\U0001d5fa\U0001d62e\U0001d662" + "\U0001d696\U00011700\u20a5\u0271\u1d6f\u1e43", + ), + ("u ", "\u1458\u1467"), + ("v ", "\u143b"), + ("w ", "\u18ed"), + (" a ", "\u249c\U0001f110"), + (" b ", "\u249d\U0001f111"), + (" c ", "\u249e\U0001f112"), + (" d ", "\u249f\U0001f113"), + (" e ", "\u24a0\U0001f114"), + (" i ", "\u24a4"), + (" k ", "\u24a6\U0001f11a"), + (" l ", "\u2474\U0001f118\u24a7\U0001f11b"), + (" n ", "\u24a9\U0001f11d"), + (" o ", "\u24aa\U0001f11e"), + (" p ", "\u24ab\U0001f11f"), + (" r ", "\u24ad\U0001f121"), + (" s ", "\u24ae\U0001f122\U0001f12a"), + (" t ", "\u24af\U0001f123"), + (" u ", "\u24b0\U0001f124"), + (" v ", "\u24b1\U0001f125"), + (" w ", "\u24b2\U0001f126"), + ("ll ", "\u2492"), + (" ll ", "\u247e"), + (" rn ", "\U0001f11c\u24a8"), +) +_CONFUSABLE_PROTOTYPES = { + character: prototype for prototype, characters in _CONFUSABLE_PROTOTYPE_GROUPS for character in characters +} +if len(_CONFUSABLE_PROTOTYPES) != UNICODE_CONFUSABLES_SUBSET_SOURCE_COUNT: + raise RuntimeError("the pinned Unicode confusable source table is incomplete or contains duplicate entries") + + +def _pinned_category(character: str) -> str: + """Return a Unicode 15.1 category on every supported Python runtime.""" + codepoint = ord(character) + # Unicode 15.1 added four ideographic-description symbols, one additional + # symbol, and CJK Extension I. Python 3.12 ships UCD 15.0 and otherwise + # reports these assigned characters as Cn. + # Sources: https://www.unicode.org/Public/15.1.0/ucd/DerivedAge.txt and + # https://www.unicode.org/Public/15.1.0/ucd/extracted/DerivedGeneralCategory.txt + if 0x2EBF0 <= codepoint <= 0x2EE5D: + return "Lo" + if 0x2FFC <= codepoint <= 0x2FFF or codepoint == 0x31EF: + return "So" + return unicodedata.category(character) + + +def _safe_text(value: object) -> str: + if isinstance(value, str): + return value.encode("utf-8", errors="replace").decode("utf-8") + if isinstance(value, bool): + return "" + if isinstance(value, int): + return str(value) if value.bit_length() <= 256 else "" + if isinstance(value, float) and math.isfinite(value): + return str(value) + return "" + + +def publication_semantic_text(value: object, *, strip_marks: bool = False) -> str: + """Normalize public text and remove invisible identity-spoofing characters.""" + normalization_form = "NFKD" if strip_marks else "NFKC" + text = unicodedata.normalize(normalization_form, _safe_text(value)) + semantic: list[str] = [] + for character in text: + category = _pinned_category(character) + if category[0] == "C" or character in _INVISIBLE_IDENTITY_CHARACTERS or (strip_marks and category[0] == "M"): + continue + if strip_marks and (category == "Pd" or character in {"\u2043", "\u2212"}): + semantic.append("-") + else: + semantic.append(_SECURITY_TEXT_CONFUSABLES.get(character, character) if strip_marks else character) + return "".join(semantic) + + +def publication_confusable_skeleton(value: object) -> str: + """Return the pinned UTS #39 skeleton subset needed by reserved identities.""" + # The embedded values already contain the effective result of the UTS #39 + # NFD-first generation algorithm. Look up the original code point before + # host normalization so newer pinned sources survive an older runtime UCD. + mapped = "".join(_CONFUSABLE_PROTOTYPES.get(character, character) for character in _safe_text(value)) + # Identity placeholders are case-insensitive. Fold after the pinned map, + # then close over characters such as ASCII M whose folded form is itself a + # source. Mapping first preserves Unicode 17 sources unknown to the host. + folded = unicodedata.normalize("NFD", mapped).casefold() + remapped = "".join(_CONFUSABLE_PROTOTYPES.get(character, character) for character in folded) + return unicodedata.normalize("NFD", remapped).casefold() + + +def _identity_key(value: object) -> str: + skeleton = publication_confusable_skeleton(value) + security_text = publication_semantic_text(skeleton, strip_marks=True) + words = "".join(character if _pinned_category(character)[0] in {"L", "N"} else " " for character in security_text) + return " ".join(words.split()).casefold() + + +_RESERVED_IDENTITY_SKELETONS = frozenset(_identity_key(identity) for identity in _RESERVED_IDENTITIES) + + +def publication_identity_present(value: object) -> bool: + """Return whether identity text records non-placeholder provenance.""" + if not isinstance(value, str): + return False + identity = _identity_key(value) + return bool(identity and identity not in _RESERVED_IDENTITY_SKELETONS) + + +__all__ = [ + "UNICODE_CONFUSABLES_GENERATOR_UCD_VERSION", + "UNICODE_CONFUSABLES_SOURCE_SHA256", + "UNICODE_CONFUSABLES_SUBSET_SOURCE_COUNT", + "UNICODE_CONFUSABLES_VERSION", + "publication_confusable_skeleton", + "publication_identity_present", + "publication_semantic_text", +] diff --git a/src/skillevaluator/reporting/base.py b/src/skillevaluator/reporting/base.py index c199274b..e014e122 100644 --- a/src/skillevaluator/reporting/base.py +++ b/src/skillevaluator/reporting/base.py @@ -14,19 +14,36 @@ from __future__ import annotations +import hashlib +import json +import math import os +import re import secrets import stat import tempfile +import unicodedata from abc import ABC, abstractmethod from contextlib import suppress +from dataclasses import dataclass +from datetime import UTC, datetime, timedelta from pathlib import Path -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Any import click +from skillevaluator.constants import ( + DIMENSION_MAPPING, + DIMENSION_VERDICT_NEUTRAL_THRESHOLD, + DIMENSION_VERDICT_PASS_THRESHOLD, +) +from skillevaluator.publication_evidence import result_publication_evidence +from skillevaluator.publication_identity import PUBLICATION_TARGET_DIGEST_ALGORITHM +from skillevaluator.publication_text import publication_identity_present, publication_semantic_text from skillevaluator.utils.path_security import canonicalize_trusted_root_alias +_AGENT_EVAL_MAX_FUTURE_CLOCK_SKEW = timedelta(minutes=5) + if TYPE_CHECKING: from skillevaluator.models import ValidationResult @@ -51,6 +68,42 @@ and os.stat in os.supports_follow_symlinks ) +# Reporters share these limits so a payload that would be truncated in a +# self-contained artifact cannot receive a different publication verdict in a +# text-only artifact. +AGENT_EVAL_REPORT_MAX_DEPTH = 64 +AGENT_EVAL_REPORT_MAX_NODES = 100_000 +AGENT_EVAL_REPORT_MAX_TEXT_BYTES = 2 * 1024 * 1024 +_AGENT_EVAL_FINGERPRINT_MAX_DEPTH = 8 +_AGENT_EVAL_FINGERPRINT_MAX_NODES = 256 +_AGENT_EVAL_FINGERPRINT_MAX_MAPPING_ITEMS = 64 +_AGENT_EVAL_FINGERPRINT_MAX_COLLECTION_ITEMS = 64 +_AGENT_EVAL_FINGERPRINT_MAX_TEXT_CHARS = 4096 +_AGENT_EVAL_FINGERPRINT_PRIORITY_KEYS = ( + "schema_version", + "verdict", + "execution_status", + "summary", + "skill_name", + "publication_target", + "run_id", + "evaluated_at", + "evaluator_version", + "dataset_digest", + "dataset_digest_algorithm", + "benchmark_policy", + "attempt_policy", + "dataset_summary", + "agents", +) +_AGENT_EVAL_DATASET_DIGEST = re.compile(r"sha256:[0-9a-f]{64}", flags=re.IGNORECASE) +_AGENT_EVAL_DATASET_DIGEST_ALGORITHM = "skill-evaluator-dataset-snapshot/1" +_PUBLICATION_TARGET_DIGEST = re.compile(r"sha256:[0-9a-f]{64}", flags=re.IGNORECASE) +_AGENT_EVAL_RUN_ID = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,159}") +_PUBLICATION_TARGET_SKILL_NAME_MAX_BYTES = 1024 +_PUBLICATION_TARGET_CONFLICT_MAX_BYTES = 256 +_PUBLICATION_TARGET_CONFLICT_FALLBACK = "publication target identity conflict" + class UnsafeReportPathError(click.ClickException, ValueError): """Raised when a report output path cannot be written without following links.""" @@ -348,18 +401,1617 @@ def is_advisory_agent_eval_skip(result: ValidationResult) -> bool: gating = result.metadata.get("gating") if isinstance(result.metadata, dict) else None if isinstance(gating, dict) and gating.get("blocking", False): return False - payload = result.metadata.get("agent_eval", {}) if result.metadata else {} - provenance = payload.get("provenance", {}) if isinstance(payload, dict) else {} + raw_payload = result.metadata.get("agent_eval") if isinstance(result.metadata, dict) else None + payload = raw_payload if isinstance(raw_payload, dict) else {} + provenance = payload.get("provenance", {}) + verdict, execution_status, truth_consistent = _agent_eval_truth_state(payload) + return bool( + isinstance(provenance, dict) + and provenance.get("advisory") + and provenance.get("reason") == "skipped" + and truth_consistent + and execution_status == "skipped" + and verdict in {"neutral", "skipped"} + ) + + +def is_cleanly_skipped(result: ValidationResult) -> bool: + """Return whether a result records a non-failing skipped execution.""" + metadata_skip = isinstance(result.metadata, dict) and bool(result.metadata.get("skipped")) + if result.passed and metadata_skip: + raw_payload = result.metadata.get("agent_eval") + if isinstance(raw_payload, dict): + verdict, execution_status, truth_consistent = _agent_eval_truth_state(raw_payload) + return bool(truth_consistent and execution_status == "skipped" and verdict in {"neutral", "skipped"}) + return True + return is_advisory_agent_eval_skip(result) + + +def get_skip_reason(result: ValidationResult) -> str: + """Return a stable human-readable reason for a skipped result.""" + metadata = result.metadata if isinstance(result.metadata, dict) else {} + reason = metadata.get("skip_reason") + if reason: + return str(reason) + + payload = metadata.get("agent_eval") + provenance = payload.get("provenance") if isinstance(payload, dict) else None + advisory_message = provenance.get("message") if isinstance(provenance, dict) else None + if advisory_message: + return str(advisory_message) + + if result.warnings: + return str(result.warnings[0]) + return "Prerequisite unavailable" + + +def is_tier2_validator_name(validator_name: str | None) -> bool: + """Return whether a normalized validator name belongs to Tier 2.""" + normalized = " ".join((validator_name or "").casefold().replace("_", " ").replace("-", " ").split()) + return any(marker in normalized for marker in ("similarity", "dedup", "context optimization")) + + +def agent_eval_report_serialization_limits() -> tuple[int, int]: + """Return the shared depth and node limits for embedded Tier 3 payloads.""" + return AGENT_EVAL_REPORT_MAX_DEPTH, AGENT_EVAL_REPORT_MAX_NODES + + +def agent_eval_report_text_limit() -> int: + """Return the aggregate UTF-8 text budget for embedded Tier 3 payloads.""" + return AGENT_EVAL_REPORT_MAX_TEXT_BYTES + + +def agent_eval_report_serialization_issue(value: object) -> str | None: + """Return why Tier 3 cannot be emitted losslessly within report limits.""" + if not isinstance(value, dict): + return "The Tier 3 payload is not a mapping." + if isinstance(value, dict) and value.get("_serialization_truncated") is True: + return "The emitted Tier 3 payload was truncated." + + max_depth, max_nodes = agent_eval_report_serialization_limits() + max_text_bytes = agent_eval_report_text_limit() + active_containers: set[int] = set() + stack: list[tuple[object, int, bool]] = [(value, 0, False)] + nodes = 0 + text_bytes = 0 + while stack: + current, depth, exiting = stack.pop() + if exiting: + active_containers.remove(id(current)) + continue + + nodes += 1 + if nodes > max_nodes: + return f"The Tier 3 payload exceeds the {max_nodes:,}-node report limit." + if depth > max_depth: + return f"The Tier 3 payload exceeds the {max_depth}-level report depth limit." + if current is None or isinstance(current, bool): + continue + if isinstance(current, str): + if len(current) > max_text_bytes - text_bytes: + return f"The Tier 3 payload exceeds the {max_text_bytes:,}-byte report text limit." + utf8_safe = current.encode("utf-8", errors="replace").decode("utf-8") + normalized = publication_semantic_text(current) + if utf8_safe != current or normalized != unicodedata.normalize("NFKC", utf8_safe): + return "The Tier 3 payload contains text that cannot be emitted losslessly." + text_bytes += len(normalized.encode("utf-8")) + if text_bytes > max_text_bytes: + return f"The Tier 3 payload exceeds the {max_text_bytes:,}-byte report text limit." + continue + if isinstance(current, int): + if current.bit_length() > 256: + return "The Tier 3 payload contains an oversized integer." + continue + if isinstance(current, float): + if not math.isfinite(current): + return "The Tier 3 payload contains a non-finite number." + continue + if isinstance(current, tuple): + return "The Tier 3 payload contains a tuple that would be normalized to a list." + if not isinstance(current, (dict, list)): + return "The Tier 3 payload contains a value that is not JSON-compatible." + if len(current) > max_nodes - nodes: + return f"The Tier 3 payload exceeds the {max_nodes:,}-node report limit." + + container_id = id(current) + if container_id in active_containers: + return "The Tier 3 payload contains a recursive container." + active_containers.add(container_id) + stack.append((current, depth, True)) + if isinstance(current, dict): + safe_keys: set[str] = set() + for key in current: + if not isinstance(key, str): + return "The Tier 3 payload contains a non-string mapping key." + if len(key) > max_text_bytes - text_bytes: + return f"The Tier 3 payload exceeds the {max_text_bytes:,}-byte report text limit." + utf8_safe_key = key.encode("utf-8", errors="replace").decode("utf-8") + safe_key = publication_semantic_text(key) + if utf8_safe_key != key or safe_key != unicodedata.normalize("NFKC", utf8_safe_key): + return "The Tier 3 payload contains a mapping key that cannot be emitted losslessly." + if not safe_key or safe_key in safe_keys: + return "The Tier 3 payload contains colliding normalized mapping keys." + text_bytes += len(safe_key.encode("utf-8")) + if text_bytes > max_text_bytes: + return f"The Tier 3 payload exceeds the {max_text_bytes:,}-byte report text limit." + safe_keys.add(safe_key) + children = current.values() + else: + children = current + for child in reversed(children): + stack.append((child, depth + 1, False)) + return None + + +def _agent_eval_finite_number(value: object) -> float | None: + if isinstance(value, bool) or not isinstance(value, (int, float)): + return None + try: + number = float(value) + except (OverflowError, ValueError): + return None + return number if math.isfinite(number) else None + + +def _agent_eval_finite_score(value: object) -> float | None: + number = _agent_eval_finite_number(value) + return number if number is not None and 0.0 <= number <= 1.0 else None + + +def _agent_eval_count(value: object) -> int: + if isinstance(value, bool) or not isinstance(value, int): + return 0 + return value if 0 <= value <= 2**63 - 1 else 0 + + +def _agent_eval_consistent_text_field(payload: dict[str, Any] | None, key: str) -> str | None: + """Return a duplicated text field only when every persisted value agrees.""" + if not isinstance(payload, dict): + return None + summary = payload.get("summary") if isinstance(payload.get("summary"), dict) else {} + values: list[str] = [] + for container in (payload, summary): + if key not in container: + continue + value = container[key] + if ( + not isinstance(value, str) + or not value + or len(value) > AGENT_EVAL_REPORT_MAX_TEXT_BYTES + or value != value.strip() + ): + return None + values.append(value) + if not values or len(set(values)) != 1: + return None + return values[0] + + +def _agent_eval_consistent_count_field(payload: dict[str, Any] | None, key: str) -> int | None: + """Return a duplicated non-negative count only when persisted values agree.""" + if not isinstance(payload, dict): + return None + summary = payload.get("summary") if isinstance(payload.get("summary"), dict) else {} + values: list[int] = [] + for container in (payload, summary): + if key not in container: + continue + value = container[key] + if isinstance(value, bool) or not isinstance(value, int) or _agent_eval_count(value) != value: + return None + values.append(value) + if not values or len(set(values)) != 1: + return None + return values[0] + + +def _agent_eval_truth_state(payload: dict[str, Any] | None) -> tuple[str, str, bool]: + """Return conservative Tier 3 truth plus whether duplicated claims agree.""" + if not isinstance(payload, dict): + return "", "", False + summary = payload.get("summary") if isinstance(payload.get("summary"), dict) else {} + + def collect(key: str, allowed: frozenset[str], rank: dict[str, int]) -> tuple[str, bool]: + values: list[str] = [] + valid = True + for container in (payload, summary): + if key not in container: + continue + raw_value = container[key] + if not isinstance(raw_value, str) or not 1 <= len(raw_value) <= 16 or raw_value != raw_value.strip(): + valid = False + continue + value = raw_value.casefold() + if value not in allowed: + valid = False + continue + values.append(value) + if not values: + return "", False + conservative = max(values, key=lambda value: rank[value]) + return conservative, valid and len(set(values)) == 1 + + verdict, verdict_consistent = collect( + "verdict", + frozenset({"pass", "neutral", "fail", "skipped", "incomplete"}), + {"pass": 0, "skipped": 1, "neutral": 2, "incomplete": 3, "fail": 4}, + ) + execution_status, execution_consistent = collect( + "execution_status", + frozenset({"succeeded", "skipped", "incomplete", "failed"}), + {"succeeded": 0, "skipped": 1, "incomplete": 2, "failed": 3}, + ) + return verdict, execution_status, verdict_consistent and execution_consistent + + +def _agent_eval_rejected_truth_state(payload: dict[str, Any]) -> tuple[str, str, bool, int]: + """Read only bounded root truth fields from a rejected Tier 3 payload.""" + summary = payload.get("summary") if isinstance(payload.get("summary"), dict) else {} + + def collect(key: str, allowed: frozenset[str], rank: dict[str, int]) -> str: + values: list[str] = [] + for container in (payload, summary): + raw_value = container.get(key) + if not isinstance(raw_value, str) or not 1 <= len(raw_value) <= 16: + continue + if raw_value != raw_value.strip(): + continue + value = raw_value.casefold() + if value in allowed: + values.append(value) + return max(values, key=lambda value: rank[value]) if values else "" + + verdict = collect( + "verdict", + frozenset({"pass", "neutral", "fail", "skipped", "incomplete"}), + {"pass": 0, "skipped": 1, "neutral": 2, "incomplete": 3, "fail": 4}, + ) + execution_status = collect( + "execution_status", + frozenset({"succeeded", "skipped", "incomplete", "failed"}), + {"succeeded": 0, "skipped": 1, "incomplete": 2, "failed": 3}, + ) + explicit_failure = verdict == "fail" or execution_status == "failed" + conservative_outcome_rank = {"pass": 0, "neutral": 1, "incomplete": 2, "fail": 3}.get(verdict, -1) + if execution_status == "failed": + conservative_outcome_rank = max(conservative_outcome_rank, 3) + elif execution_status == "incomplete": + conservative_outcome_rank = max(conservative_outcome_rank, 2) + return verdict, execution_status, explicit_failure, conservative_outcome_rank + + +def _agent_eval_evaluated_at_datetime(value: object) -> datetime | None: + """Parse one timezone-aware evaluation instant within bounded clock skew.""" + timestamp = _agent_eval_safe_text(value) + if not timestamp or timestamp != timestamp.strip(): + return None + try: + parsed = datetime.fromisoformat(timestamp) + if parsed.tzinfo is None or parsed.utcoffset() is None: + return None + normalized = parsed.astimezone(UTC) + if normalized > datetime.now(UTC) + _AGENT_EVAL_MAX_FUTURE_CLOCK_SKEW: + return None + except (OverflowError, OSError, ValueError): + return None + return normalized + + +def agent_eval_publication_evaluated_at(payload: dict[str, Any] | None) -> str | None: + """Return a canonical ISO evaluation timestamp, rejecting shaped text.""" + value = _agent_eval_consistent_text_field(payload, "evaluated_at") + if value is None or _agent_eval_evaluated_at_datetime(value) is None: + return None + return value + + +def agent_eval_publication_dataset_provenance( + payload: dict[str, Any] | None, +) -> tuple[str, str] | None: + """Return canonical digest provenance required by public benchmark cards.""" + digest = _agent_eval_consistent_text_field(payload, "dataset_digest") or "" + algorithm = _agent_eval_consistent_text_field(payload, "dataset_digest_algorithm") or "" + if not _AGENT_EVAL_DATASET_DIGEST.fullmatch(digest): + return None + if algorithm != _AGENT_EVAL_DATASET_DIGEST_ALGORITHM: + return None + return digest, algorithm + + +def _agent_eval_safe_text(value: object) -> str: + if isinstance(value, str): + return value.encode("utf-8", errors="replace").decode("utf-8") + if isinstance(value, bool): + return "" + if isinstance(value, int): + return str(value) if value.bit_length() <= 256 else "" + if isinstance(value, float) and math.isfinite(value): + return str(value) + return "" + + +@dataclass(frozen=True) +class PublicationTargetIdentity: + """Canonical source identity shared by every publication-required tier.""" + + skill_name: str + skill_key: str + skill_digest: str + skill_digest_algorithm: str + + +def _publication_target_identity(value: object) -> PublicationTargetIdentity | None: + if not isinstance(value, dict): + return None + raw_skill_name = value.get("skill_name") + raw_skill_digest = value.get("skill_digest") + raw_algorithm = value.get("skill_digest_algorithm") + if ( + not isinstance(raw_skill_name, str) + or not isinstance(raw_skill_digest, str) + or not isinstance(raw_algorithm, str) + ): + return None + if ( + len(raw_skill_name) > _PUBLICATION_TARGET_SKILL_NAME_MAX_BYTES + or len(raw_skill_digest) != 71 + or raw_algorithm != PUBLICATION_TARGET_DIGEST_ALGORITHM + ): + return None + skill_name = _agent_eval_safe_text(raw_skill_name) + if unicodedata.normalize("NFC", skill_name) != skill_name or not publication_identity_present(skill_name): + return None + if len(skill_name.encode("utf-8")) > _PUBLICATION_TARGET_SKILL_NAME_MAX_BYTES: + return None + if not _PUBLICATION_TARGET_DIGEST.fullmatch(raw_skill_digest): + return None + return PublicationTargetIdentity( + skill_name=skill_name, + skill_key=skill_name, + skill_digest=raw_skill_digest.casefold(), + skill_digest_algorithm=raw_algorithm, + ) + + +def _result_publication_target(result: ValidationResult) -> PublicationTargetIdentity | None: + metadata = result.metadata if isinstance(result.metadata, dict) else {} + if _result_has_publication_target_conflict(result): + return None + return _publication_target_identity(metadata.get("publication_target")) + + +def publication_target_dict(value: object) -> dict[str, str] | None: + """Project an untrusted target claim to its canonical three fields.""" + identity = _publication_target_identity(value) + if identity is None: + return None + return { + "skill_name": identity.skill_name, + "skill_digest": identity.skill_digest, + "skill_digest_algorithm": identity.skill_digest_algorithm, + } + + +def result_publication_target_dict(result: ValidationResult) -> dict[str, str] | None: + """Return a fresh, canonical three-field result target safe for output.""" + metadata = result.metadata if isinstance(result.metadata, dict) else {} + if _result_has_publication_target_conflict(result): + return None + return publication_target_dict(metadata.get("publication_target")) + + +def publication_target_conflict_marker(value: object) -> str: + """Flatten an untrusted conflict claim to one bounded, printable line.""" + if not isinstance(value, str): + return _PUBLICATION_TARGET_CONFLICT_FALLBACK + if len(value) > _PUBLICATION_TARGET_CONFLICT_MAX_BYTES: + return _PUBLICATION_TARGET_CONFLICT_FALLBACK + single_line = "".join(" " if character.isspace() else character for character in value) + normalized = " ".join(publication_semantic_text(single_line).split()) + if not normalized or len(normalized.encode("utf-8")) > _PUBLICATION_TARGET_CONFLICT_MAX_BYTES: + return _PUBLICATION_TARGET_CONFLICT_FALLBACK + return normalized + + +def result_publication_target_conflict_marker(result: ValidationResult) -> str | None: + """Return a safe conflict marker when a producer persisted that field.""" + metadata = result.metadata if isinstance(result.metadata, dict) else {} + if "publication_target_conflict" not in metadata: + return None + return publication_target_conflict_marker(metadata.get("publication_target_conflict")) + + +def _result_has_publication_target_conflict(result: ValidationResult) -> bool: + metadata = result.metadata if isinstance(result.metadata, dict) else {} + payload = metadata.get("agent_eval") + return "publication_target_conflict" in metadata or _agent_eval_has_publication_target_conflict( + payload if isinstance(payload, dict) else None + ) + + +def _agent_eval_has_publication_target_conflict(payload: dict[str, Any] | None) -> bool: + if not isinstance(payload, dict): + return False + summary = payload.get("summary") if isinstance(payload.get("summary"), dict) else {} + return "publication_target_conflict" in payload or "publication_target_conflict" in summary + + +def agent_eval_publication_target(payload: dict[str, Any] | None) -> PublicationTargetIdentity | None: + """Return the duplicated Tier 3 target claim only when both copies agree.""" + if not isinstance(payload, dict): + return None + summary = payload.get("summary") if isinstance(payload.get("summary"), dict) else {} + top_level = _publication_target_identity(payload.get("publication_target")) + summarized = _publication_target_identity(summary.get("publication_target")) + if top_level is None or summarized is None or top_level != summarized: + return None + return top_level + + +def agent_eval_publication_run_id(payload: dict[str, Any] | None) -> str | None: + """Return the exact bounded Tier 3 run identity duplicated in its summary.""" + if not isinstance(payload, dict): + return None + summary = payload.get("summary") + if not isinstance(summary, dict): + return None + run_id = payload.get("run_id") + summarized_run_id = summary.get("run_id") + if ( + not isinstance(run_id, str) + or not isinstance(summarized_run_id, str) + or not 1 <= len(run_id) <= 160 + or run_id != summarized_run_id + or _AGENT_EVAL_RUN_ID.fullmatch(run_id) is None + ): + return None + return run_id + + +def _selected_publication_target( + results: list[ValidationResult], + payload: dict[str, Any] | None, + expected_skill_name: str | None, +) -> PublicationTargetIdentity | None: + """Select one canonical target without accepting anonymous policy claims.""" + expected_key = _publication_identity_key(expected_skill_name) + payload_target = agent_eval_publication_target(payload) + if payload_target is not None and ( + expected_key is None or _publication_identity_key(payload_target.skill_name) == expected_key + ): + return payload_target + claims = { + claim + for result in results + if (claim := _result_publication_target(result)) is not None + and (is_tier3_result(result) or result_publication_evidence(result) is not None) + and (expected_key is None or _publication_identity_key(claim.skill_name) == expected_key) + } + return next(iter(claims)) if len(claims) == 1 else None + + +def publication_target_for_results( + results: list[ValidationResult], + payload: dict[str, Any] | None = None, + *, + expected_skill_name: str | None = None, +) -> PublicationTargetIdentity | None: + """Return the one canonical target shared by publication evidence.""" + selected_payload = payload if isinstance(payload, dict) else select_agent_eval_payload(results) + return _selected_publication_target(results, selected_payload, expected_skill_name) + + +def result_matches_publication_target( + result: ValidationResult, + target: PublicationTargetIdentity | None, +) -> bool: + """Return whether one result is bound to the selected exact source.""" + return target is not None and _result_publication_target(result) == target + + +def _agent_eval_agents(payload: dict[str, Any] | None) -> dict[str, dict[str, Any]]: + raw_agents = (payload or {}).get("agents") + if not isinstance(raw_agents, dict): + return {} + agents: dict[str, dict[str, Any]] = {} + normalized_names: set[str] = set() + for raw_name, raw_agent in raw_agents.items(): + if not isinstance(raw_name, str) or not publication_identity_present(raw_name): + return {} + name = publication_semantic_text(raw_name).strip() + normalized_name = name.casefold() + if not publication_identity_present(name) or not isinstance(raw_agent, dict): + # A malformed peer must not disappear while another agent certifies + # the same run. Treat the whole agent set as unusable evidence. + return {} + if normalized_name in normalized_names: + return {} + normalized_names.add(normalized_name) + agents[name] = raw_agent + return agents + + +def _agent_eval_attempt_coverage_complete(payload: dict[str, Any] | None) -> bool: + """Return whether succeeded Tier 3 evidence proves positive attempt coverage.""" + expected_attempts = _agent_eval_consistent_count_field(payload, "expected_attempts") + scored_attempts = _agent_eval_consistent_count_field(payload, "scored_attempts") + if ( + expected_attempts is None + or scored_attempts is None + or expected_attempts <= 0 + or scored_attempts <= 0 + or scored_attempts > expected_attempts + ): + return False + + scored_agents = 0 + summed_expected = 0 + summed_scored = 0 + for agent in _agent_eval_agents(payload).values(): + raw_expected = agent.get("expected_attempts") + raw_scored = agent.get("scored_attempts") + if ( + isinstance(raw_expected, bool) + or not isinstance(raw_expected, int) + or isinstance(raw_scored, bool) + or not isinstance(raw_scored, int) + ): + return False + agent_expected = _agent_eval_count(agent.get("expected_attempts")) + agent_scored = _agent_eval_count(agent.get("scored_attempts")) + if agent_scored > agent_expected: + return False + summed_expected += agent_expected + summed_scored += agent_scored + if _agent_eval_safe_text(agent.get("execution_status")).casefold() != "succeeded": + continue + if agent_expected <= 0 or agent_scored <= 0: + return False + scored_agents += 1 + return bool(scored_agents > 0 and summed_expected == expected_attempts and summed_scored == scored_attempts) + + +def _agent_eval_dimension_scores(agent: dict[str, Any]) -> list[float] | None: + raw_dimensions = agent.get("dimensions") + if not isinstance(raw_dimensions, list): + return None + dimensions: dict[str, dict[str, Any]] = {} + for raw_dimension in raw_dimensions: + if not isinstance(raw_dimension, dict): + continue + dimension_id = _agent_eval_safe_text(raw_dimension.get("id")) + if not dimension_id or dimension_id in dimensions: + return None + dimensions[dimension_id] = raw_dimension + + scores: list[float] = [] + for dimension_id in DIMENSION_MAPPING: + dimension = dimensions.get(dimension_id) + if dimension is None: + return None + value = dimension.get("with_skill") if "with_skill" in dimension else dimension.get("score") + score = _agent_eval_finite_score(value) + if score is None: + return None + scores.append(score) + return scores + + +def agent_eval_dimension_verdict(payload: dict[str, Any] | None) -> str | None: + """Recompute a Tier 3 verdict from complete supported-agent dimensions.""" + supported_agents = [ + agent + for agent in _agent_eval_agents(payload).values() + if _agent_eval_safe_text(agent.get("execution_status")).lower() == "succeeded" + ] + if not supported_agents: + return None + + verdicts: list[str] = [] + has_partial_evidence = False + for agent in supported_agents: + scores = _agent_eval_dimension_scores(agent) + if scores is None: + has_partial_evidence = True + continue + if any(score < DIMENSION_VERDICT_NEUTRAL_THRESHOLD for score in scores): + verdicts.append("fail") + elif any(score < DIMENSION_VERDICT_PASS_THRESHOLD for score in scores): + verdicts.append("neutral") + else: + verdicts.append("pass") + + if "pass" in verdicts: + return "pass" + if has_partial_evidence or not verdicts: + return None + if "neutral" in verdicts: + return "neutral" + return "fail" + + +def _agent_eval_dataset_count(payload: dict[str, Any]) -> int: + summary = payload.get("dataset_summary") + if isinstance(summary, dict): + return _agent_eval_count(summary.get("total_tasks")) + dataset = payload.get("dataset") + if isinstance(dataset, list): + count = sum(isinstance(item, dict) for item in dataset) + if count: + return count + trials = payload.get("trials") + task_ids: set[str] = set() + if isinstance(trials, list): + for trial in trials: + if not isinstance(trial, dict): + continue + for key in ("entry_id", "case_id", "task_id", "id"): + task_id = _agent_eval_safe_text(trial.get(key)).strip() + if task_id: + task_ids.add(task_id) + break + return len(task_ids) + + +def agent_eval_publication_evidence_complete(payload: dict[str, Any] | None) -> bool: + """Return whether a payload carries the minimum publication Tier 3 evidence.""" + if not isinstance(payload, dict): + return False + if agent_eval_report_serialization_issue(payload) is not None: + return False + agents = _agent_eval_agents(payload) + if not agents or any( + not any(publication_identity_present(agent.get(key)) for key in ("model", "model_name", "llm_model")) + for agent in agents.values() + ): + return False + verdict, execution_status, truth_consistent = _agent_eval_truth_state(payload) + evaluator_version = _agent_eval_consistent_text_field(payload, "evaluator_version") + environment = _agent_eval_consistent_text_field(payload, "environment") + skill_name = _agent_eval_consistent_text_field(payload, "skill_name") + publication_target = agent_eval_publication_target(payload) + attempt_policy = payload.get("attempt_policy") if isinstance(payload.get("attempt_policy"), dict) else {} return bool( - isinstance(provenance, dict) and provenance.get("advisory") and provenance.get("reason") == "skipped" + truth_consistent + and verdict in {"pass", "neutral", "fail"} + and execution_status == "succeeded" + and agent_eval_dimension_verdict(payload) is not None + and agent_eval_publication_evaluated_at(payload) is not None + and publication_identity_present(evaluator_version) + and agent_eval_publication_dataset_provenance(payload) is not None + and _agent_eval_dataset_count(payload) > 0 + and _agent_eval_count(attempt_policy.get("max_attempts")) > 0 + and _agent_eval_attempt_coverage_complete(payload) + and publication_identity_present(environment) + and publication_identity_present(skill_name) + and publication_target is not None + and publication_target.skill_name == skill_name + and agent_eval_publication_run_id(payload) is not None + ) + + +def _agent_eval_fingerprint_text(value: str) -> str | dict[str, object]: + """Return a bounded canonical text projection without scanning a huge tail.""" + if len(value) > _AGENT_EVAL_FINGERPRINT_MAX_TEXT_CHARS: + prefix = value[:_AGENT_EVAL_FINGERPRINT_MAX_TEXT_CHARS] + return { + "type": "text", + "length": len(value), + "prefix": publication_semantic_text(prefix), + "truncated": True, + } + return publication_semantic_text(value) + + +def _agent_eval_fingerprint_key_rank(value: object) -> tuple[object, ...]: + """Return a bounded total ordering for the sampled keys of a malformed map.""" + if isinstance(value, str): + prefix = value[:_AGENT_EVAL_FINGERPRINT_MAX_TEXT_CHARS] + return (0, len(value), publication_semantic_text(prefix)) + if isinstance(value, bool): + return (1, int(value)) + if isinstance(value, int): + return (2, value.bit_length(), value if value.bit_length() <= 256 else 0) + if isinstance(value, float): + if math.isnan(value): + return (3, 1, "nan") + if math.isinf(value): + return (3, 1, "positive-infinity" if value > 0 else "negative-infinity") + return (3, 0, value) + value_type = type(value) + return (4, value_type.__module__, value_type.__qualname__) + + +def _agent_eval_fingerprint_key(value: object) -> object: + """Return a bounded JSON-safe description of one sampled mapping key.""" + if isinstance(value, str): + return ["text", _agent_eval_fingerprint_text(value)] + if isinstance(value, bool): + return ["bool", value] + if isinstance(value, int): + return ["int", value] if value.bit_length() <= 256 else ["oversized-int", value.bit_length()] + if isinstance(value, float): + if math.isfinite(value): + return ["float", value] + if math.isnan(value): + return ["non-finite-float", "nan"] + return ["non-finite-float", "positive-infinity" if value > 0 else "negative-infinity"] + value_type = type(value) + return ["unsupported", value_type.__module__, value_type.__qualname__] + + +def _agent_eval_bounded_fingerprint_value( + value: object, + *, + depth: int = 0, + remaining_nodes: list[int] | None = None, + active_containers: set[int] | None = None, +) -> object: + """Project malformed evidence into a small deterministic fingerprint shape.""" + remaining = remaining_nodes if remaining_nodes is not None else [_AGENT_EVAL_FINGERPRINT_MAX_NODES] + active = active_containers if active_containers is not None else set() + if remaining[0] <= 0: + return {"truncated": "node-limit"} + remaining[0] -= 1 + if depth > _AGENT_EVAL_FINGERPRINT_MAX_DEPTH: + return {"truncated": "depth-limit"} + if value is None or isinstance(value, bool): + return value + if isinstance(value, str): + return _agent_eval_fingerprint_text(value) + if isinstance(value, int): + return value if value.bit_length() <= 256 else {"oversized_integer_bits": value.bit_length()} + if isinstance(value, float): + if math.isfinite(value): + return value + if math.isnan(value): + return {"non_finite_number": "nan"} + return {"non_finite_number": "positive-infinity" if value > 0 else "negative-infinity"} + if not isinstance(value, (dict, list, tuple)): + value_type = type(value) + return {"unsupported_type": [value_type.__module__, value_type.__qualname__]} + + container_id = id(value) + if container_id in active: + return {"recursive_container": True} + active.add(container_id) + try: + if isinstance(value, dict): + priority = frozenset(_AGENT_EVAL_FINGERPRINT_PRIORITY_KEYS) + priority_keys = [key for key in _AGENT_EVAL_FINGERPRINT_PRIORITY_KEYS if key in value] + sampled_keys: list[object] = [] + max_nonpriority = max(0, _AGENT_EVAL_FINGERPRINT_MAX_MAPPING_ITEMS - len(priority_keys)) + iterator = iter(value) + attempts_remaining = max_nonpriority + len(priority_keys) + while len(sampled_keys) < max_nonpriority and attempts_remaining > 0: + attempts_remaining -= 1 + try: + key = next(iterator) + except StopIteration: + break + if key not in priority: + sampled_keys.append(key) + sampled_keys.sort(key=_agent_eval_fingerprint_key_rank) + selected_keys = [*priority_keys, *sampled_keys] + entries: list[list[object]] = [] + for key in selected_keys: + if remaining[0] <= 0: + break + try: + item = value[key] + except (KeyError, RuntimeError): + entries.append([_agent_eval_fingerprint_key(key), {"unavailable": True}]) + continue + entries.append( + [ + _agent_eval_fingerprint_key(key), + _agent_eval_bounded_fingerprint_value( + item, + depth=depth + 1, + remaining_nodes=remaining, + active_containers=active, + ), + ] + ) + return { + "type": "mapping", + "length": len(value), + "entries": entries, + "truncated": len(value) > len(selected_keys), + } + + item_limit = min(len(value), _AGENT_EVAL_FINGERPRINT_MAX_COLLECTION_ITEMS) + return { + "type": "tuple" if isinstance(value, tuple) else "list", + "length": len(value), + "items": [ + _agent_eval_bounded_fingerprint_value( + value[index], + depth=depth + 1, + remaining_nodes=remaining, + active_containers=active, + ) + for index in range(item_limit) + if remaining[0] > 0 + ], + "truncated": len(value) > item_limit, + } + finally: + active.remove(container_id) + + +def _agent_eval_payload_fingerprint( + payload: dict[str, Any], + *, + serialization_issue: str | None = None, +) -> str: + """Return a stable tie-breaker without fully serializing rejected evidence.""" + if serialization_issue is None: + serialization_issue = agent_eval_report_serialization_issue(payload) + serialized: str | None = None + if serialization_issue is None: + try: + serialized = json.dumps( + payload, + allow_nan=False, + ensure_ascii=True, + separators=(",", ":"), + sort_keys=True, + ) + except (OverflowError, RecursionError, TypeError, ValueError): + serialization_issue = "Canonical JSON serialization failed." + if serialized is None: + serialized = json.dumps( + { + "serialization_issue": serialization_issue, + "payload": _agent_eval_bounded_fingerprint_value(payload), + }, + allow_nan=False, + ensure_ascii=True, + separators=(",", ":"), + sort_keys=True, + ) + return hashlib.sha256(serialized.encode("utf-8", errors="surrogatepass")).hexdigest() + + +def _agent_eval_evaluated_at_rank(value: object) -> float: + parsed = _agent_eval_evaluated_at_datetime(value) + if parsed is None: + return float("-inf") + try: + return parsed.timestamp() + except (OverflowError, OSError, ValueError): + return float("-inf") + + +def _agent_eval_candidate_rank( + result: ValidationResult, + payload: dict[str, Any], +) -> tuple[Any, ...]: + serialization_issue = agent_eval_report_serialization_issue(payload) + if serialization_issue is not None: + verdict, execution_status, explicit_failure, conservative_outcome_rank = _agent_eval_rejected_truth_state( + payload + ) + result_evidence = len(result.success_details) + len(result.findings) + result.summary.checks_performed + priority = 3 if execution_status in {"failed", "incomplete", "skipped"} else 1 + return ( + int(explicit_failure), + priority, + conservative_outcome_rank, + float("-inf"), + result_evidence, + 0, + 0, + -1.0, + -1.0, + len(payload), + "", + "", + verdict, + _agent_eval_payload_fingerprint(payload, serialization_issue=serialization_issue), + ) + + summary = payload.get("summary") if isinstance(payload.get("summary"), dict) else {} + verdict, execution_status, truth_consistent = _agent_eval_truth_state(payload) + payload_attempts = max( + _agent_eval_count(payload.get("scored_attempts")), + _agent_eval_count(summary.get("scored_attempts")), + ) + agents = payload.get("agents") if isinstance(payload.get("agents"), dict) else {} + agent_evidence = False + strong_agent_evidence = False + for agent in agents.values(): + if not isinstance(agent, dict) or agent.get("execution_status") != "succeeded": + continue + score = _agent_eval_finite_score(agent.get("with_skill")) + if score is None: + score = _agent_eval_finite_score(agent.get("overall_score")) + dimensions = agent.get("dimensions") + dimension_evidence = isinstance(dimensions, list) and any( + isinstance(dimension, dict) + and _agent_eval_finite_score(dimension.get("with_skill", dimension.get("score"))) is not None + for dimension in dimensions + ) + if score is None and not dimension_evidence: + continue + agent_evidence = True + if max(payload_attempts, _agent_eval_count(agent.get("scored_attempts"))) > 0: + strong_agent_evidence = True + + valid_verdict = truth_consistent and verdict in {"pass", "neutral", "fail"} + dimension_verdict = agent_eval_dimension_verdict(payload) + conservative_outcome_rank = max( + ({"pass": 0, "neutral": 1, "fail": 3}.get(candidate, -1) for candidate in (verdict, dimension_verdict)), + default=-1, + ) + if execution_status == "failed": + conservative_outcome_rank = max(conservative_outcome_rank, 3) + elif execution_status == "incomplete": + conservative_outcome_rank = max(conservative_outcome_rank, 2) + explicit_failure_rank = int(verdict == "fail" or dimension_verdict == "fail" or execution_status == "failed") + result_evidence = len(result.success_details) + len(result.findings) + result.summary.checks_performed + if agent_eval_publication_evidence_complete(payload): + priority = 8 + elif execution_status == "succeeded" and valid_verdict and strong_agent_evidence: + priority = 6 + elif execution_status == "succeeded" and valid_verdict and agent_evidence: + priority = 5 + elif not execution_status and valid_verdict and result_evidence: + priority = 4 + elif execution_status in {"failed", "incomplete", "skipped"}: + priority = 3 + elif payload: + priority = 1 + else: + priority = 0 + + dataset_count = _agent_eval_dataset_count(payload) + evaluated_at = _agent_eval_safe_text(payload.get("evaluated_at") or summary.get("evaluated_at")) + evaluated_at_rank = _agent_eval_evaluated_at_rank(evaluated_at) + digest = _agent_eval_safe_text(payload.get("dataset_digest") or summary.get("dataset_digest")) + best_score = max( + ( + score + for agent in agents.values() + if isinstance(agent, dict) + and (score := _agent_eval_finite_score(agent.get("with_skill", agent.get("overall_score")))) is not None + ), + default=-1.0, ) + runtime = _agent_eval_finite_number(payload.get("runtime_seconds")) + return ( + explicit_failure_rank, + priority, + conservative_outcome_rank, + evaluated_at_rank, + result_evidence, + dataset_count, + payload_attempts, + best_score, + runtime if runtime is not None else -1.0, + len(payload), + evaluated_at, + digest, + verdict, + _agent_eval_payload_fingerprint(payload), + ) + + +def select_agent_eval_candidate( + results: list[ValidationResult], +) -> tuple[ValidationResult, dict[str, Any]] | None: + """Select the strongest Tier 3 result and payload independent of input order.""" + candidates = [ + (result, payload) + for result in results + if isinstance(result.metadata, dict) and isinstance((payload := result.metadata.get("agent_eval")), dict) + ] + if not candidates: + return None + return max( + candidates, + key=lambda candidate: _agent_eval_candidate_rank(*candidate), + ) + + +def select_agent_eval_payload(results: list[ValidationResult]) -> dict[str, Any] | None: + """Select the strongest Tier 3 payload without depending on result order.""" + selected = select_agent_eval_candidate(results) + return selected[1] if selected is not None else None + + +def resolve_benchmark_policy( + results: list[ValidationResult], + agent_eval: dict[str, Any] | None, + *, + expected_skill_name: str | None = None, +) -> dict[str, bool]: + """Resolve required publication tiers from persisted policy metadata. + + Each policy key is resolved independently. The selected canonical payload + may waive evidence only when persisted peer payloads agree; conflicts and + missing values fail closed to required evidence. + """ + selected_payload = agent_eval if isinstance(agent_eval, dict) else select_agent_eval_payload(results) + expected_target = _selected_publication_target(results, selected_payload, expected_skill_name) + has_publication_target_conflict = _agent_eval_has_publication_target_conflict(selected_payload) or any( + _result_has_publication_target_conflict(result) for result in results + ) + + def payload_matches_target(payload: dict[str, Any]) -> bool: + return bool( + expected_target is not None + and agent_eval_publication_target(payload) == expected_target + and _agent_eval_target_skill_issue(payload, expected_skill_name) is None + ) + + selected_policy_payload = ( + selected_payload if isinstance(selected_payload, dict) and payload_matches_target(selected_payload) else None + ) + payloads: list[dict[str, Any]] = [selected_policy_payload] if isinstance(selected_policy_payload, dict) else [] + foreign_payloads: list[dict[str, Any]] = ( + [selected_payload] if isinstance(selected_payload, dict) and selected_policy_payload is None else [] + ) + for result in results: + metadata = result.metadata if isinstance(result.metadata, dict) else {} + payload = metadata.get("agent_eval") + if not isinstance(payload, dict): + continue + destination = payloads if payload_matches_target(payload) else foreign_payloads + if all(payload is not candidate for candidate in destination): + destination.append(payload) + result_policies: list[object] = [] + foreign_result_policies: list[object] = [] + for result in results: + metadata = result.metadata if isinstance(result.metadata, dict) else {} + policy = metadata.get("benchmark_policy") + producer = result_publication_evidence(result) + destination = ( + result_policies + if producer is not None + and expected_target is not None + and _result_publication_target(result) == expected_target + else foreign_result_policies + ) + destination.append(policy) + + def peer_value(candidates: list[object], key: str) -> bool | None: + values = [ + value + for candidate in candidates + if isinstance(candidate, dict) and isinstance((value := candidate.get(key)), bool) + ] + if not values: + return None + return values[0] if len(set(values)) == 1 else True + + def payload_value(payload: dict[str, Any], key: str) -> bool | None: + summary = payload.get("summary") + summary_policy = summary.get("benchmark_policy") if isinstance(summary, dict) else None + values: list[bool] = [] + for policy in (payload.get("benchmark_policy"), summary_policy): + if not isinstance(policy, dict) or key not in policy: + continue + value = policy[key] + if isinstance(value, bool): + values.append(value) + if not values: + return None + return values[0] if len(set(values)) == 1 else True + + resolved: dict[str, bool] = {} + for key in ("tier2_required", "tier3_required"): + if has_publication_target_conflict: + resolved[key] = True + continue + selected_value = ( + payload_value(selected_policy_payload, key) if isinstance(selected_policy_payload, dict) else None + ) + peer_payload_values = [ + payload_value(payload, key) for payload in payloads if payload is not selected_policy_payload + ] + peer_payload_values = [value for value in peer_payload_values if value is not None] + # Foreign payloads or result producers can require evidence but can + # never waive it for this report, even across policy precedence levels. + if any(payload_value(payload, key) is True for payload in foreign_payloads): + peer_payload_values.append(True) + if peer_value(foreign_result_policies, key) is True: + peer_payload_values.append(True) + if selected_value is not None: + # The canonical payload may waive a tier only when every peer + # payload that persists the same key agrees. A weak or stale peer + # can force required evidence, but cannot create a waiver. + resolved[key] = selected_value if all(value == selected_value for value in peer_payload_values) else True + continue + if any(peer_payload_values): + resolved[key] = True + continue + + # Result objects are peers, not an ordered precedence chain. A conflict + # therefore fails closed regardless of aggregator input order. + result_value = peer_value(result_policies, key) + if peer_value(foreign_result_policies, key) is True: + result_value = True + resolved[key] = result_value if result_value is not None else True + return resolved + + +def _result_agent_eval_run_id_issue(result: ValidationResult) -> str | None: + """Reject a persisted outer Tier 3 run claim that contradicts its payload.""" + metadata = result.metadata if isinstance(result.metadata, dict) else {} + if "run_id" not in metadata: + return None + outer_run_id = metadata.get("run_id") + if ( + not isinstance(outer_run_id, str) + or not 1 <= len(outer_run_id) <= 160 + or _AGENT_EVAL_RUN_ID.fullmatch(outer_run_id) is None + ): + return "Tier 3 result contains an invalid outer run identity." + payload = metadata.get("agent_eval") + payload_run_id = agent_eval_publication_run_id(payload if isinstance(payload, dict) else None) + if payload_run_id != outer_run_id: + return "Tier 3 result contains contradictory run identities." + return None + + +@dataclass(frozen=True) +class Tier3EvidenceAssessment: + """Publication-facing interpretation of the canonical Tier 3 payload.""" + + status: str + evidence_complete: bool + execution_status: str + verdict: str + payload: dict[str, Any] | None + reason: str | None = None + + +@dataclass(frozen=True) +class PublicationAssessment: + """Publication status kept separate from a command's process exit gate.""" + + status: str + benchmark_policy: dict[str, bool] + tier3: Tier3EvidenceAssessment + reasons: tuple[str, ...] = () + + +def result_has_execution_evidence(result: ValidationResult) -> bool: + """Return whether a non-skipped validator proves that it executed work.""" + return bool(result.success_details or result.findings or result.summary.checks_performed > 0) + + +def is_tier3_result(result: ValidationResult) -> bool: + """Return whether a result belongs to live Tier 3 evaluation.""" + metadata = result.metadata if isinstance(result.metadata, dict) else {} + return result.validator_name == "AGENT_EVAL" or isinstance(metadata.get("agent_eval"), dict) + + +def is_tier2_result(result: ValidationResult) -> bool: + """Return whether a result belongs to semantic Tier 2 validation.""" + return bool( + not is_tier3_result(result) + and ( + is_tier2_validator_name(result.validator_name) + or any(finding.category == "CONTENT_DEDUP" for finding in result.findings) + ) + ) + + +def _publication_identity_key(value: object) -> str | None: + """Normalize canonical aliases without folding filesystem distinctions.""" + if not publication_identity_present(value): + return None + return unicodedata.normalize("NFC", _agent_eval_safe_text(value)) + + +def _agent_eval_target_skill_issue( + payload: dict[str, Any] | None, + expected_skill_name: str | None, + *, + require_publication_target: bool = True, +) -> str | None: + """Return a fail-closed reason when Tier 3 targets a different skill.""" + expected_key = _publication_identity_key(expected_skill_name) + if expected_skill_name is not None and expected_key is None: + return "The expected target skill identity is invalid." + + persisted_skill_name = _agent_eval_consistent_text_field(payload, "skill_name") + persisted_key = _publication_identity_key(persisted_skill_name) + if persisted_key is None: + return "Tier 3 evidence lacks a consistent target skill identity." + if expected_key is not None and persisted_key != expected_key: + return "Tier 3 evidence belongs to a different target skill." + if not require_publication_target: + return None + publication_target = agent_eval_publication_target(payload) + if publication_target is None: + return "Tier 3 evidence lacks a canonical target source identity." + if publication_target.skill_name != persisted_skill_name: + return "Tier 3 evidence contains contradictory target skill identities." + return None + + +def assess_tier3_evidence( + results: list[ValidationResult], + payload: dict[str, Any] | None = None, + *, + expected_skill_name: str | None = None, +) -> Tier3EvidenceAssessment: + """Classify Tier 3 evidence once for every reporter.""" + tier3_results = [result for result in results if is_tier3_result(result)] + selected_payload = payload if isinstance(payload, dict) else select_agent_eval_payload(tier3_results) + summary = ( + selected_payload.get("summary") + if isinstance(selected_payload, dict) and isinstance(selected_payload.get("summary"), dict) + else {} + ) + serialization_issue = ( + agent_eval_report_serialization_issue(selected_payload) if isinstance(selected_payload, dict) else None + ) + if serialization_issue is not None: + raw_verdict, execution_status, explicit_failure, _ = _agent_eval_rejected_truth_state(selected_payload) + truth_consistent = False + dimension_verdict = None + else: + raw_verdict, execution_status, truth_consistent = _agent_eval_truth_state(selected_payload) + dimension_verdict = agent_eval_dimension_verdict(selected_payload) + explicit_failure = False + + if any(result.is_incomplete for result in tier3_results): + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status or "incomplete", + raw_verdict or "incomplete", + selected_payload, + "Tier 3 scanner evidence is incomplete.", + ) + if any( + not result.passed and (serialization_issue is not None or not is_cleanly_skipped(result)) + for result in tier3_results + ): + return Tier3EvidenceAssessment( + "fail", + False, + execution_status or "failed", + raw_verdict or "fail", + selected_payload, + "A Tier 3 validator failed.", + ) + if _agent_eval_has_publication_target_conflict(selected_payload) or any( + _result_has_publication_target_conflict(result) for result in tier3_results + ): + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status or "incomplete", + raw_verdict or "incomplete", + selected_payload, + "Tier 3 source identity changed during evaluation.", + ) + if run_id_issue := next( + (issue for result in tier3_results if (issue := _result_agent_eval_run_id_issue(result)) is not None), + None, + ): + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status or "incomplete", + raw_verdict or "incomplete", + selected_payload, + run_id_issue, + ) + if serialization_issue is not None: + if explicit_failure: + return Tier3EvidenceAssessment( + "fail", + False, + execution_status or "failed", + raw_verdict or "fail", + selected_payload, + f"{serialization_issue} The rejected Tier 3 payload also records an explicit failure.", + ) + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status or "incomplete", + raw_verdict or "incomplete", + selected_payload, + f"{serialization_issue} Publication completeness cannot be proven.", + ) + if tier3_results and all(is_cleanly_skipped(result) for result in tier3_results): + if selected_payload is not None and ( + target_skill_issue := _agent_eval_target_skill_issue( + selected_payload, + expected_skill_name, + require_publication_target=False, + ) + ): + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status or "incomplete", + raw_verdict or "incomplete", + selected_payload, + target_skill_issue, + ) + return Tier3EvidenceAssessment( + "skipped", + False, + execution_status or "skipped", + raw_verdict or "skipped", + selected_payload, + "Tier 3 was skipped.", + ) + if (selected_payload is not None or tier3_results) and not truth_consistent: + if raw_verdict == "fail" or dimension_verdict == "fail" or execution_status == "failed": + return Tier3EvidenceAssessment( + "fail", + False, + execution_status or "failed", + raw_verdict or "fail", + selected_payload, + "Tier 3 contains contradictory truth fields including an explicit failure.", + ) + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status or "incomplete", + raw_verdict or "incomplete", + selected_payload, + "Tier 3 truth fields are missing, invalid, or contradictory.", + ) + if raw_verdict == "fail" or dimension_verdict == "fail" or execution_status == "failed": + return Tier3EvidenceAssessment( + "fail", + agent_eval_publication_evidence_complete(selected_payload), + execution_status or "failed", + raw_verdict or "fail", + selected_payload, + "Tier 3 evidence records a failing verdict.", + ) + if execution_status == "incomplete": + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status, + raw_verdict or "incomplete", + selected_payload, + "Tier 3 execution evidence is incomplete.", + ) + if any( + isinstance(result.metadata, dict) and bool(result.metadata.get("skipped")) and not is_cleanly_skipped(result) + for result in tier3_results + ): + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status or "incomplete", + raw_verdict or "incomplete", + selected_payload, + "Tier 3 skip metadata contradicts the persisted execution evidence.", + ) + if selected_payload is not None and ( + target_skill_issue := _agent_eval_target_skill_issue(selected_payload, expected_skill_name) + ): + return Tier3EvidenceAssessment( + "incomplete", + False, + execution_status or "incomplete", + raw_verdict or "incomplete", + selected_payload, + target_skill_issue, + ) + evidence_complete = agent_eval_publication_evidence_complete(selected_payload) + if not evidence_complete: + payload_attempts = max( + _agent_eval_count((selected_payload or {}).get("scored_attempts")), + _agent_eval_count(summary.get("scored_attempts")), + ) + has_scored_agent = any( + _agent_eval_safe_text(agent.get("execution_status")).lower() == "succeeded" + and ( + _agent_eval_finite_score(agent.get("with_skill", agent.get("overall_score"))) is not None + or bool(_agent_eval_dimension_scores(agent)) + ) + and max(payload_attempts, _agent_eval_count(agent.get("scored_attempts"))) > 0 + for agent in _agent_eval_agents(selected_payload).values() + ) + if execution_status == "succeeded" and raw_verdict in {"pass", "neutral"} and has_scored_agent: + effective_verdict = "neutral" if "neutral" in {raw_verdict, dimension_verdict} else "pass" + return Tier3EvidenceAssessment( + effective_verdict, + False, + execution_status, + raw_verdict, + selected_payload, + "Tier 3 ran but lacks publication-complete provenance or dimension evidence.", + ) + return Tier3EvidenceAssessment( + "incomplete" if tier3_results or selected_payload else "not_run", + False, + execution_status or ("incomplete" if tier3_results else "not_run"), + raw_verdict or ("incomplete" if tier3_results else "not_run"), + selected_payload, + "Tier 3 lacks publication-complete execution evidence.", + ) + + effective_verdict = "neutral" if "neutral" in {raw_verdict, dimension_verdict} else "pass" + return Tier3EvidenceAssessment( + effective_verdict, + True, + execution_status, + raw_verdict, + selected_payload, + ) + + +def assess_publication( + results: list[ValidationResult], + agent_eval: dict[str, Any] | None = None, + *, + expected_skill_name: str | None = None, +) -> PublicationAssessment: + """Return a shared, fail-closed publication assessment for all reporters.""" + payload = agent_eval if isinstance(agent_eval, dict) else select_agent_eval_payload(results) + tier3_results = [result for result in results if is_tier3_result(result)] + producer_results = [ + (result, producer) for result in results if (producer := result_publication_evidence(result)) is not None + ] + tier1_results = [result for result, producer in producer_results if producer.tier == 1] + tier2_results = [result for result, producer in producer_results if producer.tier == 2] + publication_target = _selected_publication_target(results, payload, expected_skill_name) + + peer_skill_names: dict[str, str] = {} + for result in results: + if not is_tier3_result(result) and result_publication_evidence(result) is None: + continue + metadata = result.metadata if isinstance(result.metadata, dict) else {} + candidates: list[object] = [metadata.get("skill_name"), metadata.get("target_skill_name")] + quality = metadata.get("quality_scores") + if isinstance(quality, dict): + candidates.append(quality.get("skill_name")) + quality_all = metadata.get("quality_scores_all") + if isinstance(quality_all, list): + candidates.extend(item.get("skill_name") for item in quality_all if isinstance(item, dict)) + agent_eval_payload = metadata.get("agent_eval") + if isinstance(agent_eval_payload, dict): + candidates.append(_agent_eval_consistent_text_field(agent_eval_payload, "skill_name")) + if (result_target := _result_publication_target(result)) is not None: + candidates.append(result_target.skill_name) + for candidate in candidates: + if (key := _publication_identity_key(candidate)) is not None: + peer_skill_names.setdefault(key, str(candidate)) + + explicit_expected_key = _publication_identity_key(expected_skill_name) + peer_expected_key = next(iter(peer_skill_names), None) if len(peer_skill_names) == 1 else None + tier3_expected_skill_name = ( + expected_skill_name + if expected_skill_name is not None + else publication_target.skill_name + if publication_target is not None + else peer_skill_names.get(peer_expected_key) + if peer_expected_key is not None + else None + ) + policy = resolve_benchmark_policy( + results, + payload, + expected_skill_name=tier3_expected_skill_name, + ) + tier3 = assess_tier3_evidence( + results, + payload, + expected_skill_name=tier3_expected_skill_name, + ) + + reasons: list[str] = [] + if any(result.is_incomplete for result in results): + reasons.append("One or more validators reported incomplete scanner evidence.") + + def blocking_skip(result: ValidationResult) -> bool: + producer = result_publication_evidence(result) + tier2 = producer.tier == 2 if producer is not None else is_tier2_result(result) + return bool( + is_cleanly_skipped(result) + and not is_advisory_agent_eval_skip(result) + and not (tier2 and not policy["tier2_required"]) + and not (is_tier3_result(result) and not policy["tier3_required"]) + ) + + if any(blocking_skip(result) for result in results): + reasons.append("A publication-required validator was skipped.") + if reasons: + return PublicationAssessment("incomplete", policy, tier3, tuple(reasons)) + + if any(not result.passed and not is_cleanly_skipped(result) for result in results): + return PublicationAssessment("fail", policy, tier3, ("A validator failed.",)) + if tier3.status == "fail": + return PublicationAssessment("fail", policy, tier3, (tier3.reason or "Tier 3 failed.",)) + + if expected_skill_name is not None and explicit_expected_key is None: + reasons.append("The expected target skill identity is invalid.") + if len(peer_skill_names) > 1: + reasons.append("Validation results contain conflicting target skill identities.") + if ( + explicit_expected_key is not None + and peer_expected_key is not None + and explicit_expected_key != peer_expected_key + ): + reasons.append("Validation results belong to a different target skill.") + + if publication_target is None: + reasons.append("Validation results lack one canonical publication target identity.") + elif ( + explicit_expected_key is not None + and _publication_identity_key(publication_target.skill_name) != explicit_expected_key + ): + reasons.append("Validation results belong to a different publication target.") + + def require_bound_evidence( + tier_results: list[ValidationResult], + tier_label: str, + ) -> None: + for result in tier_results: + result_target = _result_publication_target(result) + if result_target is None: + reasons.append(f"{tier_label} evidence lacks a canonical publication target identity.") + elif publication_target is None or result_target != publication_target: + reasons.append(f"{tier_label} evidence belongs to a different publication target.") + + executed_tier1 = [result for result in tier1_results if not is_cleanly_skipped(result)] + if not executed_tier1: + reasons.append("Recognized built-in Tier 1 evidence is missing.") + elif any(not result_has_execution_evidence(result) for result in executed_tier1): + reasons.append("Tier 1 lacks trustworthy execution evidence.") + require_bound_evidence(executed_tier1, "Tier 1") + + executed_tier2 = [result for result in tier2_results if not is_cleanly_skipped(result)] + if any(not result_has_execution_evidence(result) for result in executed_tier2): + reasons.append("Tier 2 lacks trustworthy execution evidence.") + elif policy["tier2_required"] and not executed_tier2: + reasons.append("Required recognized built-in Tier 2 evidence is missing.") + require_bound_evidence(executed_tier2, "Tier 2") + + executed_tier3 = [result for result in tier3_results if not is_cleanly_skipped(result)] + require_bound_evidence(executed_tier3, "Tier 3") + + has_present_tier3 = bool(tier3_results) and tier3.status != "skipped" + if (policy["tier3_required"] or has_present_tier3) and not tier3.evidence_complete: + reasons.append(tier3.reason or "Required Tier 3 evidence is missing.") + + if reasons: + return PublicationAssessment("incomplete", policy, tier3, tuple(dict.fromkeys(reasons))) + if tier3.status == "neutral": + return PublicationAssessment("neutral", policy, tier3) + return PublicationAssessment("pass", policy, tier3) def passes_required_gate(result: ValidationResult) -> bool: """Return whether *result* permits the required validation gate to pass.""" gating = result.metadata.get("gating") if isinstance(result.metadata, dict) else None if isinstance(gating, dict): - return bool(result.passed or not gating.get("blocking", True)) + if not gating.get("blocking", True): + return True + if is_tier3_result(result): + selected = select_agent_eval_candidate([result]) + payload = selected[1] if selected is not None else None + verdict, execution_status, truth_consistent = _agent_eval_truth_state(payload) + return bool( + result.passed + and not result.is_incomplete + and truth_consistent + and verdict == "pass" + and execution_status == "succeeded" + and agent_eval_dimension_verdict(payload) == "pass" + and isinstance(payload, dict) + and _agent_eval_dataset_count(payload) > 0 + and _agent_eval_attempt_coverage_complete(payload) + ) + return bool(result.passed) return bool(result.passed or is_advisory_agent_eval_skip(result)) diff --git a/src/skillevaluator/reporting/benchmark.py b/src/skillevaluator/reporting/benchmark.py index 9980f73d..a252bb5b 100644 --- a/src/skillevaluator/reporting/benchmark.py +++ b/src/skillevaluator/reporting/benchmark.py @@ -7,8 +7,9 @@ import math import re +import unicodedata from collections import Counter -from datetime import datetime +from datetime import UTC, datetime from pathlib import PurePosixPath, PureWindowsPath from typing import TYPE_CHECKING, Any @@ -17,11 +18,33 @@ DIMENSION_MAPPING, DIMENSION_VERDICT_NEUTRAL_THRESHOLD, DIMENSION_VERDICT_PASS_THRESHOLD, - KEBAB_CASE_PATTERN, TIER3_LIFT_FAIL_THRESHOLD, TIER3_LIFT_PASS_THRESHOLD, ) -from skillevaluator.reporting.base import ReporterBase, is_advisory_agent_eval_skip, passes_required_gate +from skillevaluator.publication_evidence import result_has_publication_evidence, result_publication_evidence +from skillevaluator.reporting.base import ( + PublicationTargetIdentity, + ReporterBase, + agent_eval_dimension_verdict, + agent_eval_publication_dataset_provenance, + agent_eval_publication_evaluated_at, + agent_eval_publication_evidence_complete, + agent_eval_publication_run_id, + assess_publication, + assess_tier3_evidence, + get_skip_reason, + is_advisory_agent_eval_skip, + is_cleanly_skipped, + is_tier2_result, + is_tier3_result, + publication_identity_present, + publication_semantic_text, + publication_target_for_results, + resolve_benchmark_policy, + result_has_execution_evidence, + result_matches_publication_target, + select_agent_eval_payload, +) from skillevaluator.tier3_environments import HARBOR_ENV_MODES if TYPE_CHECKING: @@ -38,24 +61,38 @@ "token_efficiency": "token usage with and without the skill (reported separately; not scored as a dimension)", } -_TIER2_VALIDATORS = { - "context deduplication", - "intra-skill deduplication", -} - _RETIRED_PRODUCT_NAME = re.compile(r"\b[a-z]*[\s_-]*skills[\s_-]*eval\b", flags=re.IGNORECASE) +_RETIRED_PRODUCT_TOKEN = re.compile(r"\bskills[\s_-]*eval\b", flags=re.IGNORECASE) _RETIRED_SANDBOX_REFERENCE = re.compile( - rf"\b{re.escape(chr(97) + 'stra')}[\s_-]+sandbox\b", + rf"\b{re.escape(chr(97) + 'stra')}[\s_\-\u2010-\u2015\u2043\u2212]+sandbox\b", flags=re.IGNORECASE, ) -_PATH_START = re.compile(r"(?['\"])(?P(?:[A-Za-z]:[\\/]|\\\\|\\|/)[^'\"\r\n]+)(?P=quote)") +_POSIX_PATH_SEPARATORS = "/\u2044\u2215\u29f8" +_WINDOWS_PATH_SEPARATORS = "\\\u2216\u29f5" +_PATH_SEPARATOR_CLASS = re.escape(_POSIX_PATH_SEPARATORS + _WINDOWS_PATH_SEPARATORS) +_PATH_SEPARATOR_TRANSLATION = str.maketrans( + {**dict.fromkeys(_POSIX_PATH_SEPARATORS[1:], "/"), **dict.fromkeys(_WINDOWS_PATH_SEPARATORS[1:], "\\")} +) +_PATH_START = re.compile(rf"(?['\"])(?P(?:[A-Za-z]:[{_PATH_SEPARATOR_CLASS}]|[{_PATH_SEPARATOR_CLASS}])[^'\"\r\n]+)(?P=quote)" +) _QUOTED_FILE_URI_PATH = re.compile( - r"(?P['\"])(?:file:)(?://[^/'\"\r\n]*)?(?P/[^'\"\r\n]+)(?P=quote)", + rf"(?P['\"])(?:file:)(?://[^/'\"\r\n]*)?" + rf"(?P(?:[A-Za-z]:[{_PATH_SEPARATOR_CLASS}]|[{_PATH_SEPARATOR_CLASS}])[^'\"\r\n]+)(?P=quote)", flags=re.IGNORECASE, ) _FILE_URI_PATH = re.compile( - r"\bfile:(?://[^/\s'\"<>]*)?(?P/[^\s'\"<>]+)", + rf"\bfile:(?://[^/\s'\"<>]*)?" + rf"(?P(?:[A-Za-z]:[{_PATH_SEPARATOR_CLASS}]|[{_PATH_SEPARATOR_CLASS}])[^\s'\"<>]+)", + flags=re.IGNORECASE, +) +_PUBLIC_USER_PATH = re.compile( + rf"(?P(?:" + rf"[{re.escape(_POSIX_PATH_SEPARATORS)}](?:Users|home)[{re.escape(_POSIX_PATH_SEPARATORS)}]" + rf"|[A-Za-z]:[{_PATH_SEPARATOR_CLASS}]Users[{_PATH_SEPARATOR_CLASS}]" + rf"|[{re.escape(_WINDOWS_PATH_SEPARATORS)}]Users[{re.escape(_WINDOWS_PATH_SEPARATORS)}]" + rf")[^\s'\"<>]+)", flags=re.IGNORECASE, ) _MARKDOWN_INLINE_SPECIAL = re.compile(r"([\\*_\[\]~])") @@ -64,6 +101,8 @@ _PUBLICATION_URL_SCHEME = re.compile(r"(?Phttps?|ftp)://", flags=re.IGNORECASE) _PUBLICATION_WWW_PREFIX = re.compile(r"\bwww\.", flags=re.IGNORECASE) _TRAILING_PATH_PUNCTUATION = ".,;!?)]}>`'\"" +_LEGACY_AMBIGUOUS_UPLIFT = re.compile(r"\b(?P\d+%)\s+\((?P[+-]\d+%)\)") +_LEGACY_NUM_HEADER = re.compile(r"\|\s*Dimension\s*\|\s*Num\s*\|", flags=re.IGNORECASE) class BenchmarkReporter(ReporterBase): @@ -93,10 +132,11 @@ def render(self, result: ValidationResult) -> str: def render_all(self, results: list[ValidationResult]) -> str: ae = _agent_eval_payload(results) - skill_name = _publication_safe_skill_name(self.skill_name or _skill_name(results, ae)) + expected_skill_name = self.skill_name or _skill_name(results, ae) + skill_name = _publication_safe_skill_name(expected_skill_name) private_labels = _private_environment_labels(ae) - policy = _benchmark_policy(results, ae) - status = _overall_status(results, ae, policy) + policy = _benchmark_policy(results, ae, expected_skill_name=expected_skill_name) + status = _overall_status(results, ae, policy, expected_skill_name=expected_skill_name) lines: list[str] = [ f"# Skill Benchmark: {skill_name}", @@ -144,10 +184,18 @@ def render_all(self, results: list[ValidationResult]) -> str: skill_name, policy, private_labels=private_labels, + expected_skill_name=expected_skill_name, ) self._render_report_purpose(lines) self._render_results_at_a_glance(lines, ae, private_labels) - self._render_tier_status(lines, results, ae, policy, private_labels) + self._render_tier_status( + lines, + results, + ae, + policy, + private_labels, + expected_skill_name=expected_skill_name, + ) self._render_findings(lines, results, private_labels) self._render_methodology(lines, ae, private_labels) self._render_freshness(lines) @@ -179,20 +227,35 @@ def _render_metadata( benchmark_policy: dict[str, bool], *, private_labels: tuple[str, ...], + expected_skill_name: str | None, ) -> None: lines.extend(["## Evaluation Metadata", "", f"- Skill: `{skill_name}`"]) + publication_target = publication_target_for_results( + results, + ae, + expected_skill_name=expected_skill_name, + ) + if publication_target is not None: + lines.append( + f"- Source digest: `{publication_target.skill_digest}` ({publication_target.skill_digest_algorithm})" + ) + else: + lines.append("- Source digest: not recorded (legacy or unbound result)") + evaluated_at = _evaluated_at(ae) lines.append( - f"- Evaluation date: {_publication_safe_inline(_evaluation_date(evaluated_at), private_labels)}" + f"- Evaluation date: {_publication_safe_inline(_evaluation_date(evaluated_at))}" if evaluated_at else "- Evaluation date: not recorded (legacy or non-live result)" ) summary = _mapping((ae or {}).get("summary")) - version = (ae or {}).get("evaluator_version") or summary.get("evaluator_version") + version = _first_publication_safe_label( + ((ae or {}).get("evaluator_version"), summary.get("evaluator_version")), + ) lines.append( - f"- Evaluator version: `{_publication_safe_inline(version, private_labels)}`" + f"- Evaluator version: `{version}`" if version else "- Evaluator version: not recorded (legacy or non-live result)" ) @@ -206,7 +269,7 @@ def _render_metadata( requested = (ae or {}).get("requested_agents") or [] if isinstance(requested, list) and requested: labels = ", ".join( - f"{_human_agent_name(_publication_safe_label(agent, private_labels))} (model not recorded)" + f"{_human_agent_name(_first_publication_safe_label((agent,)) or 'Agent')} (model not recorded)" for agent in requested ) lines.append("- Agents: requested but not run — " + labels) @@ -220,21 +283,26 @@ def _render_metadata( else: lines.append("- Tasks: not recorded (legacy or non-live result)") - digest = (ae or {}).get("dataset_digest") or summary.get("dataset_digest") - digest_algorithm = (ae or {}).get("dataset_digest_algorithm") or summary.get("dataset_digest_algorithm") - if digest: - safe_digest = _publication_safe_inline(digest, private_labels) - algorithm_label = ( - f" ({_publication_safe_inline(digest_algorithm, private_labels)})" if digest_algorithm else "" - ) + dataset_provenance = agent_eval_publication_dataset_provenance(ae) + if dataset_provenance is not None: + digest, digest_algorithm = dataset_provenance + safe_digest = _publication_safe_inline(digest) + algorithm_label = f" ({_publication_safe_inline(digest_algorithm)})" lines.append(f"- Dataset digest: `{safe_digest}`{algorithm_label}") else: lines.append("- Dataset digest: not recorded (legacy or non-live result)") + tier3_run_id = agent_eval_publication_run_id(ae) + lines.append( + f"- Tier 3 run ID: `{_publication_safe_inline(tier3_run_id)}`" + if tier3_run_id + else "- Tier 3 run ID: not recorded (Tier 3 did not complete)" + ) + policy = _mapping((ae or {}).get("attempt_policy")) - attempts = policy.get("max_attempts") - if attempts is not None: - lines.append(f"- Attempts per task: {_publication_safe_inline(attempts, private_labels)}") + attempts = _nonnegative_int(policy.get("max_attempts")) + if attempts > 0: + lines.append(f"- Attempts per task: {attempts}") else: lines.append("- Attempts per task: not recorded (legacy or non-live result)") @@ -244,6 +312,8 @@ def _render_metadata( else: lines.append("- Environment: not recorded (legacy or non-live result)") + tier2_requirement = "required for publication" if benchmark_policy["tier2_required"] else "optional by policy" + lines.append(f"- Tier 2 evidence: {tier2_requirement}") tier3_requirement = "required for publication" if benchmark_policy["tier3_required"] else "optional by policy" lines.append(f"- Tier 3 evidence: {tier3_requirement}") if skip_message := _advisory_agent_eval_skip_message(results): @@ -350,7 +420,14 @@ def _render_tier_status( ae: dict[str, Any] | None, benchmark_policy: dict[str, bool], private_labels: tuple[str, ...], + *, + expected_skill_name: str | None, ) -> None: + publication_target = publication_target_for_results( + results, + ae, + expected_skill_name=expected_skill_name, + ) tier_groups = [ ("Tier 1", "Static validation", _tier1_results(results)), ("Tier 2", "Semantic deduplication", _tier2_results(results)), @@ -366,12 +443,19 @@ def _render_tier_status( ] ) for tier, purpose, tier_results in tier_groups: - status, evidence = _tier_status(tier, tier_results, ae, benchmark_policy) + status, evidence = _tier_status( + tier, + tier_results, + ae, + benchmark_policy, + expected_skill_name=expected_skill_name, + expected_publication_target=publication_target, + ) lines.append(f"| {tier} | {purpose} | **{status}** | {_md_cell(evidence, private_labels)} |") lines.append("") for tier, _purpose, tier_results in tier_groups: - if tier_results and all(_result_skipped(result) for result in tier_results): + if tier_results and all(is_cleanly_skipped(result) for result in tier_results): lines.append(f"{tier} validation was skipped and executed 0 checks.") for reason in _skip_reasons(tier_results): lines.append(f"- {_publication_safe_inline(reason, private_labels)}") @@ -555,11 +639,7 @@ def get_file_extension(self) -> str: def _agent_eval_payload(results: list[ValidationResult]) -> dict[str, Any] | None: - for result in results: - payload = result.metadata.get("agent_eval") if isinstance(result.metadata, dict) else None - if isinstance(payload, dict): - return payload - return None + return select_agent_eval_payload(results) def _mapping(value: object) -> dict[str, Any]: @@ -569,162 +649,61 @@ def _mapping(value: object) -> dict[str, Any]: def _benchmark_policy( results: list[ValidationResult], ae: dict[str, Any] | None, + *, + expected_skill_name: str | None = None, ) -> dict[str, bool]: - """Resolve the persisted publication policy, defaulting Tier 3 to required.""" - candidates: list[object] = [ - (ae or {}).get("benchmark_policy"), - _mapping((ae or {}).get("summary")).get("benchmark_policy"), - ] - candidates.extend( - result.metadata.get("benchmark_policy") for result in results if isinstance(result.metadata, dict) - ) - for candidate in candidates: - if not isinstance(candidate, dict): - continue - required = candidate.get("tier3_required") - if isinstance(required, bool): - return {"tier3_required": required} - return {"tier3_required": True} + """Resolve the persisted publication policy for all configurable tiers.""" + return resolve_benchmark_policy(results, ae, expected_skill_name=expected_skill_name) def _tier3_evidence_complete(ae: dict[str, Any] | None) -> bool: """Require a succeeded run with the minimum publication provenance.""" - if not isinstance(ae, dict): - return False - agents = _agents(ae) - if not agents or any( - not str(agent.get("model") or agent.get("model_name") or agent.get("llm_model") or "").strip() - for agent in agents.values() - ): - return False - summary = _mapping(ae.get("summary")) - verdict = str(ae.get("verdict") or summary.get("verdict") or "").lower() - if verdict not in {"pass", "neutral", "fail"}: - return False - execution_status = str(ae.get("execution_status") or summary.get("execution_status") or "").lower() - evaluated_at = ae.get("evaluated_at") or summary.get("evaluated_at") - evaluator_version = ae.get("evaluator_version") or summary.get("evaluator_version") - dataset_digest = ae.get("dataset_digest") or summary.get("dataset_digest") - attempt_policy = _mapping(ae.get("attempt_policy")) - attempts = _nonnegative_int(attempt_policy.get("max_attempts")) - environment = summary.get("environment") or ae.get("environment") - return bool( - execution_status == "succeeded" - and _tier3_dimension_verdict(ae) is not None - and str(evaluated_at or "").strip() - and str(evaluator_version or "").strip() - and str(dataset_digest or "").strip() - and _dataset_summary(ae)["total_tasks"] > 0 - and attempts > 0 - and str(environment or "").strip() - ) + return agent_eval_publication_evidence_complete(ae) def _tier3_dimension_verdict(ae: dict[str, Any] | None) -> str | None: """Recompute the canonical verdict from every supported agent's dimensions.""" - supported_agents = [agent for agent in _agents(ae).values() if agent.get("execution_status") == "succeeded"] - if not supported_agents: - return None - - agent_verdicts: list[str] = [] - has_partial_evidence = False - for agent in supported_agents: - scores = _agent_dimension_scores(agent) - if scores is None: - has_partial_evidence = True - continue - if any(score < DIMENSION_VERDICT_NEUTRAL_THRESHOLD for score in scores): - agent_verdicts.append("fail") - elif any(score < DIMENSION_VERDICT_PASS_THRESHOLD for score in scores): - agent_verdicts.append("neutral") - else: - agent_verdicts.append("pass") - - if "pass" in agent_verdicts: - return "pass" - if has_partial_evidence or not agent_verdicts: - return None - if "neutral" in agent_verdicts: - return "neutral" - return "fail" - - -def _agent_dimension_scores(agent: dict[str, Any]) -> list[float] | None: - """Return all configured in-range dimension scores, rejecting partial evidence.""" - raw_dimensions = agent.get("dimensions") - if not isinstance(raw_dimensions, list): - return None - dimensions: dict[str, dict[str, Any]] = {} - for dimension in raw_dimensions: - if not isinstance(dimension, dict): - continue - dimension_id = str(dimension.get("id") or "") - if dimension_id in dimensions: - return None - dimensions[dimension_id] = dimension - - scores: list[float] = [] - for dimension_id in DIMENSION_MAPPING: - dimension = dimensions.get(dimension_id) - if dimension is None: - return None - value = dimension.get("with_skill") if "with_skill" in dimension else dimension.get("score") - score = _number(value) - if score is None or not 0.0 <= score <= 1.0: - return None - scores.append(score) - return scores + return agent_eval_dimension_verdict(ae) def _advisory_agent_eval_skip_message(results: list[ValidationResult]) -> str | None: + if assess_tier3_evidence(results).status != "skipped": + return None for result in results: if not is_advisory_agent_eval_skip(result): continue payload = result.metadata.get("agent_eval", {}) if result.metadata else {} provenance = payload.get("provenance", {}) if isinstance(payload, dict) else {} message = provenance.get("message") if isinstance(provenance, dict) else None - return str(message or "Live evaluation did not run.") + return _safe_scalar_text(message).strip() or "Live evaluation did not run." return None -def _result_skipped(result: ValidationResult) -> bool: - """Return whether a result records a skipped validator run.""" - return bool(result.metadata.get("skipped")) or is_advisory_agent_eval_skip(result) - - def _overall_status( results: list[ValidationResult], ae: dict[str, Any] | None, benchmark_policy: dict[str, bool], + *, + expected_skill_name: str | None = None, ) -> str: - if any(result.is_incomplete for result in results): - return "INCOMPLETE" + # Keep this compatibility wrapper so existing callers continue to use the + # benchmark's uppercase vocabulary while every reporter shares one + # publication assessment implementation. + del benchmark_policy + return assess_publication(results, ae, expected_skill_name=expected_skill_name).status.upper() - blocking_skips = [ - result - for result in results - if _result_skipped(result) and not result.metadata.get("optional") and not is_advisory_agent_eval_skip(result) - ] - if blocking_skips: - return "INCOMPLETE" - has_failures = not all(passes_required_gate(result) for result in results) - summary = _mapping((ae or {}).get("summary")) - verdict = str((ae or {}).get("verdict") or summary.get("verdict") or "").lower() - dimension_verdict = _tier3_dimension_verdict(ae) - if has_failures or verdict == "fail" or dimension_verdict == "fail": - return "FAIL" - - tier3_results = _tier3_results(results) - has_present_tier3_result = bool(tier3_results) and not all(_result_skipped(result) for result in tier3_results) - if (benchmark_policy["tier3_required"] or has_present_tier3_result) and not _tier3_evidence_complete(ae): - return "INCOMPLETE" - execution_status = str((ae or {}).get("execution_status") or summary.get("execution_status") or "").lower() - if verdict == "neutral" and execution_status in {"succeeded", ""}: - return "NEUTRAL" - if verdict == "pass" and dimension_verdict == "neutral": - return "NEUTRAL" - return "PASS" +def _is_blocking_publication_skip( + result: ValidationResult, + benchmark_policy: dict[str, bool], +) -> bool: + """Return whether a clean skip blocks this publication policy.""" + return bool( + is_cleanly_skipped(result) + and not is_advisory_agent_eval_skip(result) + and not (_is_tier2(result) and not benchmark_policy["tier2_required"]) + and not (_is_tier3(result) and not benchmark_policy["tier3_required"]) + ) def _verdict_callout(status: str) -> str: @@ -741,20 +720,45 @@ def _skill_name(results: list[ValidationResult], ae: dict[str, Any] | None) -> s if ae: summary = _mapping(ae.get("summary")) candidate = ae.get("skill_name") or summary.get("skill_name") - if candidate: - return str(candidate) + if candidate_text := _safe_scalar_text(candidate).strip(): + return candidate_text for result in results: quality = result.metadata.get("quality_scores") if isinstance(result.metadata, dict) else None - if isinstance(quality, dict) and quality.get("skill_name"): - return str(quality["skill_name"]) + if isinstance(quality, dict) and (candidate := _safe_scalar_text(quality.get("skill_name")).strip()): + return candidate return "skill" +def _safe_scalar_text(value: object) -> str: + if isinstance(value, str): + return value.encode("utf-8", errors="replace").decode("utf-8") + if isinstance(value, bool): + return str(value) + if isinstance(value, int): + return str(value) if value.bit_length() <= 256 else "" + if isinstance(value, float) and math.isfinite(value): + return str(value) + return "" + + def _agents(ae: dict[str, Any] | None) -> dict[str, dict[str, Any]]: agents = (ae or {}).get("agents") if not isinstance(agents, dict): return {} - return {str(name): agent for name, agent in agents.items() if isinstance(agent, dict)} + normalized: dict[str, dict[str, Any]] = {} + normalized_names: set[str] = set() + for raw_name, agent in agents.items(): + if not publication_identity_present(raw_name): + return {} + safe_name = publication_semantic_text(raw_name).strip() + identity_key = safe_name.casefold() + if not publication_identity_present(safe_name) or not isinstance(agent, dict): + return {} + if identity_key in normalized_names: + return {} + normalized_names.add(identity_key) + normalized[safe_name] = agent + return normalized def _agent_label( @@ -762,12 +766,12 @@ def _agent_label( agent: dict[str, Any], private_labels: tuple[str, ...] = (), ) -> str: - display = _human_agent_name( - _publication_safe_label(agent.get("display_name") or agent.get("label") or name, private_labels) + display = _agent_display_label(name, agent, private_labels).replace(",", ",") + safe_model = _first_publication_safe_label( + (agent.get("model"), agent.get("model_name"), agent.get("llm_model")), ) - model = agent.get("model") or agent.get("model_name") or agent.get("llm_model") - if model: - return f"{display} (`{_publication_safe_label(model, private_labels)}`)" + if safe_model: + return f"{display} (`{safe_model}`)" return f"{display} (model not recorded)" @@ -776,37 +780,86 @@ def _agent_table_label( agent: dict[str, Any], private_labels: tuple[str, ...] = (), ) -> str: - display = _human_agent_name( - _publication_safe_label(agent.get("display_name") or agent.get("label") or name, private_labels) - ) + display = _agent_display_label(name, agent, private_labels) return f"{display} (Baseline → Skill Uplift)" +def _agent_display_label( + name: str, + agent: dict[str, Any], + private_labels: tuple[str, ...] = (), +) -> str: + label = "" + for value in (agent.get("display_name"), agent.get("label")): + if not publication_identity_present(value): + continue + candidate = _normalized_publication_text(value) + contains_private_label = any( + re.search( + rf"(? str: + for value in values: + if not publication_identity_present(value): + continue + label = _publication_safe_label(value, private_labels) + if publication_identity_present(label): + return label + return "" + + def _human_agent_name(name: str) -> str: if name == "claude-code": return "Claude Code" if re.fullmatch(r"[A-Za-z0-9]+(?:[-_][A-Za-z0-9]+)*", name): - return name.replace("_", " ").replace("-", " ").title() - return name.title() + humanized = name.replace("_", " ").replace("-", " ").title() + else: + humanized = name.title() + return humanized.replace("Skill" + "evaluator", "SkillEvaluator") def _evaluated_at(ae: dict[str, Any] | None) -> str | None: - value = (ae or {}).get("evaluated_at") or _mapping((ae or {}).get("summary")).get("evaluated_at") - return str(value).strip() if value else None + return agent_eval_publication_evaluated_at(ae) def _evaluation_date(value: str) -> str: candidate = value.replace("Z", "+00:00") try: - return datetime.fromisoformat(candidate).date().isoformat() + parsed = datetime.fromisoformat(candidate) + if parsed.tzinfo is not None and parsed.utcoffset() is not None: + parsed = parsed.astimezone(UTC) + return parsed.date().isoformat() except ValueError: return value[:10] if re.fullmatch(r"\d{4}-\d{2}-\d{2}.*", value) else value def _environment(ae: dict[str, Any] | None) -> str | None: summary = _mapping((ae or {}).get("summary")) - value = summary.get("environment") or (ae or {}).get("environment") - return str(value) if value else None + for value in (summary.get("environment"), (ae or {}).get("environment")): + if publication_identity_present(value): + return _safe_scalar_text(value).strip() + return None def _environment_note(environment: str | None) -> str | None: @@ -838,7 +891,7 @@ def _dataset_summary(ae: dict[str, Any] | None) -> dict[str, int | str]: "positive_tasks": _nonnegative_int(summary.get("positive_tasks")), "negative_tasks": _nonnegative_int(summary.get("negative_tasks")), "unclassified_tasks": _nonnegative_int(summary.get("unclassified_tasks")), - "source": str(summary.get("source") or "payload"), + "source": _safe_scalar_text(summary.get("source")) or "payload", } dataset = _dataset(ae) @@ -853,13 +906,16 @@ def _dataset_summary(ae: dict[str, Any] | None) -> dict[str, int | str]: } task_ids: set[str] = set() - for trial in (ae or {}).get("trials") or []: + raw_trials = (ae or {}).get("trials") + trials = raw_trials if isinstance(raw_trials, list) else [] + for trial in trials: if not isinstance(trial, dict): continue for key in ("entry_id", "case_id", "task_id", "id"): value = trial.get(key) - if value is not None and str(value).strip(): - task_ids.add(str(value).strip()) + safe_value = _safe_scalar_text(value).strip() + if safe_value: + task_ids.add(safe_value) break return { "total_tasks": len(task_ids), @@ -897,23 +953,25 @@ def _dataset_composition_label(summary: dict[str, int | str]) -> str: def _nonnegative_int(value: object) -> int: - if isinstance(value, bool): - return 0 - try: - return max(0, int(value)) - except (TypeError, ValueError): + if isinstance(value, bool) or not isinstance(value, int): return 0 + return value if 0 <= value <= 2**63 - 1 else 0 def _number(value: object) -> float | None: if not isinstance(value, (int, float)) or isinstance(value, bool): return None - number = float(value) + try: + number = float(value) + except (OverflowError, ValueError): + return None return number if math.isfinite(number) else None def _agent_dimension(agent: dict[str, Any], dim_id: str) -> dict[str, Any] | None: - for dimension in agent.get("dimensions") or []: + raw_dimensions = agent.get("dimensions") + dimensions = raw_dimensions if isinstance(raw_dimensions, list) else [] + for dimension in dimensions: if isinstance(dimension, dict) and dimension.get("id") == dim_id: return dimension return None @@ -950,6 +1008,9 @@ def _tier_status( results: list[ValidationResult], ae: dict[str, Any] | None, benchmark_policy: dict[str, bool], + *, + expected_skill_name: str | None = None, + expected_publication_target: PublicationTargetIdentity | None = None, ) -> tuple[str, str]: if not results: return "NOT RUN", "No result was recorded" @@ -957,50 +1018,112 @@ def _tier_status( if incomplete: tools = list(dict.fromkeys(tool for result in incomplete for tool in result.incomplete_scans)) return "INCOMPLETE", f"Missing trustworthy evidence from {', '.join(tools)}" - if all(_result_skipped(result) for result in results): - optional = all(result.metadata.get("optional") or is_advisory_agent_eval_skip(result) for result in results) + if all(is_cleanly_skipped(result) for result in results): + if tier == "Tier 3" and ae: + tier3 = assess_tier3_evidence( + results, + ae, + expected_skill_name=expected_skill_name, + ) + if tier3.status != "skipped": + return tier3.status.upper(), tier3.reason or "Tier 3 skip evidence is incomplete" + if tier == "Tier 1": + optional = False + elif tier == "Tier 2": + optional = not benchmark_policy["tier2_required"] + else: + optional = not benchmark_policy["tier3_required"] or all( + is_advisory_agent_eval_skip(result) for result in results + ) return ("SKIPPED (ADVISORY)" if optional else "INCOMPLETE"), "; ".join(_skip_reasons(results)) findings = [finding for result in results for finding in result.findings] - if any(not result.passed and not _result_skipped(result) for result in results): + if any(not result.passed and not is_cleanly_skipped(result) for result in results): return "FAILED", f"{len(results)} validator(s); {len(findings)} finding(s)" - if tier == "Tier 3" and ae: - summary = _mapping(ae.get("summary")) - execution = str(ae.get("execution_status") or summary.get("execution_status") or "").lower() - verdict = str(ae.get("verdict") or summary.get("verdict") or "").lower() - dimension_verdict = _tier3_dimension_verdict(ae) - if verdict == "fail" or dimension_verdict == "fail": - return "FAIL", f"{len(_agents(ae))} agent(s); {_dataset_summary(ae)['total_tasks']} task(s)" - if execution and execution != "succeeded": - return "INCOMPLETE", f"Execution status: {execution}" - if not _tier3_evidence_complete(ae): + publication_results = results + if tier in {"Tier 1", "Tier 2"}: + expected_tier = 1 if tier == "Tier 1" else 2 + publication_results = [ + result for result in results if result_has_publication_evidence(result, tier=expected_tier) + ] + if not publication_results: + return "INCOMPLETE", f"No recognized built-in {tier} producer evidence was recorded" + + unbound_results = [ + result + for result in publication_results + if not is_cleanly_skipped(result) and not result_matches_publication_target(result, expected_publication_target) + ] + if tier in {"Tier 1", "Tier 2"} and unbound_results: + validator_names = list( + dict.fromkeys(result.validator_name or f"{tier} validator" for result in unbound_results) + ) + return "INCOMPLETE", f"Missing canonical source identity from {', '.join(validator_names)}" + + if tier in {"Tier 1", "Tier 2"} and (missing_evidence := _results_without_execution_evidence(publication_results)): + validator_names = list( + dict.fromkeys(result.validator_name or f"{tier} validator" for result in missing_evidence) + ) + return "INCOMPLETE", f"Missing trustworthy execution evidence from {', '.join(validator_names)}" + + skipped = [result for result in publication_results if is_cleanly_skipped(result)] + if tier in {"Tier 1", "Tier 2"} and skipped: + reasons = "; ".join(_skip_reasons(skipped)) + completed = len(publication_results) - len(skipped) + evidence = f"{completed} completed validator(s); {len(findings)} finding(s); {reasons}" + if any(_is_blocking_publication_skip(result, benchmark_policy) for result in skipped): + return "INCOMPLETE", evidence + return "PASSED WITH OBSERVATIONS", evidence + + if tier == "Tier 3": + if not ae: evidence = ( "Required Tier 3 evidence is missing" if benchmark_policy["tier3_required"] else "Present Tier 3 result lacks complete evidence" ) return "INCOMPLETE", evidence - effective_verdict = verdict - if verdict == "pass" and dimension_verdict == "neutral": - effective_verdict = "neutral" - if effective_verdict in {"pass", "neutral"}: - return effective_verdict.upper(), ( - f"{len(_agents(ae))} agent(s); {_dataset_summary(ae)['total_tasks']} task(s)" + tier3 = assess_tier3_evidence(results, ae, expected_skill_name=expected_skill_name) + if tier3.status == "fail": + return "FAIL", f"{len(_agents(ae))} agent(s); {_dataset_summary(ae)['total_tasks']} task(s)" + if not tier3.evidence_complete: + evidence = tier3.reason or ( + "Required Tier 3 evidence is missing" + if benchmark_policy["tier3_required"] + else "Present Tier 3 result lacks complete evidence" ) + return "INCOMPLETE", evidence + if unbound_results: + validator_names = list( + dict.fromkeys(result.validator_name or "Tier 3 validator" for result in unbound_results) + ) + return "INCOMPLETE", f"Missing canonical source identity from {', '.join(validator_names)}" + if tier3.status in {"pass", "neutral"}: + return tier3.status.upper(), (f"{len(_agents(ae))} agent(s); {_dataset_summary(ae)['total_tasks']} task(s)") status = "PASSED WITH OBSERVATIONS" if findings else "PASSED" return status, f"{len(results)} validator(s); {len(findings)} finding(s)" +def _results_without_execution_evidence( + results: list[ValidationResult], +) -> list[ValidationResult]: + """Return non-skipped results that do not prove a validator ran.""" + return [ + result for result in results if not is_cleanly_skipped(result) and not result_has_execution_evidence(result) + ] + + +def _tier2_results_without_execution_evidence( + results: list[ValidationResult], +) -> list[ValidationResult]: + """Backward-compatible alias for the generalized evidence check.""" + return _results_without_execution_evidence(results) + + def _skip_reasons(results: list[ValidationResult]) -> list[str]: - reasons: list[str] = [] - for result in results: - payload = result.metadata.get("agent_eval", {}) if result.metadata else {} - provenance = payload.get("provenance", {}) if isinstance(payload, dict) else {} - advisory_message = provenance.get("message") if isinstance(provenance, dict) else None - reasons.append(str(result.metadata.get("skip_reason") or advisory_message or "Prerequisite unavailable")) - return list(dict.fromkeys(reasons)) + return list(dict.fromkeys(get_skip_reason(result) for result in results)) def _metric_signals(ae: dict[str, Any] | None) -> list[str]: @@ -1011,15 +1134,20 @@ def _metric_signals(ae: dict[str, Any] | None) -> list[str]: if not isinstance(evaluators, dict): continue for name, values in evaluators.items(): - if name in seen or not isinstance(values, dict): + safe_name = _safe_scalar_text(name).strip() + if not safe_name or safe_name in seen or not isinstance(values, dict): continue if any(values.get(field) is not None for field in ("with_skill", "baseline", "lift")): - seen.add(str(name)) - signals.append(str(name)) + seen.add(safe_name) + signals.append(safe_name) if signals: return signals metric_ids = (ae or {}).get("metric_ids") - return [str(item) for item in metric_ids] if isinstance(metric_ids, list) else [] + return ( + [safe for item in metric_ids if (safe := _safe_scalar_text(item).strip())] + if isinstance(metric_ids, list) + else [] + ) def _metric_labels(ae: dict[str, Any] | None) -> dict[str, str]: @@ -1037,7 +1165,7 @@ def _weighted_signals(config: dict[str, Any]) -> str: def _tier1_results(results: list[ValidationResult]) -> list[ValidationResult]: - return [result for result in results if not _is_tier2(result) and not _is_tier3(result)] + return [result for result in results if not _is_tier3(result) and not _is_tier2(result)] def _tier2_results(results: list[ValidationResult]) -> list[ValidationResult]: @@ -1049,14 +1177,14 @@ def _tier3_results(results: list[ValidationResult]) -> list[ValidationResult]: def _is_tier2(result: ValidationResult) -> bool: - name = result.validator_name.lower() - if name in _TIER2_VALIDATORS or "dedup" in name: - return True - return any(finding.category == "CONTENT_DEDUP" for finding in result.findings) + producer = result_publication_evidence(result) + if producer is not None: + return producer.tier == 2 + return is_tier2_result(result) def _is_tier3(result: ValidationResult) -> bool: - return bool(result.metadata.get("agent_eval")) or result.validator_name == "AGENT_EVAL" + return is_tier3_result(result) def _top_findings(findings: list[Finding], *, limit: int) -> list[Finding]: @@ -1081,13 +1209,13 @@ def _trusted_md_cell(value: object) -> str: def _publication_safe_skill_name(value: object) -> str: - """Return a canonical target identity or a non-injectable public fallback.""" - candidate = " ".join(str(value).split()) - if re.fullmatch(KEBAB_CASE_PATTERN, candidate) is not None: - return candidate - if _RETIRED_PRODUCT_NAME.fullmatch(candidate): + """Return an exact NFC target identity or a non-injectable fallback.""" + candidate = _exact_publication_text(value) + if not publication_identity_present(candidate): + return "skill" + if _RETIRED_PRODUCT_NAME.search(publication_semantic_text(candidate, strip_marks=True)): return "SkillEvaluator" - return "skill" + return _publication_safe_inline(candidate, preserve_exact_nfc=True) def _private_environment_labels(ae: dict[str, Any] | None) -> tuple[str, ...]: @@ -1103,33 +1231,46 @@ def _private_environment_labels(ae: dict[str, Any] | None) -> tuple[str, ...]: ] labels: list[str] = [] for value in candidates: - label = " ".join(str(value or "").split()) - if label and label.casefold() not in HARBOR_ENV_MODES and label not in labels: + if not publication_identity_present(value): + continue + label = _normalized_publication_text(value) + if publication_identity_present(label) and label.casefold() not in HARBOR_ENV_MODES and label not in labels: labels.append(label) return tuple(labels) def _publication_safe_label(value: object, private_labels: tuple[str, ...] = ()) -> str: - """Sanitize a classified display label and normalize only an exact retired product name.""" - label = _publication_safe_inline(value, private_labels) - if _RETIRED_PRODUCT_NAME.fullmatch(label): - return "SkillEvaluator" - return label + """Sanitize a classified display label for a public benchmark card.""" + label = _normalized_publication_text(value) + is_private_label = any(label.casefold() == private_label.casefold() for private_label in private_labels) + matching_label = publication_semantic_text(label, strip_marks=True) + if not is_private_label and _RETIRED_PRODUCT_NAME.fullmatch(matching_label): + label = "SkillEvaluator" + return _publication_safe_inline(label, private_labels) -def _publication_safe_inline(value: object, private_labels: tuple[str, ...] = ()) -> str: +def _publication_safe_inline( + value: object, + private_labels: tuple[str, ...] = (), + *, + preserve_exact_nfc: bool = False, +) -> str: """Render untrusted metadata as one publication-safe Markdown line.""" - text = " ".join(str(value).split()) + text = _exact_publication_text(value) if preserve_exact_nfc else _normalized_publication_text(value) text = _redact_absolute_paths(text) text = _RETIRED_SANDBOX_REFERENCE.sub("isolated sandbox", text) for label in sorted(private_labels, key=len, reverse=True): text = re.sub( - re.escape(label), + rf"(?", ">") + text = _replace_semantic_tokens(text, _RETIRED_SANDBOX_REFERENCE, "isolated sandbox") + text = _replace_retired_product_tokens(text) + text = _LEGACY_AMBIGUOUS_UPLIFT.sub(r"\g [change \g]", text) + text = _LEGACY_NUM_HEADER.sub("| Dimension | Count |", text) + text = text.replace("&", "&").replace("`", "'").replace("<", "<").replace(">", ">") text = _PUBLICATION_URL_SCHEME.sub(lambda match: f"{match.group('scheme')}://", text) text = _PUBLICATION_WWW_PREFIX.sub(lambda match: f"{match.group(0)[:-1]}.", text) text = text.replace("@", "@") @@ -1144,6 +1285,56 @@ def _publication_safe_inline(value: object, private_labels: tuple[str, ...] = () return text +def _normalized_publication_text(value: object) -> str: + """Canonicalize text and discard invisible controls before publication decisions.""" + return " ".join(publication_semantic_text(value).split()) + + +def _exact_publication_text(value: object) -> str: + """Preserve a filesystem identity only when its exact NFC text is safe.""" + if not isinstance(value, str): + return "" + text = value.encode("utf-8", errors="replace").decode("utf-8") + if text != value: + return "" + text = unicodedata.normalize("NFC", text) + if publication_semantic_text(text) != unicodedata.normalize("NFKC", text): + return "" + return text + + +def _replace_retired_product_tokens(value: str) -> str: + """Replace retired identity tokens even when Unicode marks split the spelling.""" + return _replace_semantic_tokens(value, _RETIRED_PRODUCT_TOKEN, "SkillEvaluator") + + +def _replace_semantic_tokens(value: str, pattern: re.Pattern[str], replacement: str) -> str: + """Replace match-only normalized tokens with one linear source reconstruction.""" + searchable: list[str] = [] + source_offsets: list[int] = [] + for index, character in enumerate(value): + for semantic_character in publication_semantic_text(character, strip_marks=True): + searchable.append(semantic_character) + source_offsets.append(index) + matches = list(pattern.finditer("".join(searchable))) + if not matches: + return value + + # Reconstruct once. Repeated whole-string slicing here is quadratic for + # dense untrusted metadata such as thousands of adjacent retired tokens. + parts: list[str] = [] + cursor = 0 + for match in matches: + start = source_offsets[match.start()] + end = source_offsets[match.end() - 1] + 1 + if start < cursor: + continue + parts.extend((value[cursor:start], replacement)) + cursor = end + parts.append(value[cursor:]) + return "".join(parts) + + def _redact_absolute_paths(value: str) -> str: """Reduce absolute POSIX and Windows paths embedded in free text to basenames.""" @@ -1163,8 +1354,26 @@ def redact_quoted(match: re.Match[str]) -> str: basename = _absolute_path_basename(path) return f"{match.group('quote')}{basename}{match.group('quote')}" if basename else match.group(0) + def redact_public_user_path(match: re.Match[str]) -> str: + candidate = match.group("path") + core = candidate.rstrip(_TRAILING_PATH_PUNCTUATION) + suffix = candidate[len(core) :] + basename = _absolute_path_basename(core) + if not basename: + return match.group(0) + is_drive_path = len(core) >= 2 and core[1] == ":" + preceding = value[match.start("path") - 1] if match.start("path") > 0 else "" + preserve_separator = ( + not is_drive_path + and candidate[0] in _POSIX_PATH_SEPARATORS + and (preceding.isalnum() or preceding in {":", "/"}) + ) + separator = candidate[0] if preserve_separator else "" + return f"{separator}{basename}{suffix}" + text = _QUOTED_FILE_URI_PATH.sub(redact_quoted_file_uri, value) text = _FILE_URI_PATH.sub(redact_file_uri, text) + text = _PUBLIC_USER_PATH.sub(redact_public_user_path, text) text = _QUOTED_ABSOLUTE_PATH.sub(redact_quoted, text) tokens: list[str] = [] for token in text.split(" "): @@ -1182,9 +1391,10 @@ def redact_quoted(match: re.Match[str]) -> str: def _absolute_path_basename(value: str) -> str | None: - posix_path = PurePosixPath(value) - windows_path = PureWindowsPath(value) - if posix_path.is_absolute() and not value.startswith("//"): + canonical_value = value.translate(_PATH_SEPARATOR_TRANSLATION) + posix_path = PurePosixPath(canonical_value) + windows_path = PureWindowsPath(canonical_value) + if posix_path.is_absolute() and not canonical_value.startswith("//"): return posix_path.name or "redacted-path" if windows_path.is_absolute() or windows_path.root: return windows_path.name or "redacted-path" @@ -1192,7 +1402,7 @@ def _absolute_path_basename(value: str) -> str | None: def _publication_safe_environment(value: object) -> str: - environment = str(value).strip() + environment = _normalized_publication_text(value) return environment if environment.casefold() in HARBOR_ENV_MODES else "Isolated sandbox" diff --git a/src/skillevaluator/reporting/cli.py b/src/skillevaluator/reporting/cli.py index 988e78e9..7d803614 100644 --- a/src/skillevaluator/reporting/cli.py +++ b/src/skillevaluator/reporting/cli.py @@ -27,7 +27,7 @@ DIMENSION_VERDICT_NEUTRAL_THRESHOLD, DIMENSION_VERDICT_PASS_THRESHOLD, ) -from skillevaluator.reporting.base import ReporterBase, passes_required_gate +from skillevaluator.reporting.base import ReporterBase, is_advisory_agent_eval_skip, passes_required_gate from skillevaluator.reporting.harbor_viewer import ( harbor_evidence_link_text, normalize_harbor_viewer_for_display, @@ -584,10 +584,4 @@ def _static_test_evidence_message(result: ValidationResult) -> str | None: @staticmethod def _is_advisory_agent_eval_skip(result: ValidationResult) -> bool: """Return whether an AGENT_EVAL result records a skipped live run.""" - if result.validator_name != "AGENT_EVAL": - return False - payload = result.metadata.get("agent_eval", {}) if result.metadata else {} - provenance = payload.get("provenance", {}) if isinstance(payload, dict) else {} - return bool( - isinstance(provenance, dict) and provenance.get("advisory") and provenance.get("reason") == "skipped" - ) + return is_advisory_agent_eval_skip(result) diff --git a/src/skillevaluator/reporting/harbor_viewer.py b/src/skillevaluator/reporting/harbor_viewer.py index 1a7eff03..9ab88a02 100644 --- a/src/skillevaluator/reporting/harbor_viewer.py +++ b/src/skillevaluator/reporting/harbor_viewer.py @@ -80,6 +80,7 @@ def normalize_agent_eval_harbor_links(agent_eval: dict[str, Any]) -> dict[str, A } normalized["summary"] = normalized_summary else: + normalized.pop("harbor_viewer", None) summary = normalized.get("summary") if isinstance(summary, dict) and "harbor_viewer" in summary: normalized_summary = dict(summary) diff --git a/src/skillevaluator/reporting/html.py b/src/skillevaluator/reporting/html.py index 8bd15a5c..2da3bda3 100644 --- a/src/skillevaluator/reporting/html.py +++ b/src/skillevaluator/reporting/html.py @@ -24,8 +24,13 @@ import json import math import pkgutil +import re +import sys +import unicodedata +from collections.abc import Iterator from dataclasses import dataclass from datetime import UTC, datetime +from decimal import Decimal, InvalidOperation from importlib import resources from pathlib import Path from typing import TYPE_CHECKING, Any, ClassVar @@ -35,15 +40,38 @@ from skillevaluator import __version__ from skillevaluator.constants import TIER3_LIFT_FAIL_THRESHOLD, TIER3_LIFT_PASS_THRESHOLD -from skillevaluator.reporting.base import ReporterBase, is_advisory_agent_eval_skip, passes_required_gate +from skillevaluator.publication_evidence import result_publication_evidence_dict +from skillevaluator.reporting.base import ( + ReporterBase, + agent_eval_publication_evidence_complete, + agent_eval_report_serialization_issue, + agent_eval_report_serialization_limits, + agent_eval_report_text_limit, + assess_publication, + assess_tier3_evidence, + get_skip_reason, + is_advisory_agent_eval_skip, + is_cleanly_skipped, + is_tier2_result, + is_tier3_result, + passes_required_gate, + publication_identity_present, + publication_semantic_text, + publication_target_conflict_marker, + publication_target_dict, + result_publication_target_conflict_marker, + result_publication_target_dict, + select_agent_eval_candidate, +) +from skillevaluator.reporting.base import ( + is_tier2_validator_name as _is_tier2_validator_name, +) from skillevaluator.reporting.harbor_viewer import normalize_agent_eval_harbor_links if TYPE_CHECKING: from skillevaluator.models import ValidationResult -_TIER2_VALIDATOR_MARKERS = ("similarity", "dedup", "context optimization") - # Tier 3 already enforces a 2 MiB canonical payload limit. HTML needs a # separate bound because script-safe escaping (``<`` -> ``\u003c``), pretty # diagnostics, and visible dataset fields can otherwise multiply that payload @@ -55,6 +83,980 @@ _TIER3_HTML_PREVIEW_STRING_CHARS = 4 * 1024 _TIER3_HTML_PREVIEW_COLLECTION_ITEMS = 64 _TIER3_PREVIEW_MARKER = "... [HTML preview truncated; download the full Tier 3 payload]" +_TIER3_JSON_SAFE_PRIORITY_KEYS = ( + "schema_version", + "verdict", + "execution_status", + "summary", + "skill_name", + "publication_target", + "run_id", + "evaluated_at", + "evaluator_version", + "dataset_digest", + "dataset_digest_algorithm", + "benchmark_policy", + "attempt_policy", + "dataset_summary", + "agents", + "model", + "model_name", + "llm_model", + "with_skill", + "overall_score", + "scored_attempts", + "dimensions", + "id", + "score", + "evaluators", +) +_TIER3_PROBABILITY_TEXT = re.compile(r"(?:\d{1,16}(?:\.\d{0,16})?|\.\d{1,16})(?:[eE][+-]?\d{1,9})?") + + +def is_tier2_validator_name(validator_name: str | None) -> bool: + """Compatibility wrapper for callers that imported this helper here.""" + return _is_tier2_validator_name(validator_name) + + +def _finite_number( + value: object, + *, + minimum: float | None = None, + maximum: float | None = None, +) -> float | None: + """Return a bounded finite number, rejecting booleans and shaped values.""" + if isinstance(value, bool) or not isinstance(value, (int, float)): + return None + try: + number = float(value) + except (OverflowError, ValueError): + return None + if not math.isfinite(number): + return None + if minimum is not None and number < minimum: + return None + if maximum is not None and number > maximum: + return None + return number + + +def _nonnegative_count(value: object) -> int: + return count if (count := _strict_nonnegative_count(value)) is not None else 0 + + +def _strict_nonnegative_count(value: object) -> int | None: + if isinstance(value, bool) or not isinstance(value, int) or value < 0 or value.bit_length() > 63: + return None + return value + + +def _json_safe_tier3_text(value: str) -> str: + """Normalize template-facing text and replace invalid UTF-8 surrogates.""" + safe_value = value.encode("utf-8", errors="replace").decode("utf-8") + return publication_semantic_text(safe_value) + + +def _json_safe_tier3_mapping_keys(value: dict[object, object]) -> Iterator[object]: + """Yield proof keys first without materializing an untrusted wide mapping.""" + priority = frozenset(_TIER3_JSON_SAFE_PRIORITY_KEYS) + for key in _TIER3_JSON_SAFE_PRIORITY_KEYS: + if key in value: + yield key + for key in value: + if key not in priority: + yield key + + +def _json_safe_tier3_value( + value: object, + *, + _depth: int = 0, + _active_containers: set[int] | None = None, + _remaining_nodes: list[int] | None = None, + _remaining_text_bytes: list[int] | None = None, + _truncated: list[bool] | None = None, + _max_depth: int | None = None, + _preserve_exact_text: bool = False, +) -> Any: + """Copy untrusted Tier 3 metadata into a finite JSON-compatible shape.""" + max_depth, max_nodes = agent_eval_report_serialization_limits() + if _max_depth is not None: + max_depth = _max_depth + active_containers = _active_containers if _active_containers is not None else set() + remaining_nodes = _remaining_nodes if _remaining_nodes is not None else [max_nodes] + remaining_text_bytes = ( + _remaining_text_bytes if _remaining_text_bytes is not None else [agent_eval_report_text_limit()] + ) + truncated = _truncated if _truncated is not None else [False] + if remaining_nodes[0] <= 0: + truncated[0] = True + return None + remaining_nodes[0] -= 1 + if _depth > max_depth: + truncated[0] = True + return None + if value is None or isinstance(value, bool): + return value + if isinstance(value, str): + if len(value) > remaining_text_bytes[0]: + truncated[0] = True + return None + utf8_safe_value = value.encode("utf-8", errors="replace").decode("utf-8") + safe_value = _json_safe_tier3_text(value) + if utf8_safe_value != value or safe_value != unicodedata.normalize("NFKC", utf8_safe_value): + truncated[0] = True + elif _preserve_exact_text and unicodedata.normalize("NFC", utf8_safe_value) == utf8_safe_value: + # Publication target names are filesystem identities. Preserve + # their exact NFC spelling instead of collapsing compatibility- + # distinct ASCII and fullwidth spellings. + safe_value = utf8_safe_value + encoded_size = len(safe_value.encode("utf-8")) + if encoded_size > remaining_text_bytes[0]: + truncated[0] = True + return None + remaining_text_bytes[0] -= encoded_size + return safe_value + if isinstance(value, int): + if value.bit_length() <= 256: + return value + truncated[0] = True + return None + if isinstance(value, float): + if math.isfinite(value): + return value + truncated[0] = True + return None + if isinstance(value, dict): + container_id = id(value) + if container_id in active_containers: + truncated[0] = True + return None + active_containers.add(container_id) + safe: dict[str, Any] = {} + collided_keys: set[str] = set() + try: + total_keys = len(value) + processed_keys = 0 + if total_keys and remaining_nodes[0] <= 0: + truncated[0] = True + return safe + for key in _json_safe_tier3_mapping_keys(value): + processed_keys += 1 + if isinstance(key, str): + if len(key) > remaining_text_bytes[0]: + truncated[0] = True + remaining_nodes[0] -= 1 + if remaining_nodes[0] <= 0 and processed_keys < total_keys: + break + continue + utf8_safe_key = key.encode("utf-8", errors="replace").decode("utf-8") + safe_key = _json_safe_tier3_text(key) + if utf8_safe_key != key or safe_key != unicodedata.normalize("NFKC", utf8_safe_key): + truncated[0] = True + elif ( + isinstance(key, bool) + or (isinstance(key, int) and key.bit_length() <= 256) + or (isinstance(key, float) and math.isfinite(key)) + ): + safe_key = str(key) + truncated[0] = True + else: + truncated[0] = True + remaining_nodes[0] -= 1 + if remaining_nodes[0] <= 0 and processed_keys < total_keys: + break + continue + if len(safe_key) > remaining_text_bytes[0]: + truncated[0] = True + remaining_nodes[0] -= 1 + if remaining_nodes[0] <= 0 and processed_keys < total_keys: + break + continue + encoded_key_size = len(safe_key.encode("utf-8")) + if encoded_key_size > remaining_text_bytes[0]: + truncated[0] = True + remaining_nodes[0] -= 1 + if remaining_nodes[0] <= 0 and processed_keys < total_keys: + break + continue + remaining_text_bytes[0] -= encoded_key_size + if not safe_key or safe_key in collided_keys: + truncated[0] = True + remaining_nodes[0] -= 1 + if remaining_nodes[0] <= 0 and processed_keys < total_keys: + break + continue + if safe_key in safe: + # Drop every alias in a normalized-key collision rather + # than retaining an order-dependent first or last value. + safe.pop(safe_key) + collided_keys.add(safe_key) + truncated[0] = True + remaining_nodes[0] -= 1 + if remaining_nodes[0] <= 0 and processed_keys < total_keys: + break + continue + item = value[key] + if safe_key == "publication_target": + item = publication_target_dict(item) + elif safe_key == "publication_target_conflict": + item = publication_target_conflict_marker(item) + safe[safe_key] = _json_safe_tier3_value( + item, + _depth=_depth + 1, + _active_containers=active_containers, + _remaining_nodes=remaining_nodes, + _remaining_text_bytes=remaining_text_bytes, + _truncated=truncated, + _max_depth=max_depth, + _preserve_exact_text=_preserve_exact_text or safe_key in {"publication_target", "skill_name"}, + ) + if remaining_nodes[0] <= 0 and processed_keys < total_keys: + truncated[0] = True + break + finally: + active_containers.remove(container_id) + return safe + if isinstance(value, (list, tuple)): + if isinstance(value, tuple): + truncated[0] = True + container_id = id(value) + if container_id in active_containers: + truncated[0] = True + return None + active_containers.add(container_id) + safe_items: list[Any] = [] + try: + for item in value: + if remaining_nodes[0] <= 0: + truncated[0] = True + break + safe_items.append( + _json_safe_tier3_value( + item, + _depth=_depth + 1, + _active_containers=active_containers, + _remaining_nodes=remaining_nodes, + _remaining_text_bytes=remaining_text_bytes, + _truncated=truncated, + _max_depth=max_depth, + _preserve_exact_text=_preserve_exact_text, + ) + ) + finally: + active_containers.remove(container_id) + return safe_items + truncated[0] = True + return None + + +def _json_safe_tier3_payload(value: object) -> dict[str, Any]: + """Return a bounded payload while disclosing any serialization truncation.""" + truncated = [False] + normalized = _json_safe_tier3_value(value, _truncated=truncated) + payload = normalized if isinstance(normalized, dict) else {} + if not isinstance(value, dict): + truncated[0] = True + if truncated[0]: + payload["_serialization_truncated"] = True + return payload + + +def _string_list(value: object) -> list[str]: + return [item for item in value if isinstance(item, str) and item] if isinstance(value, list) else [] + + +def _dict_list(value: object) -> list[dict[str, Any]]: + return [item for item in value if isinstance(item, dict)] if isinstance(value, list) else [] + + +def _tier3_display_identifier(value: object) -> str: + if not publication_identity_present(value): + return "" + identifier = publication_semantic_text(value).strip() + return identifier if publication_identity_present(identifier) else "" + + +def _tier3_identifier_list(value: object) -> list[str]: + """Return visible, normalized, collision-free identifiers in source order.""" + if not isinstance(value, list): + return [] + identifiers: list[str] = [] + seen: set[str] = set() + for raw_identifier in value: + identifier = _tier3_display_identifier(raw_identifier) + identity_key = identifier.casefold() + if not identifier or identity_key in seen: + continue + seen.add(identity_key) + identifiers.append(identifier) + return identifiers + + +def _tier3_agent_entries(value: object) -> list[tuple[str, dict[str, Any]]]: + """Normalize agent keys and fail closed on malformed or colliding peers.""" + if not isinstance(value, dict): + return [] + entries: list[tuple[str, dict[str, Any]]] = [] + seen: set[str] = set() + for raw_name, raw_agent in value.items(): + if not isinstance(raw_name, str): + return [] + name = _tier3_display_identifier(raw_name) + identity_key = name.casefold() + if not name or not isinstance(raw_agent, dict) or identity_key in seen: + return [] + seen.add(identity_key) + entries.append((name, raw_agent)) + return entries + + +def _score_mapping( + value: object, + *, + minimum: float = 0.0, + maximum: float = 1.0, +) -> dict[str, float]: + if not isinstance(value, dict): + return {} + scores: dict[str, float] = {} + seen: set[str] = set() + for key, item in value.items(): + identifier = _tier3_display_identifier(key) + identity_key = identifier.casefold() + if not identifier: + continue + if identity_key in seen: + return {} + seen.add(identity_key) + score = _finite_number(item, minimum=minimum, maximum=maximum) + if score is not None: + scores[identifier] = score + return scores + + +def _tier3_text_mapping(value: object) -> dict[str, str]: + """Normalize public label maps without silently overwriting aliases.""" + if not isinstance(value, dict): + return {} + labels: dict[str, str] = {} + seen: set[str] = set() + for raw_name, raw_label in value.items(): + name = _tier3_display_identifier(raw_name) + identity_key = name.casefold() + if not publication_identity_present(raw_label): + continue + label = publication_semantic_text(raw_label).strip() + if not name or not publication_identity_present(label): + continue + if identity_key in seen: + return {} + seen.add(identity_key) + labels[name] = label + return labels + + +def _sanitize_tier3_dimension(value: dict[str, Any]) -> dict[str, Any]: + dimension = dict(value) + for key in ("baseline", "with_skill", "score"): + dimension[key] = _finite_number(dimension.get(key), minimum=0.0, maximum=1.0) + dimension["lift"] = _finite_number(dimension.get("lift"), minimum=-1.0, maximum=1.0) + for key in ("id", "dimension", "label"): + if key in dimension: + dimension[key] = _tier3_display_identifier(dimension[key]) + if "explanation" in dimension and not isinstance(dimension["explanation"], str): + dimension["explanation"] = "" + if "verdict" in dimension: + raw_verdict = dimension["verdict"] + dimension["verdict"] = None if raw_verdict is None else _tier3_display_identifier(raw_verdict) + dimension["evaluators"] = _tier3_identifier_list(dimension.get("evaluators")) + dimension["reasoning_bullets"] = _string_list(dimension.get("reasoning_bullets")) + return dimension + + +def _sanitize_tier3_pass_condition(value: object) -> dict[str, Any]: + if not isinstance(value, dict): + return {} + condition = dict(value) + if "rate" in condition: + rate = _finite_number(condition.get("rate"), minimum=0.0, maximum=1.0) + if rate is None: + condition.pop("rate", None) + else: + condition["rate"] = rate + for key in ( + "passed_cases", + "failed_cases", + "total_cases", + "attempts_used", + "max_attempts_possible", + ): + if key in condition: + condition[key] = _nonnegative_count(condition.get(key)) + + cases: dict[str, dict[str, Any]] = {} + raw_cases = condition.get("cases") + if isinstance(raw_cases, dict): + case_ids: set[str] = set() + for raw_case_id, raw_case in raw_cases.items(): + case_id = _tier3_display_identifier(raw_case_id) + identity_key = case_id.casefold() + if not isinstance(raw_case, dict): + continue + if not case_id: + continue + if identity_key in case_ids: + cases = {} + break + case_ids.add(identity_key) + case = dict(raw_case) + for key in ("attempts_used", "attempts_missing", "attempts_skipped"): + case[key] = _nonnegative_count(case.get(key)) + case["best_score"] = _finite_number(case.get("best_score"), minimum=0.0, maximum=1.0) + first_pass = case.get("first_pass_attempt") + case["first_pass_attempt"] = ( + _nonnegative_count(first_pass) if isinstance(first_pass, int) and first_pass > 0 else None + ) + case["passed"] = case.get("passed") if isinstance(case.get("passed"), bool) else False + attempts: list[dict[str, Any]] = [] + for raw_attempt in _dict_list(case.get("attempts")): + attempt = dict(raw_attempt) + attempt["score"] = _finite_number(attempt.get("score"), minimum=0.0, maximum=1.0) + attempt_number = attempt.get("attempt") + attempt["attempt"] = ( + _nonnegative_count(attempt_number) + if isinstance(attempt_number, int) and attempt_number > 0 + else None + ) + attempts.append(attempt) + case["attempts"] = attempts + cases[case_id] = case + condition["cases"] = cases + return condition + + +def _sanitize_tier3_probability_text(value: object) -> tuple[str, Decimal] | None: + if not isinstance(value, str): + return None + text = publication_semantic_text(value).strip() + if not _TIER3_PROBABILITY_TEXT.fullmatch(text): + return None + try: + probability = Decimal(text) + except InvalidOperation: + return None + if not probability.is_finite() or not Decimal(0) <= probability <= Decimal(1): + return None + return text, probability + + +def _probability_text_matches_number(probability: Decimal, number: float) -> bool: + numeric_probability = Decimal(str(number)) + if numeric_probability == 0: + return probability == 0 + relative_match = abs(probability - numeric_probability) <= abs(numeric_probability) * Decimal("1e-9") + if abs(number) < sys.float_info.min: + return abs(float(probability) - number) <= math.ulp(number) or relative_match + return relative_match + + +def _sanitize_tier3_probability_fields( + value: dict[str, Any], + *, + numeric_key: str, + text_key: str, + underflow_key: str, +) -> dict[str, Any] | None: + if numeric_key not in value: + return None + raw_number = value.get(numeric_key) + raw_text = value.get(text_key) + parsed_text = _sanitize_tier3_probability_text(raw_text) + if text_key in value and raw_text is not None and parsed_text is None: + return None + raw_underflow = value.get(underflow_key) + if underflow_key in value and not isinstance(raw_underflow, bool): + return None + + if raw_number is None: + if not ( + parsed_text is not None and parsed_text[1] > 0 and float(parsed_text[1]) == 0.0 and raw_underflow is True + ): + return None + return { + numeric_key: None, + text_key: parsed_text[0], + underflow_key: True, + } + + number = _finite_number(raw_number, minimum=0.0, maximum=1.0) + if number is None or number <= 0.0 or raw_underflow is True: + return None + if parsed_text is not None and not _probability_text_matches_number(parsed_text[1], number): + return None + if number < sys.float_info.min and parsed_text is None: + return None + + fields: dict[str, Any] = {numeric_key: number} + if parsed_text is not None: + fields[text_key] = parsed_text[0] + if underflow_key in value: + fields[underflow_key] = False + return fields + + +def _sanitize_tier3_mcnemar_exact(value: object) -> dict[str, Any]: + if not isinstance(value, dict): + return {} + + p_value_fields = _sanitize_tier3_probability_fields( + value, + numeric_key="p_value", + text_key="p_value_text", + underflow_key="p_value_numeric_underflow", + ) + if p_value_fields is None: + return {} + + exact = dict(p_value_fields) + minimum_keys = { + "minimum_attainable_p_value", + "minimum_attainable_p_value_text", + "minimum_attainable_p_value_numeric_underflow", + } + if minimum_keys.intersection(value): + minimum_fields = _sanitize_tier3_probability_fields( + value, + numeric_key="minimum_attainable_p_value", + text_key="minimum_attainable_p_value_text", + underflow_key="minimum_attainable_p_value_numeric_underflow", + ) + if minimum_fields is not None: + exact.update(minimum_fields) + + for key in ("method", "null_hypothesis"): + text = _tier3_display_identifier(value.get(key)) + if text: + exact[key] = text + for key, omitted_key in ( + ("p_value_exact", "p_value_exact_omitted"), + ("minimum_attainable_p_value_exact", "minimum_attainable_p_value_exact_omitted"), + ): + raw_text = value.get(key) + if isinstance(raw_text, str): + text = publication_semantic_text(raw_text).strip() + if text: + exact[key] = text + elif key in value and raw_text is None and value.get(omitted_key) is True: + exact[key] = None + for key in ( + "p_value_exact_omitted", + "p_value_numeric_underflow", + "minimum_attainable_p_value_exact_omitted", + "minimum_attainable_p_value_numeric_underflow", + "resolution_limited_at_alpha_0_05", + ): + if isinstance(value.get(key), bool): + exact[key] = value[key] + for key in ( + "p_value_exact_omitted_reason", + "minimum_attainable_p_value_exact_omitted_reason", + ): + text = _tier3_display_identifier(value.get(key)) + if text: + exact[key] = text + return exact + + +def _sanitize_tier3_paired_comparison(value: object) -> dict[str, Any]: + if not isinstance(value, dict): + return {} + + paired: dict[str, Any] = {} + if "pairing_status" in value: + status = _tier3_display_identifier(value.get("pairing_status")).casefold() + paired["pairing_status"] = status if status in {"complete", "partial", "unavailable"} else "unavailable" + count_keys = ( + "paired_cases", + "with_skill_unpaired_case_count", + "without_skill_unpaired_case_count", + "with_skill_unidentified_cases", + "without_skill_unidentified_cases", + "both_pass", + "with_skill_only_pass", + "without_skill_only_pass", + "neither_pass", + "discordant_cases", + ) + valid_counts: dict[str, int] = {} + for key in count_keys: + if key in value: + count = _strict_nonnegative_count(value.get(key)) + paired[key] = count if count is not None else 0 + if count is not None: + valid_counts[key] = count + for key in ("with_skill_unpaired_case_ids", "without_skill_unpaired_case_ids"): + if key in value: + paired[key] = _tier3_identifier_list(value.get(key)) + for key in ( + "with_skill_unpaired_case_ids_truncated", + "without_skill_unpaired_case_ids_truncated", + ): + if key in value: + paired[key] = value.get(key) is True + outcome_keys = ( + "with_skill_only_pass", + "without_skill_only_pass", + "both_pass", + "neither_pass", + ) + counts_are_consistent = ( + valid_counts.get("paired_cases", 0) > 0 + and all(key in valid_counts for key in (*outcome_keys, "discordant_cases")) + and valid_counts["paired_cases"] == sum(valid_counts[key] for key in outcome_keys) + and valid_counts["discordant_cases"] + == valid_counts["with_skill_only_pass"] + valid_counts["without_skill_only_pass"] + ) + if counts_are_consistent: + paired["paired_rate_delta"] = ( + valid_counts["with_skill_only_pass"] - valid_counts["without_skill_only_pass"] + ) / valid_counts["paired_cases"] + if paired.get("pairing_status") == "complete" and counts_are_consistent: + mcnemar_exact = _sanitize_tier3_mcnemar_exact(value.get("mcnemar_exact")) + if mcnemar_exact: + paired["mcnemar_exact"] = mcnemar_exact + return paired + + +def _sanitize_tier3_pass_at_k(value: object) -> dict[str, Any]: + if not isinstance(value, dict): + return {} + pass_at_k = dict(value) + pass_at_k["with_skill"] = _sanitize_tier3_pass_condition(pass_at_k.get("with_skill")) + pass_at_k["without_skill"] = _sanitize_tier3_pass_condition(pass_at_k.get("without_skill")) + lift = pass_at_k.get("lift") if isinstance(pass_at_k.get("lift"), dict) else {} + delta = _finite_number(lift.get("delta"), minimum=-1.0, maximum=1.0) + if delta is None: + lift.pop("delta", None) + else: + lift["delta"] = delta + paired_comparison = _sanitize_tier3_paired_comparison(lift.get("paired_comparison")) + if paired_comparison: + lift["paired_comparison"] = paired_comparison + else: + lift.pop("paired_comparison", None) + pass_at_k["lift"] = lift + return pass_at_k + + +def _sanitize_tier3_evaluator_card(value: dict[str, Any]) -> dict[str, Any] | None: + card = dict(value) + card["with_skill"] = _finite_number(card.get("with_skill"), minimum=0.0, maximum=1.0) + if card["with_skill"] is None: + return None + card["baseline"] = _finite_number(card.get("baseline"), minimum=0.0, maximum=1.0) + card["lift"] = _finite_number(card.get("lift"), minimum=-1.0, maximum=1.0) + for key in ("id", "label", "evaluator", "status"): + if key in card: + card[key] = _tier3_display_identifier(card[key]) + evidence: list[dict[str, Any]] = [] + for raw_entry in _dict_list(card.get("evidence")): + entry = dict(raw_entry) + entry["entry_id"] = _tier3_display_identifier(entry.get("entry_id")) + occurrences = _nonnegative_count(entry.get("occurrences")) + entry["occurrences"] = occurrences if occurrences > 0 else 1 + entry["score"] = _finite_number(entry.get("score"), minimum=0.0, maximum=1.0) + for key in ("notes", "failures", "checks", "evidence_refs"): + entry[key] = _string_list(entry.get(key)) + evidence.append(entry) + card["evidence"] = evidence + sampling = card.get("evidence_sampling") if isinstance(card.get("evidence_sampling"), dict) else {} + for key in ("represented_cases", "represented_trials", "total_trials", "scanned_trials"): + if key in sampling: + sampling[key] = _nonnegative_count(sampling.get(key)) + sampling["truncated"] = sampling.get("truncated") is True + card["evidence_sampling"] = sampling + return card + + +def _sanitize_tier3_agent(value: dict[str, Any]) -> dict[str, Any]: + agent = dict(value) + for key in ("model", "model_name", "llm_model", "display_name", "label"): + if key not in agent: + continue + if not publication_identity_present(agent[key]): + agent.pop(key) + continue + normalized = publication_semantic_text(agent[key]) + if publication_identity_present(normalized): + agent[key] = normalized + else: + agent.pop(key) + agent["baseline"] = _finite_number(agent.get("baseline"), minimum=0.0, maximum=1.0) + agent["with_skill"] = _finite_number(agent.get("with_skill"), minimum=0.0, maximum=1.0) + agent["overall_score"] = _finite_number(agent.get("overall_score"), minimum=0.0, maximum=1.0) + agent["lift"] = _finite_number(agent.get("lift"), minimum=-1.0, maximum=1.0) + agent["num_trials"] = _nonnegative_count(agent.get("num_trials")) + agent["num_trials_baseline"] = _nonnegative_count(agent.get("num_trials_baseline")) + agent["expected_attempts"] = _nonnegative_count(agent.get("expected_attempts")) + agent["scored_attempts"] = _nonnegative_count(agent.get("scored_attempts")) + agent["dimensions"] = [_sanitize_tier3_dimension(item) for item in _dict_list(agent.get("dimensions"))] + agent["evaluator_cards"] = [ + card + for item in _dict_list(agent.get("evaluator_cards")) + if (card := _sanitize_tier3_evaluator_card(item)) is not None + ] + agent["findings"] = [_sanitize_tier3_dimension(item) for item in _dict_list(agent.get("findings"))] + agent["pass_at_k"] = _sanitize_tier3_pass_at_k(agent.get("pass_at_k")) + return agent + + +def _scrub_untrusted_tier3_agents(value: object) -> dict[str, dict[str, Any]]: + """Retain agent identity/coverage while removing untrusted score evidence.""" + agents: dict[str, dict[str, Any]] = {} + for name, raw_agent in _tier3_agent_entries(value): + agent: dict[str, Any] = { + "baseline": None, + "with_skill": None, + "overall_score": None, + "lift": None, + } + for key in ("model", "model_name", "llm_model", "display_name", "label", "execution_status"): + if isinstance(raw_agent.get(key), str): + agent[key] = raw_agent[key] + for key in ( + "num_trials", + "num_trials_baseline", + "expected_attempts", + "scored_attempts", + ): + agent[key] = _nonnegative_count(raw_agent.get(key)) + for key in ("execution_errors", "warnings"): + agent[key] = _string_list(raw_agent.get(key)) + agent["dimensions"] = [ + { + **{ + key: raw_dimension[key] + for key in ("id", "dimension", "label") + if isinstance(raw_dimension.get(key), str) + }, + "baseline": None, + "with_skill": None, + "score": None, + "lift": None, + "verdict": None, + } + for raw_dimension in _dict_list(raw_agent.get("dimensions")) + ] + agents[name] = _sanitize_tier3_agent(agent) + return agents + + +def _sanitize_tier3_trial(value: dict[str, Any]) -> dict[str, Any]: + trial = dict(value) + for key in ("agent", "entry_id", "trial_id"): + trial[key] = _tier3_display_identifier(trial.get(key)) + trial["overall"] = _finite_number(trial.get("overall"), minimum=0.0, maximum=1.0) + trial["scores"] = _score_mapping(trial.get("scores")) + trial["baseline_scores"] = _score_mapping(trial.get("baseline_scores")) + trial["lift_scores"] = _score_mapping(trial.get("lift_scores"), minimum=-1.0, maximum=1.0) + tokens = trial.get("tokens") if isinstance(trial.get("tokens"), dict) else {} + trial["tokens"] = { + "prompt": _nonnegative_count(tokens.get("prompt")), + "completion": _nonnegative_count(tokens.get("completion")), + } + trial["steps"] = _nonnegative_count(trial.get("steps")) + trial["warnings"] = _string_list(trial.get("warnings")) + recovery = trial.get("error_recovery") if isinstance(trial.get("error_recovery"), dict) else {} + recovery["details"] = _string_list(recovery.get("details")) + trial["error_recovery"] = recovery + return trial + + +def _sanitize_tier3_harbor_viewer(value: object) -> dict[str, Any]: + if not isinstance(value, dict): + return {} + viewer = dict(value) + viewer["jobs"] = _dict_list(viewer.get("jobs")) + viewer["evidence_links"] = _dict_list(viewer.get("evidence_links")) + return viewer + + +def _sanitize_tier3_insight_items(value: object) -> list[dict[str, Any]]: + items = _dict_list(value) + for item in items: + if not isinstance(item.get("evidence"), dict): + item.pop("evidence", None) + if not isinstance(item.get("harbor_evidence"), dict): + item.pop("harbor_evidence", None) + return items + + +def _sanitize_tier3_dataset_case(value: dict[str, Any]) -> dict[str, Any]: + case = dict(value) + case["id"] = _tier3_display_identifier(case.get("id")) + case["assertions"] = _string_list(case.get("assertions")) + case["expected_behavior"] = _string_list(case.get("expected_behavior")) + return case + + +def _sanitize_tier3_display_payload(value: dict[str, Any]) -> dict[str, Any]: + """Normalize template-facing types after merging canonical and fallback data.""" + payload = _json_safe_tier3_payload(value) + payload["summary"] = payload.get("summary") if isinstance(payload.get("summary"), dict) else {} + payload["agents"] = { + identifier: _sanitize_tier3_agent(agent) for identifier, agent in _tier3_agent_entries(payload.get("agents")) + } + for container in (payload, payload["summary"]): + for key in ("skill_name", "evaluator_version", "environment"): + if key not in container: + continue + if key == "skill_name": + exact_name = container[key] + if ( + isinstance(exact_name, str) + and unicodedata.normalize("NFC", exact_name) == exact_name + and publication_identity_present(exact_name) + ): + continue + container.pop(key) + continue + normalized = publication_semantic_text(container[key]) + if publication_identity_present(normalized): + container[key] = normalized + else: + container.pop(key) + evaluators: dict[str, dict[str, float | None]] = {} + raw_evaluators = payload.get("evaluators") + if isinstance(raw_evaluators, dict): + evaluator_names: set[str] = set() + for raw_name, scores in raw_evaluators.items(): + name = _tier3_display_identifier(raw_name) + identity_key = name.casefold() + if not name or not isinstance(scores, dict) or identity_key in evaluator_names: + evaluators = {} + break + evaluator_names.add(identity_key) + evaluators[name] = { + "with_skill": _finite_number(scores.get("with_skill"), minimum=0.0, maximum=1.0), + "baseline": _finite_number(scores.get("baseline"), minimum=0.0, maximum=1.0), + "lift": _finite_number(scores.get("lift"), minimum=-1.0, maximum=1.0), + } + payload["evaluators"] = evaluators + for key in ( + "insights", + "dataset_summary", + "verdict_policy", + "provenance", + ): + if not isinstance(payload.get(key), dict): + payload[key] = {} + payload["metric_labels"] = _tier3_text_mapping(payload.get("metric_labels")) + payload["dimension_hints"] = _tier3_text_mapping(payload.get("dimension_hints")) + evaluator_paths = payload["provenance"].get("evaluator_paths") + payload["provenance"]["evaluator_paths"] = evaluator_paths if isinstance(evaluator_paths, dict) else {} + harbor_viewer = _sanitize_tier3_harbor_viewer(payload.get("harbor_viewer")) + if harbor_viewer: + payload["harbor_viewer"] = harbor_viewer + else: + payload.pop("harbor_viewer", None) + summary_harbor_viewer = _sanitize_tier3_harbor_viewer(payload["summary"].get("harbor_viewer")) + if summary_harbor_viewer: + payload["summary"]["harbor_viewer"] = summary_harbor_viewer + else: + payload["summary"].pop("harbor_viewer", None) + attempt_policy = payload.get("attempt_policy") if isinstance(payload.get("attempt_policy"), dict) else {} + if "max_attempts" in attempt_policy: + max_attempts = _nonnegative_count(attempt_policy.get("max_attempts")) + if max_attempts < 1: + attempt_policy.pop("max_attempts", None) + else: + attempt_policy["max_attempts"] = max_attempts + if "pass_threshold" in attempt_policy: + threshold = _finite_number(attempt_policy.get("pass_threshold"), minimum=0.0, maximum=1.0) + if threshold is None: + attempt_policy["pass_threshold"] = None + else: + attempt_policy["pass_threshold"] = threshold + else: + attempt_policy["pass_threshold"] = None + if "stop_on_pass" in attempt_policy: + attempt_policy["stop_on_pass"] = attempt_policy.get("stop_on_pass") is True + payload["attempt_policy"] = attempt_policy + payload["pass_at_k"] = _sanitize_tier3_pass_at_k(payload.get("pass_at_k")) + payload["evaluator_cards"] = [ + card + for item in _dict_list(payload.get("evaluator_cards")) + if (card := _sanitize_tier3_evaluator_card(item)) is not None + ] + payload["cases"] = _dict_list(payload.get("cases")) + payload["dimensions"] = [ + dimension + for item in _dict_list(payload.get("dimensions")) + if ( + (dimension := _sanitize_tier3_dimension(item)).get("score") is not None + or dimension.get("with_skill") is not None + ) + ] + for key in ("conclusions", "recommendations", "suggestions_v2"): + payload[key] = _sanitize_tier3_insight_items(payload.get(key)) + payload["dataset"] = [_sanitize_tier3_dataset_case(item) for item in _dict_list(payload.get("dataset"))] + payload["trials"] = [_sanitize_tier3_trial(item) for item in _dict_list(payload.get("trials"))] + timing: list[dict[str, Any]] = [] + for item in _dict_list(payload.get("timing")): + entry = dict(item) + entry["seconds"] = _finite_number(entry.get("seconds"), minimum=0.0) + if entry["seconds"] is not None: + timing.append(entry) + payload["timing"] = timing + for key in ("agents_run", "metric_ids"): + payload[key] = _tier3_identifier_list(payload.get(key)) + for key in ("execution_errors", "suggestions"): + payload[key] = _string_list(payload.get(key)) + # ``skill_name`` was already validated above as an exact NFC filesystem + # identity. Do not pass it through compatibility-normalizing display text. + for key in ("best_agent", "verdict", "execution_status"): + for container in (payload, payload["summary"]): + if key not in container: + continue + normalized = _tier3_display_identifier(container.get(key)) + if normalized: + container[key] = normalized + else: + container.pop(key, None) + for key in ("overall_score",): + payload[key] = _finite_number(payload.get(key), minimum=0.0, maximum=1.0) + payload["summary"][key] = _finite_number(payload["summary"].get(key), minimum=0.0, maximum=1.0) + for key in ("overall_lift", "composite_lift"): + payload[key] = _finite_number(payload.get(key), minimum=-1.0, maximum=1.0) + if key in payload["summary"]: + payload["summary"][key] = _finite_number( + payload["summary"].get(key), + minimum=-1.0, + maximum=1.0, + ) + payload["runtime_seconds"] = _finite_number(payload.get("runtime_seconds"), minimum=0.0) or 0.0 + payload["summary"]["runtime_seconds"] = ( + _finite_number(payload["summary"].get("runtime_seconds"), minimum=0.0) or 0.0 + ) + for key in ("expected_attempts", "scored_attempts"): + for container in (payload, payload["summary"]): + if key in container: + container[key] = _nonnegative_count(container.get(key)) + truncation = payload.get("report_truncation") if isinstance(payload.get("report_truncation"), dict) else {} + omitted = truncation.get("omitted") if isinstance(truncation.get("omitted"), dict) else {} + sanitized_omitted = { + str(section): count for section, value in omitted.items() if (count := _nonnegative_count(value)) > 0 + } + if sanitized_omitted: + truncation["omitted"] = sanitized_omitted + else: + truncation.pop("omitted", None) + if truncation: + payload["report_truncation"] = truncation + else: + payload.pop("report_truncation", None) + return payload @dataclass @@ -142,12 +1144,6 @@ def _bounded_tier3_preview(payload: dict[str, Any] | None) -> tuple[dict[str, An return preview, notice -def is_tier2_validator_name(validator_name: str | None) -> bool: - """Return whether a validator name belongs to Tier 2 reporting.""" - normalized_name = " ".join((validator_name or "").casefold().replace("_", " ").replace("-", " ").split()) - return any(marker in normalized_name for marker in _TIER2_VALIDATOR_MARKERS) - - def _related_paths(finding: object) -> list[str]: """Return distinct path-like string values carried in finding metadata.""" metadata = finding.get("metadata", {}) if isinstance(finding, dict) else getattr(finding, "metadata", {}) @@ -277,6 +1273,7 @@ def __init__( target_path: str | None = None, content_label: str = "Skill", profile: str | None = None, + expected_skill_name: str | None = None, ) -> None: self.include_timestamp = include_timestamp self.title = title or "SkillEvaluator Validation Report" @@ -288,14 +1285,17 @@ def __init__( # in the report header so reviewers can tell which audience the # validation gate was applied for. self.profile = profile + self.expected_skill_name = expected_skill_name self._env = self._create_environment() def _create_environment(self) -> Environment: loader = PackageLoader("skillevaluator.reporting", "templates") environment = Environment(loader=loader, autoescape=True) + environment.filters["cleanly_skipped"] = is_cleanly_skipped environment.filters["related_paths"] = _related_paths environment.filters["adaptive_percent"] = _adaptive_percent environment.filters["adaptive_interval_percent"] = _adaptive_interval_percent + environment.filters["skip_reason"] = get_skip_reason return environment @staticmethod @@ -913,8 +1913,6 @@ def _compute_friendly_skill_label(target: str | None) -> str: # to bucket combined run results back into per-tier summaries for the # hero card so the chip for "Tier 1" reflects only Tier 1 validators # instead of leaking Tier 2 / Tier 3 stats from the global totals. - _TIER3_VALIDATOR_NAMES: ClassVar[frozenset[str]] = frozenset({"AGENT_EVAL"}) - @classmethod def _split_results_by_tier( cls, results: list[ValidationResult] @@ -930,10 +1928,9 @@ def _split_results_by_tier( tier2: list[ValidationResult] = [] tier3: list[ValidationResult] = [] for r in results: - name = getattr(r, "validator_name", None) or "" - if name in cls._TIER3_VALIDATOR_NAMES: + if is_tier3_result(r): tier3.append(r) - elif is_tier2_validator_name(name): + elif is_tier2_result(r): tier2.append(r) else: tier1.append(r) @@ -949,9 +1946,12 @@ def _compute_tier_summary(results: list[ValidationResult]) -> dict[str, Any]: extra "did this tier run?" predicate. """ total = len(results) - passed_count = sum(1 for r in results if r.passed) + skipped_count = sum(1 for r in results if is_cleanly_skipped(r)) + executed_count = total - skipped_count + passed_count = sum(1 for r in results if r.passed and not r.is_incomplete and not is_cleanly_skipped(r)) advisory_skipped_count = sum(1 for r in results if is_advisory_agent_eval_skip(r)) - failed_count = sum(1 for r in results if not passes_required_gate(r)) + failed_count = sum(1 for r in results if not r.passed and not is_cleanly_skipped(r)) + incomplete_count = sum(1 for r in results if r.is_incomplete) issue_count = 0 critical = high = medium = low = 0 for r in results: @@ -968,12 +1968,14 @@ def _compute_tier_summary(results: list[ValidationResult]) -> dict[str, Any]: low += 1 return { "total": total, + "executed_count": executed_count, "passed_count": passed_count, + "skipped_count": skipped_count, "advisory_skipped_count": advisory_skipped_count, "failed_count": failed_count, "issue_count": issue_count, - "all_passed": total > 0 and failed_count == 0, - "incomplete_count": sum(1 for r in results if r.is_incomplete), + "all_passed": executed_count > 0 and failed_count == 0 and incomplete_count == 0, + "incomplete_count": incomplete_count, "critical": critical, "high": high, "medium": medium, @@ -983,11 +1985,14 @@ def _compute_tier_summary(results: list[ValidationResult]) -> dict[str, Any]: def _results_to_dict(self, results: list[ValidationResult]) -> list[dict[str, Any]]: output = [] for result in results: + skipped = is_cleanly_skipped(result) result_dict = { "validator_name": result.validator_name, "validator_description": result.validator_description, "passed": result.passed, - "status": "skipped" if is_advisory_agent_eval_skip(result) else result.status, + "status": "skipped" if skipped else result.status, + "skipped": skipped, + "skip_reason": get_skip_reason(result) if skipped else None, "incomplete_scans": result.incomplete_scans, "summary": { "files_scanned": result.summary.files_scanned, @@ -1005,6 +2010,15 @@ def _results_to_dict(self, results: list[ValidationResult]) -> list[dict[str, An "errors": result.errors, "warnings": result.warnings, } + publication_target = result_publication_target_dict(result) + if publication_target is not None: + result_dict["publication_target"] = publication_target + publication_target_conflict = result_publication_target_conflict_marker(result) + if publication_target_conflict is not None: + result_dict["publication_target_conflict"] = publication_target_conflict + publication_evidence = result_publication_evidence_dict(result) + if publication_evidence is not None: + result_dict["publication_evidence"] = publication_evidence for finding in result.findings: result_dict["findings"].append( { @@ -1033,64 +2047,369 @@ def _results_to_dict(self, results: list[ValidationResult]) -> list[dict[str, An @staticmethod def _tier3_report_data(results: list[ValidationResult]) -> dict[str, Any] | None: - """Return canonical Tier 3 data, with a visible fallback for bare results.""" - for result in results: - payload = result.metadata.get("agent_eval") if result.metadata else None - if isinstance(payload, dict) and payload: - return normalize_agent_eval_harbor_links(payload) - + """Return safe Tier 3 display data without certifying partial evidence.""" if not results: return None - # A validator failure should remain visible even if normalization - # failed before canonical ``agent_eval`` metadata could be attached. - result = results[0] - verdict = "pass" if result.passed else "fail" - execution_status = "succeeded" if result.passed else "failed" - messages = [*result.errors, *result.warnings, *result.messages] - message = messages[0] if messages else "Tier 3 did not provide canonical evaluation details." - return { - "schema_version": "2.0", - "summary": { + def fallback(verdict: str, execution_status: str) -> dict[str, Any]: + messages = [ + message for result in results for message in (*result.errors, *result.warnings, *result.messages) + ] + if execution_status == "skipped": + skipped_result = next(result for result in results if is_cleanly_skipped(result)) + message = get_skip_reason(skipped_result) + else: + incomplete_scans = list(dict.fromkeys(tool for result in results for tool in result.incomplete_scans)) + if incomplete_scans: + message = f"Missing trustworthy evidence from {', '.join(incomplete_scans)}." + else: + message = messages[0] if messages else "Tier 3 did not provide canonical evaluation details." + provenance = {"source": "validation_result", "message": message} + if execution_status == "skipped": + provenance.update({"reason": "skipped", "advisory": True}) + execution_errors = [error for result in results for error in result.errors] + return { "schema_version": "2.0", + "summary": { + "schema_version": "2.0", + "verdict": verdict, + "skill_name": "", + "best_agent": "", + "agents_run": [], + "overall_score": None, + "overall_lift": None, + "environment": None, + "runtime_seconds": 0.0, + "execution_status": execution_status, + "execution_errors": execution_errors, + "expected_attempts": 0, + "scored_attempts": 0, + }, + "skill_name": "", "verdict": verdict, + "best_agent": "", + "agents_run": [], + "environment": None, + "overall_score": None, + "overall_lift": None, + "composite_lift": None, "execution_status": execution_status, - "execution_errors": list(result.errors), - }, - "skill_name": "", - "verdict": verdict, - "overall_score": None, - "overall_lift": None, - "execution_status": execution_status, - "execution_errors": list(result.errors), - "agents_run": [], - "agents": {}, - "dimensions": [], - "evaluators": {}, - "evaluator_cards": [], - "cases": [], - "suggestions": messages, - "metric_ids": [], - "metric_labels": {}, - "dataset": [], - "provenance": {"source": "validation_result", "message": message}, - } + "execution_errors": execution_errors, + "expected_attempts": 0, + "scored_attempts": 0, + "runtime_seconds": 0.0, + "agents": {}, + "dimensions": [], + "evaluators": {}, + "evaluator_cards": [], + "cases": [], + "trials": [], + "insights": {}, + "conclusions": [], + "recommendations": [], + "suggestions": messages, + "suggestions_v2": [], + "metric_ids": [], + "metric_labels": {}, + "attempt_policy": {}, + "dataset": [], + "dataset_summary": { + "total_tasks": 0, + "positive_tasks": 0, + "negative_tasks": 0, + "unclassified_tasks": 0, + "source": "unavailable", + }, + "verdict_policy": {}, + "provenance": provenance, + } + + def payload_summary(candidate: dict[str, Any]) -> dict[str, Any]: + summary = candidate.get("summary") + return summary if isinstance(summary, dict) else {} + + def payload_verdict(candidate: dict[str, Any]) -> str: + summary = payload_summary(candidate) + value = candidate.get("verdict") or summary.get("verdict") or "" + return value.lower() if isinstance(value, str) else "" + + def payload_execution_status(candidate: dict[str, Any]) -> str: + summary = payload_summary(candidate) + value = candidate.get("execution_status") or summary.get("execution_status") or "" + return value.lower() if isinstance(value, str) else "" + + def result_has_execution_evidence(result: ValidationResult) -> bool: + return bool(result.success_details or result.findings or result.summary.checks_performed > 0) + + def payload_has_agent_evidence(candidate: dict[str, Any]) -> bool: + """Require a succeeded, scored agent with a positive attempt count.""" + summary = payload_summary(candidate) + agents = candidate.get("agents") + if not isinstance(agents, dict): + return False + payload_attempts = max( + _nonnegative_count(candidate.get("scored_attempts")), + _nonnegative_count(summary.get("scored_attempts")), + ) + for agent in agents.values(): + if not isinstance(agent, dict): + continue + status = agent.get("execution_status") + if not isinstance(status, str) or status.lower() != "succeeded": + continue + score = _finite_number(agent.get("with_skill"), minimum=0.0, maximum=1.0) + if score is None: + score = _finite_number(agent.get("overall_score"), minimum=0.0, maximum=1.0) + attempts = max(payload_attempts, _nonnegative_count(agent.get("scored_attempts"))) + if score is not None and attempts > 0: + return True + return False + + def payload_has_execution_evidence( + result: ValidationResult, + candidate: dict[str, Any], + ) -> bool: + if agent_eval_publication_evidence_complete(candidate): + return True + if payload_verdict(candidate) not in {"pass", "neutral", "fail"}: + return False + execution_status = payload_execution_status(candidate) + if execution_status == "succeeded": + return payload_has_agent_evidence(candidate) + return not execution_status and result_has_execution_evidence(result) + + selected = select_agent_eval_candidate(results) + payload_result, payload = selected if selected is not None else (None, None) + serialization_issue = agent_eval_report_serialization_issue(payload) if payload is not None else None + rejected_assessment = ( + assess_tier3_evidence(results, payload) if payload is not None and serialization_issue is not None else None + ) + bounded_rejected_has_execution_evidence = False + if payload is not None and serialization_issue is not None: + bounded_rejected_payload = _json_safe_tier3_payload(payload) + bounded_evidence_candidate = dict(bounded_rejected_payload) + bounded_evidence_candidate.pop("_serialization_truncated", None) + bounded_rejected_has_execution_evidence = payload_has_execution_evidence( + payload_result, + bounded_evidence_candidate, + ) + selected_verdict = ( + rejected_assessment.verdict + if rejected_assessment is not None + else payload_verdict(payload) + if payload is not None + else "" + ) + selected_execution_status = ( + rejected_assessment.execution_status + if rejected_assessment is not None + else payload_execution_status(payload) + if payload is not None + else "" + ) + has_trustworthy_payload = bool( + payload_result is not None + and payload is not None + and serialization_issue is None + and payload_has_execution_evidence(payload_result, payload) + ) + has_explicit_failure_payload = bool( + payload_result is not None + and payload is not None + and ( + rejected_assessment.status == "fail" + if rejected_assessment is not None + else selected_execution_status in {"failed", "incomplete"} + ) + ) + + def normalized_payload( + *, + forced_verdict: str | None = None, + forced_execution_status: str | None = None, + scrub_untrusted_evidence: bool = False, + ) -> dict[str, Any]: + assert payload is not None + verdict = selected_verdict or "incomplete" + execution_status = selected_execution_status or "incomplete" + display = fallback(forced_verdict or verdict, forced_execution_status or execution_status) + normalized = normalize_agent_eval_harbor_links(_sanitize_tier3_display_payload(payload)) + normalized_summary = normalized.get("summary") if isinstance(normalized.get("summary"), dict) else {} + display.update(normalized) + display["summary"] = {**display["summary"], **normalized_summary} + + def merge_strings(value: object, additions: list[str]) -> list[str]: + existing = value if isinstance(value, list) else [] + return list(dict.fromkeys(item for item in (*existing, *additions) if isinstance(item, str) and item)) + + if forced_verdict is not None: + display["verdict"] = forced_verdict + display["summary"]["verdict"] = forced_verdict + if forced_execution_status is not None: + display["execution_status"] = forced_execution_status + display["summary"]["execution_status"] = forced_execution_status + display = _sanitize_tier3_display_payload(display) + if scrub_untrusted_evidence: + preserved = display + display = fallback( + forced_verdict or verdict, + forced_execution_status or execution_status, + ) + display["agents"] = _scrub_untrusted_tier3_agents(preserved.get("agents")) + display["agents_run"] = list(display["agents"]) + for key in ( + "schema_version", + "skill_name", + "publication_target", + "publication_target_conflict", + "run_id", + "environment", + "evaluated_at", + "evaluator_version", + "benchmark_policy", + "attempt_policy", + "provenance", + "harbor_viewer", + ): + if key in preserved: + display[key] = preserved[key] + for key in ( + "schema_version", + "skill_name", + "publication_target", + "publication_target_conflict", + "run_id", + "environment", + "evaluated_at", + "evaluator_version", + "benchmark_policy", + ): + if key in preserved["summary"]: + display["summary"][key] = preserved["summary"][key] + if "harbor_viewer" in preserved["summary"]: + display["summary"]["harbor_viewer"] = preserved["summary"]["harbor_viewer"] + if preserved.get("_serialization_truncated") is True: + display["_serialization_truncated"] = True + + result_errors = [error for result in results for error in result.errors] + result_diagnostics = [ + message for result in results for message in (*result.errors, *result.warnings, *result.messages) + ] + display["execution_errors"] = merge_strings(display.get("execution_errors"), result_errors) + display["summary"]["execution_errors"] = merge_strings( + display["summary"].get("execution_errors"), + result_errors, + ) + display["suggestions"] = merge_strings(display.get("suggestions"), result_diagnostics) + return _sanitize_tier3_display_payload(display) + + # Incomplete scanner evidence takes precedence over failures, matching + # ValidationResult.status and the publication benchmark. + if any(result.is_incomplete for result in results): + if payload is not None: + return normalized_payload( + forced_verdict="incomplete", + forced_execution_status="incomplete", + scrub_untrusted_evidence=not has_trustworthy_payload, + ) + return fallback("incomplete", "incomplete") + + # Aggregate all Tier 3 results before falling back so the display does + # not depend on result ordering. + if any(not result.passed and not is_cleanly_skipped(result) for result in results): + if payload is not None: + return normalized_payload( + forced_verdict="fail", + forced_execution_status="failed", + scrub_untrusted_evidence=not (has_trustworthy_payload or has_explicit_failure_payload), + ) + return fallback("fail", "failed") + + if all(is_cleanly_skipped(result) for result in results): + if payload is not None: + return normalized_payload( + forced_verdict="skipped", + forced_execution_status="skipped", + scrub_untrusted_evidence=True, + ) + return fallback("skipped", "skipped") + + if payload is not None: + verdict = selected_verdict + execution_status = selected_execution_status + if serialization_issue is not None: + if rejected_assessment is not None and rejected_assessment.status == "fail": + return normalized_payload( + forced_verdict="fail", + forced_execution_status="failed", + scrub_untrusted_evidence=True, + ) + return normalized_payload( + forced_verdict=None if bounded_rejected_has_execution_evidence else "incomplete", + forced_execution_status=None if bounded_rejected_has_execution_evidence else "incomplete", + scrub_untrusted_evidence=True, + ) + if verdict == "fail" or execution_status == "failed": + return normalized_payload( + forced_verdict="fail", + forced_execution_status="failed", + scrub_untrusted_evidence=not (has_trustworthy_payload or has_explicit_failure_payload), + ) + if execution_status == "incomplete": + return normalized_payload( + forced_verdict="incomplete", + forced_execution_status="incomplete", + scrub_untrusted_evidence=not has_trustworthy_payload, + ) + if has_trustworthy_payload: + return normalized_payload( + forced_execution_status="incomplete" if not execution_status else None, + ) + return normalized_payload( + forced_verdict="incomplete", + forced_execution_status="incomplete", + scrub_untrusted_evidence=True, + ) + + # A default-passed bare result or partial payload is not proof that + # Tier 3 completed. + return fallback("incomplete", "incomplete") def render(self, result: ValidationResult) -> str: return self.render_all([result]) def render_all(self, results: list[ValidationResult]) -> str: + tier1_results, tier2_results, tier3_results = self._split_results_by_tier(results) + tier3_data = self._tier3_report_data(tier3_results) + # Preserve the raw policy/provenance assessment, then let the emitted + # payload make the decision only more conservative. Display shaping + # must never invent a successful run or erase a persisted waiver. + raw_publication = assess_publication(results, expected_skill_name=self.expected_skill_name) + emitted_publication = assess_publication( + results, + tier3_data, + expected_skill_name=self.expected_skill_name, + ) + publication = max( + (raw_publication, emitted_publication), + key=lambda assessment: {"pass": 0, "neutral": 1, "incomplete": 2, "fail": 3}.get( + assessment.status, + 3, + ), + ) + all_passed = all(passes_required_gate(r) for r in results) has_incomplete = any(r.is_incomplete for r in results) overall_status = "incomplete" if has_incomplete else "passed" if all_passed else "failed" total_errors = sum(r.summary.errors for r in results) total_warnings = sum(r.summary.warnings for r in results) - passed_count = sum(1 for r in results if r.passed) + skipped_count = sum(1 for r in results if is_cleanly_skipped(r)) + passed_count = sum(1 for r in results if r.passed and not r.is_incomplete and not is_cleanly_skipped(r)) advisory_skipped_count = sum(1 for r in results if is_advisory_agent_eval_skip(r)) failed_count = sum(1 for r in results if not passes_required_gate(r)) total_issues = total_errors + total_warnings total_validators = len(results) - executed_count = total_validators - advisory_skipped_count + executed_count = total_validators - skipped_count pass_percentage = round((passed_count / executed_count * 100) if executed_count > 0 else 0, 1) timestamp = "" @@ -1139,11 +2458,9 @@ def render_all(self, results: list[ValidationResult]) -> str: # Per-tier summaries so the hero card chips can show "Tier 1: 6/7 # passed (0 critical)" etc. without leaking Tier 2 / Tier 3 stats # from the global aggregate counters. - tier1_results, tier2_results, tier3_results = self._split_results_by_tier(results) tier1_summary = self._compute_tier_summary(tier1_results) tier2_summary = self._compute_tier_summary(tier2_results) tier3_summary = self._compute_tier_summary(tier3_results) - tier3_data = self._tier3_report_data(tier3_results) tier3_preview, tier3_preview_notice = _bounded_tier3_preview(tier3_data) tier3_canonical_data, tier3_canonical_encoding = _canonical_tier3_embed(tier3_data) tier3_truncation = tier3_data.get("report_truncation", {}) if isinstance(tier3_data, dict) else {} @@ -1155,14 +2472,23 @@ def render_all(self, results: list[ValidationResult]) -> str: tier1_skills = self._reorganize_by_skill(tier1_display_results) tier1_top_issues = self._compute_top_issues(tier1_skills) tier1_contributors = self._extract_contributors(tier1_skills, tier1_display_results) - tier1_display_total = len(tier1_display_results) - tier1_display_passed = sum(1 for result in tier1_display_results if result.passed) + tier1_display_skipped = sum(1 for result in tier1_display_results if is_cleanly_skipped(result)) + tier1_display_total = len(tier1_display_results) - tier1_display_skipped + tier1_display_passed = sum( + 1 + for result in tier1_display_results + if result.passed and not result.is_incomplete and not is_cleanly_skipped(result) + ) + tier1_display_failed = sum( + 1 for result in tier1_display_results if not result.passed and not is_cleanly_skipped(result) + ) tier1_display_total_skills = len(tier1_skills) tier1_display_passed_skills = sum(1 for skill in tier1_skills.values() if skill["passed"]) tier1_display_summary = { "total_validators": tier1_display_total, "passed_count": tier1_display_passed, - "failed_count": tier1_display_total - tier1_display_passed, + "skipped_count": tier1_display_skipped, + "failed_count": tier1_display_failed, "total_issues": sum(result.summary.errors + result.summary.warnings for result in tier1_display_results), "pass_percentage": round( (tier1_display_passed / tier1_display_total * 100) if tier1_display_total else 0, @@ -1213,9 +2539,7 @@ def _severity_totals(tier_results: list[ValidationResult]) -> dict[str, int]: for result in tier_results: for finding in result.findings: severity = ( - finding.severity.value - if hasattr(finding.severity, "value") - else str(finding.severity).lower() + finding.severity.value if hasattr(finding.severity, "value") else str(finding.severity).lower() ) if severity in totals: totals[severity] += 1 @@ -1237,7 +2561,7 @@ def _severity_totals(tier_results: list[ValidationResult]) -> dict[str, int]: "blocking": blocking_totals, "advisory": advisory_totals, "blocking_findings": blocking_totals["critical"] + blocking_totals["high"], - "would_block": any(not result.passed for result in blocking_results), + "would_block": any(not passes_required_gate(result) for result in blocking_results), } # Extract quality scores from results for per-skill quality display @@ -1289,7 +2613,9 @@ def _attach_quality_scores(skills: dict[str, dict[str, Any]]) -> None: "status": overall_status, "incomplete_scans": list(dict.fromkeys(tool for result in results for tool in result.incomplete_scans)), "total_validators": total_validators, + "executed_count": executed_count, "passed_count": passed_count, + "skipped_count": skipped_count, "advisory_skipped_count": advisory_skipped_count, "failed_count": failed_count, "total_issues": total_issues, @@ -1297,6 +2623,7 @@ def _attach_quality_scores(skills: dict[str, dict[str, Any]]) -> None: "total_skills": total_skills, "passed_skills": passed_skills, "failed_skills": failed_skills, + "publication_status": publication.status, }, "results": self._results_to_dict(results), "skills": skills_by_name, @@ -1308,6 +2635,16 @@ def _attach_quality_scores(skills: dict[str, dict[str, Any]]) -> None: # copy inside ``#report-data``. "tier3": {"$ref": "#tier3-full"} if tier3_data else None, "gating": gating, + "publication": { + "status": publication.status, + "eligible": publication.status == "pass", + "reasons": list(publication.reasons), + "tier3": { + "status": publication.tier3.status, + "evidence_complete": publication.tier3.evidence_complete, + "reason": publication.tier3.reason, + }, + }, } report_json = ( json.dumps(report_data, indent=2, allow_nan=False) @@ -1347,8 +2684,14 @@ def _attach_quality_scores(skills: dict[str, dict[str, Any]]) -> None: all_passed=all_passed, has_incomplete=has_incomplete, overall_status=overall_status, + publication_status=publication.status, + publication_reasons=publication.reasons, + tier3_effective_status=publication.tier3.status, total_validators=total_validators, + executed_count=executed_count, passed_count=passed_count, + skipped_count=skipped_count, + advisory_skipped_count=advisory_skipped_count, failed_count=failed_count, total_issues=total_issues, pass_percentage=pass_percentage, diff --git a/src/skillevaluator/reporting/json_reporter.py b/src/skillevaluator/reporting/json_reporter.py index 2fd32749..77eff02b 100644 --- a/src/skillevaluator/reporting/json_reporter.py +++ b/src/skillevaluator/reporting/json_reporter.py @@ -18,12 +18,42 @@ from datetime import UTC, datetime from typing import TYPE_CHECKING, Any -from skillevaluator.reporting.base import ReporterBase, is_advisory_agent_eval_skip, passes_required_gate +from skillevaluator.publication_evidence import result_publication_evidence_dict +from skillevaluator.reporting.base import ( + ReporterBase, + assess_publication, + get_skip_reason, + is_advisory_agent_eval_skip, + is_cleanly_skipped, + passes_required_gate, + result_publication_target_conflict_marker, + result_publication_target_dict, + select_agent_eval_payload, +) if TYPE_CHECKING: from skillevaluator.models import ValidationResult +def _json_safe_agent_eval(value: object) -> dict[str, Any]: + """Bound and normalize Tier 3 metadata before machine-readable embedding.""" + from skillevaluator.reporting.html import _json_safe_tier3_payload + + return _json_safe_tier3_payload(value) + + +def _json_safe_benchmark_policy(value: object) -> dict[str, bool] | None: + """Project untrusted per-result policy metadata onto its two typed keys.""" + if not isinstance(value, dict): + return None + policy = { + key: candidate + for key in ("tier2_required", "tier3_required") + if isinstance((candidate := value.get(key)), bool) + } + return policy or None + + class JSONReporter(ReporterBase): """JSON export for machine-readable output. @@ -31,7 +61,13 @@ class JSONReporter(ReporterBase): Supports both compact and pretty-printed output formats. """ - def __init__(self, *, indent: int | None = 2, include_timestamp: bool = True) -> None: + def __init__( + self, + *, + indent: int | None = 2, + include_timestamp: bool = True, + expected_skill_name: str | None = None, + ) -> None: """Initialize JSON reporter. Args: @@ -40,6 +76,7 @@ def __init__(self, *, indent: int | None = 2, include_timestamp: bool = True) -> """ self.indent = indent self.include_timestamp = include_timestamp + self.expected_skill_name = expected_skill_name @property def name(self) -> str: @@ -61,6 +98,7 @@ def render_all(self, results: list[ValidationResult]) -> str: from skillevaluator.reporting.html import HTMLReporter all_passed = all(passes_required_gate(r) for r in results) + skip_count = sum(1 for r in results if is_cleanly_skipped(r)) advisory_skip_count = sum(1 for r in results if is_advisory_agent_eval_skip(r)) incomplete_scans = list(dict.fromkeys(tool for result in results for tool in result.incomplete_scans)) overall_status = "incomplete" if incomplete_scans else "passed" if all_passed else "failed" @@ -90,6 +128,7 @@ def render_all(self, results: list[ValidationResult]) -> str: "overall_status": overall_status, "incomplete_scans": incomplete_scans, "total_validators": len(results), + "total_skipped": skip_count, "total_advisory_skipped": advisory_skip_count, "total_errors": total_errors, "total_warnings": total_warnings, @@ -104,12 +143,45 @@ def render_all(self, results: list[ValidationResult]) -> str: "results": [self._result_to_dict(r) for r in results], } + agent_eval = select_agent_eval_payload(results) + emitted_agent_eval = _json_safe_agent_eval(agent_eval) if agent_eval else None + # Publication claims must be supported by both the raw evidence and the + # normalized evidence actually emitted to consumers. Normalization must + # neither invent proof from malformed keys/values nor hide proof through + # truncation, so retain the more conservative assessment. + raw_publication = assess_publication( + results, + agent_eval, + expected_skill_name=self.expected_skill_name, + ) + emitted_publication = assess_publication( + results, + emitted_agent_eval, + expected_skill_name=self.expected_skill_name, + ) + status_rank = {"pass": 0, "neutral": 1, "incomplete": 2, "fail": 3} + publication = ( + emitted_publication + if status_rank[emitted_publication.status] > status_rank[raw_publication.status] + else raw_publication + ) + data["benchmark_policy"] = publication.benchmark_policy + data["publication_status"] = publication.status + data["publication"] = { + "status": publication.status, + "eligible": publication.status == "pass", + "reasons": list(publication.reasons), + "tier3": { + "status": publication.tier3.status, + "evidence_complete": publication.tier3.evidence_complete, + "execution_status": publication.tier3.execution_status, + "verdict": publication.tier3.verdict, + "reason": publication.tier3.reason, + }, + } + policy = next( - ( - result.metadata.get("policy") - for result in results - if isinstance(result.metadata.get("policy"), dict) - ), + (result.metadata.get("policy") for result in results if isinstance(result.metadata.get("policy"), dict)), None, ) if policy is not None: @@ -140,9 +212,8 @@ def render_all(self, results: list[ValidationResult]) -> str: data["rubric_eval"] = rubric_results[0] # Tier 3: Agent evaluation summary - tier3_results = [r.metadata["agent_eval"] for r in results if r.metadata.get("agent_eval")] - if tier3_results: - data["tier3"] = tier3_results[0] + if emitted_agent_eval: + data["tier3"] = emitted_agent_eval applicability = next( ( result.metadata.get("tier3_applicability") @@ -161,11 +232,14 @@ def render_all(self, results: list[ValidationResult]) -> str: def _result_to_dict(self, result: ValidationResult) -> dict[str, Any]: """Convert ValidationResult to serializable dictionary.""" + skipped = is_cleanly_skipped(result) data: dict[str, Any] = { "validator": result.validator_name, "description": result.validator_description, "passed": result.passed, - "status": "skipped" if is_advisory_agent_eval_skip(result) else result.status, + "status": "skipped" if skipped else result.status, + "skipped": skipped, + "skip_reason": get_skip_reason(result) if skipped else None, "incomplete_scans": result.incomplete_scans, "summary": { "files_scanned": result.summary.files_scanned, @@ -215,7 +289,7 @@ def _result_to_dict(self, result: ValidationResult) -> dict[str, Any]: # Tier 3: Agent evaluation data ae = result.metadata.get("agent_eval") if ae: - data["tier3"] = ae + data["tier3"] = _json_safe_agent_eval(ae) rubric = result.metadata.get("rubric_eval") if rubric: @@ -225,6 +299,20 @@ def _result_to_dict(self, result: ValidationResult) -> dict[str, Any]: if isinstance(gating, dict): data["gating"] = gating + benchmark_policy = _json_safe_benchmark_policy(result.metadata.get("benchmark_policy")) + if benchmark_policy is not None: + data["benchmark_policy"] = benchmark_policy + + publication_target = result_publication_target_dict(result) + if publication_target is not None: + data["publication_target"] = publication_target + publication_target_conflict = result_publication_target_conflict_marker(result) + if publication_target_conflict is not None: + data["publication_target_conflict"] = publication_target_conflict + publication_evidence = result_publication_evidence_dict(result) + if publication_evidence is not None: + data["publication_evidence"] = publication_evidence + return data def get_file_extension(self) -> str: diff --git a/src/skillevaluator/reporting/markdown.py b/src/skillevaluator/reporting/markdown.py index f5d34a24..c1d6b6bc 100644 --- a/src/skillevaluator/reporting/markdown.py +++ b/src/skillevaluator/reporting/markdown.py @@ -15,10 +15,20 @@ from __future__ import annotations import html +import math +import unicodedata from datetime import UTC, datetime from typing import TYPE_CHECKING -from skillevaluator.reporting.base import ReporterBase, is_advisory_agent_eval_skip, passes_required_gate +from skillevaluator.reporting.base import ( + ReporterBase, + assess_publication, + get_skip_reason, + is_advisory_agent_eval_skip, + is_cleanly_skipped, + passes_required_gate, + select_agent_eval_payload, +) from skillevaluator.reporting.harbor_viewer import ( harbor_evidence_link_text, normalize_harbor_viewer_for_display, @@ -36,6 +46,78 @@ def _markdown_table_cell(value: object) -> str: return escaped.replace("|", "|").replace("`", "`").replace("\n", "
") +def _finite_report_number(value: object) -> float | None: + if isinstance(value, bool) or not isinstance(value, (int, float)): + return None + try: + number = float(value) + except (OverflowError, ValueError): + return None + return number if math.isfinite(number) else None + + +def _format_report_number(value: object, *, signed: bool = False) -> str: + number = _finite_report_number(value) + if number is None: + return "N/A" + return f"{number:+.2f}" if signed else f"{number:.2f}" + + +def _markdown_inline_text(value: object, *, limit: int | None = None) -> str: + """Flatten untrusted metadata into one inert Markdown text fragment.""" + if isinstance(value, str): + text = value + elif ( + isinstance(value, bool) + or (isinstance(value, int) and value.bit_length() <= 256) + or (isinstance(value, float) and math.isfinite(value)) + ): + text = str(value) + else: + return "" + flattened = " ".join(text.replace("\r\n", "\n").replace("\r", "\n").split()) + if limit is not None: + flattened = flattened[:limit] + escaped = html.escape(flattened, quote=False) + for character, entity in ( + ("#", "#"), + ("\\", "\"), + ("`", "`"), + ("*", "*"), + ("_", "_"), + ("[", "["), + ("]", "]"), + ("@", "@"), + ): + escaped = escaped.replace(character, entity) + return escaped + + +def _mapping(value: object) -> dict: + return value if isinstance(value, dict) else {} + + +def _dict_items(value: object) -> list[dict]: + return [item for item in value if isinstance(item, dict)] if isinstance(value, list) else [] + + +def _markdown_untrusted_table_cell(value: object, *, limit: int | None = None) -> str: + """Return one inert table cell without double-escaping HTML entities.""" + return _markdown_inline_text(value, limit=limit).replace("|", "|") + + +def _markdown_safe_url(value: object) -> str | None: + """Return an HTTP(S) URL that cannot terminate a Markdown destination.""" + url = safe_url(value) + if url is None: + return None + if any(character.isspace() or unicodedata.category(character).startswith("C") for character in url): + return None + if any(character in "()<>" for character in url): + return None + return url + + def _related_paths(finding: Finding) -> list[str]: """Return distinct path-like string values carried in finding metadata.""" metadata = finding.metadata if isinstance(finding.metadata, dict) else {} @@ -62,6 +144,7 @@ def __init__( include_timestamp: bool = True, include_details: bool = True, max_findings_shown: int = 10, + expected_skill_name: str | None = None, ) -> None: """Initialize Markdown reporter. @@ -73,6 +156,7 @@ def __init__( self.include_timestamp = include_timestamp self.include_details = include_details self.max_findings_shown = max_findings_shown + self.expected_skill_name = expected_skill_name @property def name(self) -> str: @@ -99,16 +183,23 @@ def render_all(self, results: list[ValidationResult]) -> str: # Overall status all_passed = all(passes_required_gate(r) for r in results) has_incomplete = any(r.is_incomplete for r in results) + skip_count = sum(1 for r in results if is_cleanly_skipped(r)) advisory_skip_count = sum(1 for r in results if is_advisory_agent_eval_skip(r)) + executed_count = len(results) - skip_count + passed_count = sum(1 for r in results if r.passed and not r.is_incomplete and not is_cleanly_skipped(r)) status = "⚠️ INCOMPLETE" if has_incomplete else "✅ PASSED" if all_passed else "❌ FAILED" lines.append(f"**Status:** {status}") + publication = assess_publication(results, expected_skill_name=self.expected_skill_name) + publication_label = { + "pass": "✅ PASS", + "fail": "❌ FAIL", + "neutral": "⚠️ NEUTRAL", + "incomplete": "⚠️ INCOMPLETE", + }.get(publication.status, publication.status.upper()) + lines.append(f"**Publication status:** {publication_label}") policy = next( - ( - result.metadata.get("policy") - for result in results - if isinstance(result.metadata.get("policy"), dict) - ), + (result.metadata.get("policy") for result in results if isinstance(result.metadata.get("policy"), dict)), None, ) if policy is not None: @@ -127,11 +218,14 @@ def render_all(self, results: list[ValidationResult]) -> str: lines.append("| Metric | Value |") lines.append("|--------|-------|") lines.append(f"| Validator Results | {len(results)} |") - lines.append(f"| ✅ Passed | {sum(1 for r in results if r.status == 'passed')} |") + lines.append(f"| Validators Run | {executed_count} |") + lines.append(f"| ✅ Passed | {passed_count} |") lines.append( f"| ❌ Failed | {sum(1 for r in results if r.status == 'failed' and not is_advisory_agent_eval_skip(r))} |" ) lines.append(f"| ⚠️ Incomplete | {sum(1 for r in results if r.is_incomplete)} |") + if skip_count: + lines.append(f"| ⏭️ Skipped | {skip_count} |") if advisory_skip_count: lines.append(f"| ⏭️ Advisory skips | {advisory_skip_count} |") @@ -178,89 +272,103 @@ def render_all(self, results: list[ValidationResult]) -> str: lines.append("") # Tier 3: Agent Evaluation summary (if present) - tier3_results = [r for r in results if r.metadata.get("agent_eval")] - if tier3_results: - ae = tier3_results[0].metadata["agent_eval"] - verdict = ae.get("verdict", "unknown").upper() + ae = publication.tier3.payload or select_agent_eval_payload(results) + if ae: + verdict = publication.tier3.status.upper() composite = ae.get("composite_lift") runtime = ae.get("runtime_seconds", 0.0) lines.append("## Tier 3: Agent Evaluation") lines.append("") - composite_text = f"{composite:+.2f}" if isinstance(composite, int | float) else "N/A" + composite_text = _format_report_number(composite, signed=True) lines.append(f"**Verdict:** {verdict} (composite lift = {composite_text})") - lines.append(f"**Runtime:** {runtime:.1f}s") + runtime_number = _finite_report_number(runtime) + runtime_text = f"{runtime_number:.1f}s" if runtime_number is not None else "N/A" + lines.append(f"**Runtime:** {runtime_text}") harbor_viewer = normalize_harbor_viewer_for_display(ae) - if harbor_viewer.get("job_url"): - lines.append(f"**Harbor logs:** [Open Harbor logs]({harbor_viewer['job_url']})") - if harbor_viewer.get("analysis_url"): - lines.append(f"**Harbor analysis:** [Open Harbor analysis]({harbor_viewer['analysis_url']})") + if job_url := _markdown_safe_url(harbor_viewer.get("job_url")): + lines.append(f"**Harbor logs:** [Open Harbor logs]({job_url})") + if analysis_url := _markdown_safe_url(harbor_viewer.get("analysis_url")): + lines.append(f"**Harbor analysis:** [Open Harbor analysis]({analysis_url})") lines.append("") evaluators = ae.get("evaluators", {}) - if evaluators: + if isinstance(evaluators, dict) and evaluators: lines.append("### Evaluator Scores") lines.append("") lines.append("| Evaluator | With Skill | Baseline | Lift |") lines.append("|-----------|-----------|----------|------|") for name, scores in evaluators.items(): + if not isinstance(name, str) or not isinstance(scores, dict): + continue ws = scores.get("with_skill", 0.0) bl = scores.get("baseline", 0.0) lift = scores.get("lift", 0.0) - lines.append(f"| {name.replace('_', ' ').title()} | {ws:.2f} | {bl:.2f} | {lift:+.2f} |") + lines.append( + f"| {_markdown_untrusted_table_cell(name.replace('_', ' ').title())} " + f"| {_format_report_number(ws)} | " + f"{_format_report_number(bl)} | {_format_report_number(lift, signed=True)} |" + ) lines.append("") - insights = ae.get("insights", {}) - if any(v.get("score") is not None for v in insights.values()): + insights = _mapping(ae.get("insights")) + insight_rows = [ + (dimension, info) + for dimension, info in insights.items() + if isinstance(dimension, str) and isinstance(info, dict) and info.get("score") is not None + ] + if insight_rows: lines.append("### LLM-as-Judge Insights") lines.append("") lines.append("| Dimension | Score | Explanation |") lines.append("|-----------|-------|-------------|") - for dim, info in insights.items(): + for dim, info in insight_rows: score = info.get("score") - if score is None: - continue - score_str = str(score).upper() if isinstance(score, str) else f"{score:.2f}" - explanation = info.get("explanation", "")[:60] - lines.append(f"| {dim.title()} | {score_str} | {explanation} |") + score_number = _finite_report_number(score) + score_str = ( + f"{score_number:.2f}" + if score_number is not None + else _markdown_inline_text(score, limit=32).upper() or "N/A" + ) + explanation = _markdown_untrusted_table_cell(info.get("explanation"), limit=60) + lines.append(f"| {_markdown_untrusted_table_cell(dim.title())} | {score_str} | {explanation} |") lines.append("") - suggestions_v2 = ae.get("suggestions_v2") or [] + suggestions_v2 = _dict_items(ae.get("suggestions_v2")) if suggestions_v2: lines.append("### Evidence-Backed Suggestions") lines.append("") for idx, suggestion in enumerate(suggestions_v2, start=1): - recommendation = str(suggestion.get("recommendation") or "").strip() + recommendation = _markdown_inline_text(suggestion.get("recommendation")) if not recommendation: continue - metric = suggestion.get("metric", "unknown") + metric = _markdown_inline_text(suggestion.get("metric")) or "unknown" lines.append(f"{idx}. **{metric}**: {recommendation}") harbor_evidence = suggestion.get("harbor_evidence") or suggestion.get("evidence") if isinstance(harbor_evidence, dict): - url = safe_url(harbor_evidence.get("url")) + url = _markdown_safe_url(harbor_evidence.get("url")) if url: - label = harbor_evidence_link_text(harbor_evidence) + label = _markdown_inline_text(harbor_evidence_link_text(harbor_evidence)) or "Evidence" lines.append(f" - Evidence: [{label}]({url})") - for ref in (suggestion.get("evidence_refs") or [])[:3]: - pointer = ref.get("json_pointer") or ref.get("path") or "" - excerpt = str(ref.get("excerpt") or ref.get("label") or "")[:120] - lines.append(f" - Evidence: `{ref.get('kind', 'evidence')}` `{pointer}` {excerpt}") + for ref in _dict_items(suggestion.get("evidence_refs"))[:3]: + pointer = _markdown_inline_text(ref.get("json_pointer") or ref.get("path")) + excerpt = _markdown_inline_text(ref.get("excerpt") or ref.get("label"), limit=120) + kind = _markdown_inline_text(ref.get("kind")) or "evidence" + lines.append(f" - Evidence: `{kind}` `{pointer}` {excerpt}") lines.append("") - elif ae.get("recommendations"): + elif recommendations := _dict_items(ae.get("recommendations")): lines.append("### Recommendations") lines.append("") - for idx, recommendation in enumerate(ae.get("recommendations") or [], start=1): - if not isinstance(recommendation, dict): - continue - message = str(recommendation.get("message") or recommendation.get("title") or "").strip() + for idx, recommendation in enumerate(recommendations, start=1): + message = _markdown_inline_text(recommendation.get("message") or recommendation.get("title")) if not message: continue lines.append(f"{idx}. {message}") evidence = recommendation.get("evidence") if isinstance(evidence, dict): - url = safe_url(evidence.get("url")) + url = _markdown_safe_url(evidence.get("url")) if url: - label = harbor_evidence_link_text(evidence) + label = _markdown_inline_text(harbor_evidence_link_text(evidence)) or "Evidence" lines.append(f" - Evidence: [{label}]({url})") lines.append("") @@ -281,11 +389,11 @@ def _render_result(self, result: ValidationResult, lines: list[str]) -> None: """Render a single validation result.""" qs = result.metadata.get("quality_scores") - advisory_skip = is_advisory_agent_eval_skip(result) + clean_skip = is_cleanly_skipped(result) if result.is_incomplete: status_emoji = "⚠️ INCOMPLETE" lines.append(f"### {status_emoji} {result.validator_name}") - elif advisory_skip: + elif clean_skip: lines.append(f"### ⏭️ SKIPPED {result.validator_name}") elif qs and qs.get("grade"): grade = qs["grade"] @@ -314,11 +422,8 @@ def _render_result(self, result: ValidationResult, lines: list[str]) -> None: if result.is_incomplete: self._render_incomplete(result, lines) - elif advisory_skip: - payload = result.metadata.get("agent_eval", {}) if result.metadata else {} - provenance = payload.get("provenance", {}) if isinstance(payload, dict) else {} - message = provenance.get("message") if isinstance(provenance, dict) else None - lines.append(f"- {message or 'Live evaluation did not run.'}") + elif clean_skip: + lines.append(f"- Skip reason: {get_skip_reason(result)}") elif result.passed: self._render_success(result, lines) if result.findings: diff --git a/src/skillevaluator/reporting/templates/report.html.j2 b/src/skillevaluator/reporting/templates/report.html.j2 index 6b344c14..cf054bee 100644 --- a/src/skillevaluator/reporting/templates/report.html.j2 +++ b/src/skillevaluator/reporting/templates/report.html.j2 @@ -1233,6 +1233,7 @@ {% set tier3_execution_status = (tier3.get('execution_status') or tier3_summary_data.get('execution_status')) if has_tier3 else none %} {% set tier3_is_skipped = tier3_execution_status == 'skipped' or (tier3_provenance_data.get('advisory') and tier3_provenance_data.get('reason') == 'skipped') %} {% set tier3_skip_message = tier3_provenance_data.get('message') or 'Live evaluation did not run.' %} + {% set tier3_verdict = (tier3_effective_status or tier3.get('verdict') or 'unknown') | lower if has_tier3 else 'unknown' %}
@@ -1253,7 +1254,12 @@ {% if target_path %}

Target: {% if target_path.startswith('https://') %}{{ target_display or target_path }}{% else %}{{ target_path }}{% endif %}

{% endif %} {% if profile %}

Profile: {{ profile }}

{% endif %} {% if timestamp %}

Generated: {{ timestamp }}{% if version %} · SkillEvaluator v{{ version }}{% endif %}

{% endif %} - {{ 'Incomplete' if has_incomplete else ('All Passed' if all_passed else 'Issues Found') }} + Process: {{ 'Incomplete' if has_incomplete else ('All Passed' if all_passed else 'Issues Found') }} + + Publication: {{ publication_status | title }} +
@@ -1324,9 +1330,10 @@ {% endif %} {% else %}
-
{{ pass_percentage }}%
+
{{ pass_percentage }}%
-
Validators: {{ passed_count }}/{{ total_validators }} passed
+
Validators: {{ passed_count }}/{{ executed_count }} passed
+ {% if skipped_count %}
Skipped: {{ skipped_count }}
{% endif %}
{{ content_label_plural }}: {{ passed_skills }}/{{ total_skills }} passed
{% if total_issues %}
Issues: {{ total_issues }}
{% endif %}
@@ -1351,7 +1358,7 @@ {% set initial_tab = tabs[0].id if tabs else 'tier1' %}
{% if has_tier1 %} - {% set t1_class = 'warn' if tier1_summary.incomplete_count else ('pass' if (tier1_summary.total > 0 and tier1_summary.all_passed) else ('fail' if tier1_summary.total > 0 else 'skipped')) %} + {% set t1_class = 'warn' if tier1_summary.incomplete_count else ('skipped' if tier1_summary.executed_count == 0 else ('pass' if tier1_summary.all_passed else 'fail')) %}
Tier 1 - {{ 'PASS' if t1_class == 'pass' else ('FAIL' if t1_class == 'fail' else 'N/A') }} + {{ 'PASS' if t1_class == 'pass' else ('FAIL' if t1_class == 'fail' else ('INCOMPLETE' if t1_class == 'warn' else 'SKIPPED')) }}

Security & Static Validation

- {{ tier1_summary.passed_count }}/{{ tier1_summary.total }} validators passed + {{ tier1_summary.passed_count }}/{{ tier1_summary.executed_count }} validators passed + {% if tier1_summary.skipped_count %}{{ tier1_summary.skipped_count }} skipped{% endif %} {% if tier1_summary.issue_count %}{{ tier1_summary.issue_count }} issue(s){% endif %}
{% if tier1_summary.critical or tier1_summary.high or tier1_summary.medium or tier1_summary.low %} @@ -1380,7 +1388,7 @@
{% endif %} {% if has_tier2 %} - {% set t2_class = 'pass' if (tier2_summary.total > 0 and tier2_summary.all_passed) else ('fail' if tier2_summary.total > 0 else 'skipped') %} + {% set t2_class = 'warn' if tier2_summary.incomplete_count else ('skipped' if tier2_summary.executed_count == 0 else ('pass' if tier2_summary.all_passed else 'fail')) %}
Tier 2 - {{ 'PASS' if t2_class == 'pass' else ('FAIL' if t2_class == 'fail' else 'N/A') }} + {{ 'PASS' if t2_class == 'pass' else ('FAIL' if t2_class == 'fail' else ('INCOMPLETE' if t2_class == 'warn' else 'SKIPPED')) }}

Deduplication

- {{ tier2_summary.passed_count }}/{{ tier2_summary.total }} checks passed + {{ tier2_summary.passed_count }}/{{ tier2_summary.executed_count }} checks passed + {% if tier2_summary.skipped_count %}{{ tier2_summary.skipped_count }} skipped{% endif %} {% if tier2_summary.issue_count %}{{ tier2_summary.issue_count }} duplicate(s){% endif %}
{% endif %} {% if has_tier3 %} - {% set tier3_verdict = (tier3.verdict or 'unknown') | lower %} {% set tier3_display_verdict = 'SKIPPED' if tier3_is_skipped else tier3_verdict | upper %} {% set t3_class = 'skipped' if tier3_is_skipped else ('pass' if tier3_verdict == 'pass' else ('fail' if tier3_verdict == 'fail' else 'warn')) %}
{{ passed_count }}

{{ pass_percentage }}% success rate

+ {% if skipped_count %} +
+

Skipped

+

{{ skipped_count }}

+

Not executed

+
+ {% endif %}

Failed

{{ failed_count }}

@@ -1551,8 +1567,8 @@
{% for skill in issue.skills_affected[:5] %} - {{ skill }} - Q + {{ skill }} + Q {% endfor %} {% if issue.skills_affected | length > 5 %} @@ -1610,7 +1626,7 @@
{% for item in contrib['items'][:8] %} - {{ item.name }} + {{ item.name }} {% endfor %} {% if contrib['items'] | length > 8 %} +{{ contrib['items'] | length - 8 }} more @@ -1815,9 +1831,10 @@ ValidatorStatusFiles ScannedChecks PerformedErrorsWarnings {% for result in results %} - + {% set result_is_skipped = result | cleanly_skipped %} + {{ result.validator_name }} - {{ 'Incomplete' if result.is_incomplete else ('Pass' if result.passed else 'Fail') }} + {{ 'Skipped' if result_is_skipped else ('Incomplete' if result.is_incomplete else ('Pass' if result.passed else 'Fail')) }} {{ result.summary.files_scanned }} {{ result.summary.checks_performed }} {{ result.summary.errors }} @@ -1830,19 +1847,20 @@
Detailed Results
{% for result in results %} -
- -
+
{% if result.summary.files_scanned > 0 or result.summary.checks_performed > 0 %}
{% if result.summary.files_scanned > 0 %}{{ icons.file | safe }} {{ result.summary.files_scanned }} files scanned{% endif %} @@ -1856,7 +1874,11 @@ Incomplete scanner evidence: {{ result.incomplete_scans | join(', ') }} did not produce a trustworthy report. Restore the scanner and rerun validation before publication.
{% endif %} - {% if result.passed %} + {% if result_is_skipped %} +
+ Skip reason: {{ result | skip_reason }} +
+ {% elif result.passed %}
    {% if result.success_details %}{% for detail in result.success_details %}
  • {{ icons.checkmark | safe }}{{ detail.check_name }}: {{ detail.message }}{% if detail.metadata %} ({% for k, v in detail.metadata.items() %}{{ k }}={{ v }}{% if not loop.last %}, {% endif %}{% endfor %}){% endif %}
  • {% endfor %} {% elif result.messages %}{% for msg in result.messages %}
  • {{ icons.checkmark | safe }}{{ msg }}
  • {% endfor %} @@ -1912,7 +1934,7 @@

    Checks Run

    -

    {{ tier2_results | length }}

    +

    {{ tier2_summary.executed_count }}

    Semantic overlap analysis

    @@ -1927,7 +1949,16 @@ Checks for redundant content within one skill and semantic overlap across skills or local catalog entries.

    {% for result in tier2_results %} - {% if result.passed and not result.findings %} + {% if result | cleanly_skipped %} +
    + Skip reason: {{ result | skip_reason }} +
    + {% elif result.is_incomplete %} +
    + Incomplete scanner evidence: + {{ result.incomplete_scans | join(', ') }} did not produce trustworthy evidence. Restore the scanner and rerun validation. +
    + {% elif result.passed and not result.findings %}
    {{ icons.checkmark | safe }} No semantic overlaps detected. {% if result.success_details %} @@ -1994,8 +2025,10 @@ {% set t3_harbor_analysis_url = t3_harbor.get('analysis_url') %} {% set t3_evidence_links = t3_harbor.get('evidence_links') or [] %} {% set t3_agent_anchor_ids = {} %} + {% set t3_agent_panel_ids = {} %} {% for name, _agent in t3_agents.items() %} - {% set _ = t3_agent_anchor_ids.update({name: 'tier3-agent-' ~ loop.index0}) %} + {% set _ = t3_agent_anchor_ids.update({name: 'tier3-agent-anchor-' ~ loop.index0}) %} + {% set _ = t3_agent_panel_ids.update({name: 'tier3-agent-panel-' ~ loop.index0}) %} {% endfor %} ", + "", + "", + ], +) +def test_gate_ignores_hidden_html_recommendation_for_nonpass_verdict( + tmp_path: Path, + hidden_html: str, +) -> None: + benchmark = tmp_path / "BENCHMARK.md" + benchmark.write_text( + _valid_benchmark() + .replace("> **Overall verdict: PASS**", "> **Overall verdict: INCOMPLETE**") + .replace("## Freshness", f"{hidden_html}\n\n## Freshness"), + encoding="utf-8", + ) + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + assert offenders == [] + + +@pytest.mark.parametrize( + "visible_html", + [ + "Recommended for publication.", + " Recommended for publication.", + "Recommended for publication.", + ], +) +def test_gate_rejects_visible_recommendation_after_hidden_html_control( + tmp_path: Path, + visible_html: str, +) -> None: + benchmark = tmp_path / "BENCHMARK.md" + benchmark.write_text( + _valid_benchmark() + .replace("> **Overall verdict: PASS**", "> **Overall verdict: INCOMPLETE**") + .replace("## Freshness", f"{visible_html}\n\n## Freshness"), + encoding="utf-8", + ) + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + assert {offender.reason for offender in offenders} == {"non-PASS verdict recommends publication"} + + +@pytest.mark.parametrize( + "visible_html", + [ + "
    Overall verdict: PASS
    ", + "Overall verdict: PASS", + "Overall verdict: PASS", + ], +) +def test_gate_rejects_visible_raw_html_verdict_callout( + tmp_path: Path, + visible_html: str, +) -> None: + benchmark = tmp_path / "BENCHMARK.md" + benchmark.write_text( + _valid_benchmark() + .replace("> **Overall verdict: PASS**", "> **Overall verdict: INCOMPLETE**") + .replace("## Freshness", f"{visible_html}\n\n## Freshness"), + encoding="utf-8", + ) + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + assert "duplicate Overall verdict field" in {offender.reason for offender in offenders} + + +@pytest.mark.parametrize( + "hidden_verdict", + [ + "", + '', + '
    Overall verdict: PASS — Recommended for publication
    ', + '
    Overall verdict: PASS — Recommended for publication
    ', + ], +) +def test_gate_rejects_raw_html_as_the_only_verdict_field( + tmp_path: Path, + hidden_verdict: str, +) -> None: + benchmark = tmp_path / "BENCHMARK.md" + benchmark.write_text( + _valid_benchmark().replace("> **Overall verdict: PASS**", hidden_verdict), + encoding="utf-8", + ) + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + assert "invalid Overall verdict field" in {offender.reason for offender in offenders} + + +def test_gate_rejects_linked_overall_verdict_field(tmp_path: Path) -> None: + benchmark = tmp_path / "BENCHMARK.md" + benchmark.write_text( + _valid_benchmark().replace( + "> **Overall verdict: PASS**", + "> **[Overall verdict: PASS](https://example.invalid/phish)**", + ), + encoding="utf-8", + ) + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + assert "invalid Overall verdict field" in {offender.reason for offender in offenders} + + +@pytest.mark.parametrize( + ("tier", "status"), + [ + (1, "PASSED"), + (2, "PASSED"), + (3, "PASS"), + ], +) +@pytest.mark.parametrize( + "wrapped_status", + [ + "", + '', + '{status}', + "{status}", + ], +) +def test_gate_rejects_html_wrapped_tier_completion_status( + tmp_path: Path, + tier: int, + status: str, + wrapped_status: str, +) -> None: + benchmark = tmp_path / "BENCHMARK.md" + original_row = next(line for line in _valid_benchmark().splitlines() if line.startswith(f"| Tier {tier} |")) + wrapped_row = original_row.replace(f"**{status}**", wrapped_status.format(status=status)) + benchmark.write_text( + _valid_benchmark().replace(original_row, wrapped_row), + encoding="utf-8", + ) + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + assert f"invalid Tier {tier} status" in {offender.reason for offender in offenders} + + +def test_gate_rejects_linked_tier_completion_status(tmp_path: Path) -> None: + benchmark = tmp_path / "BENCHMARK.md" + benchmark.write_text( + _valid_benchmark().replace( + "**PASSED** | Complete |", + "**[PASSED](https://example.invalid/phish)** | Complete |", + 1, + ), + encoding="utf-8", + ) + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + assert "invalid Tier 1 status" in {offender.reason for offender in offenders} + + +@pytest.mark.parametrize("section", ["Tier Status", "Evaluation Metadata"]) +def test_gate_rejects_critical_markdown_inside_raw_html_container( + tmp_path: Path, + section: str, +) -> None: + benchmark = tmp_path / "BENCHMARK.md" + text = _valid_benchmark() + start = text.index(f"## {section}") + len(f"## {section}") + next_heading = text.index("\n## ", start) + text = f"{text[:start]}\n\n\n{text[next_heading:]}" + benchmark.write_text(text, encoding="utf-8") + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + reasons = {offender.reason for offender in offenders} + if section == "Tier Status": + assert "missing Tier 1 status row" in reasons + assert "missing Tier 2 status row" in reasons + assert "missing Tier 3 status row" in reasons + else: + assert "missing metadata field: - Evaluation date:" in reasons + + +@pytest.mark.parametrize( + "boundary_heading", + [ + "## [Unrelated](https://example.invalid/phish)", + "## Unrelated", + "

    Unrelated

    ", + ], + ids=["linked-heading", "inline-html-heading", "raw-html-heading"], +) +@pytest.mark.parametrize("section", ["Tier Status", "Evaluation Metadata"]) +def test_gate_does_not_attribute_evidence_across_untrusted_section_boundary( + tmp_path: Path, + section: str, + boundary_heading: str, +) -> None: + benchmark = tmp_path / "BENCHMARK.md" + text = _valid_benchmark() + start = text.index(f"## {section}") + len(f"## {section}") + next_heading = text.index("\n## ", start) + original_body = text[start:next_heading] + text = f"{text[:start]}\n\n{boundary_heading}{original_body}{text[next_heading:]}" + benchmark.write_text(text, encoding="utf-8") + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + reasons = {offender.reason for offender in offenders} + if section == "Tier Status": + assert "missing Tier 1 status row" in reasons + assert "missing Tier 2 status row" in reasons + assert "missing Tier 3 status row" in reasons + else: + assert "missing metadata field: - Evaluation date:" in reasons + + +def test_gate_rejects_entire_card_inside_closed_raw_html_container(tmp_path: Path) -> None: + benchmark = tmp_path / "BENCHMARK.md" + benchmark.write_text(f"
    \n\n{_valid_benchmark()}\n\n
    \n", encoding="utf-8") + + _files, offenders = benchmark_gate.find_offenders([benchmark]) + + reasons = {offender.reason for offender in offenders} + assert "missing required section: # Skill Benchmark:" in reasons + assert "invalid Overall verdict field" in reasons + + +@pytest.mark.parametrize( + "opening_html", + [ + "", + "