From 5182015b889371aaa25c1cd7476fc75dc9cef7b9 Mon Sep 17 00:00:00 2001 From: Brad Edwards Date: Thu, 30 Jul 2026 05:24:17 +0200 Subject: [PATCH 1/3] docs(cyborg): complete CAGE-2 source ledger --- .gitignore | 3 + .../cyborg-cage2-source-ledger-guardrails.md | 230 ++++++++ docs/index.md | 1 + mkdocs.yml | 1 + src/raes_adapters/cyborg/mapping/README.md | 36 +- .../cyborg/mapping/cage2-loss-disclosures.md | 71 ++- .../cyborg/mapping/cage2-source-ledger.jsonl | 40 ++ src/raes_adapters/cyborg/qualification.json | 52 ++ src/raes_adapters/cyborg/source_ledger.py | 509 ++++++++++++++++++ tests/test_cyborg_source_ledger.py | 297 ++++++++++ tools/verify_cyborg_qualification.py | 7 + 11 files changed, 1230 insertions(+), 17 deletions(-) create mode 100644 docs/decisions/cyborg-cage2-source-ledger-guardrails.md create mode 100644 src/raes_adapters/cyborg/source_ledger.py create mode 100644 tests/test_cyborg_source_ledger.py diff --git a/.gitignore b/.gitignore index 5e7add2..d063a53 100644 --- a/.gitignore +++ b/.gitignore @@ -30,5 +30,8 @@ htmlcov/ # Local MCP server registration (machine-specific paths) .mcp.json +# Local Codex workspace configuration +.codex/ + # Ground Control local operational cache (per-run sonar-watch state) .gc/sonar/ diff --git a/docs/decisions/cyborg-cage2-source-ledger-guardrails.md b/docs/decisions/cyborg-cage2-source-ledger-guardrails.md new file mode 100644 index 0000000..30c8fa0 --- /dev/null +++ b/docs/decisions/cyborg-cage2-source-ledger-guardrails.md @@ -0,0 +1,230 @@ +# CybORG/CAGE-2 source-ledger guardrails + +Issue #13 is the authority for this deliverable. This note fixes the repository, +contract, validation, and claim boundaries the implementation must respect. It +does not author the ledger, classify a source fact, define a row schema, or +describe an implementation plan. + +## Keep source, mapping, and claim authority separate + +The deliverable is backend-local evidence under `raes_adapters.cyborg`: + +- `qualification.json` remains the only owner of the selected profile id, + repository, commit and tree identities, selected-file digests, legal + disposition, stochastic-source inventory, known defects, runtime evidence, + and admissibility decision. +- `cage2-source-ledger.jsonl` accounts for facts consumed from that selection + and binds mapped facts only to published RAES SDL, contract, or evidence + surfaces. +- `cage2-loss-disclosures.md` explains each lossy fact and identifies the + ADR-069 equivalence tier or tiers whose claims it weakens. + +The ledger repeats immutable source coordinates because each row must be +reviewable in isolation, but deterministic joins must prove that those +coordinates equal the qualification-owned selection. It is not a second source +profile, legal record, RAES contract, backend manifest, or experiment artifact. + +A target reference means “this published surface can carry the mapped fact.” +It does not prove that a corresponding SDL scenario, experiment input, backend, +run result, evidence record, or equivalence claim exists. None is authored by +issue #13. + +## Close the source and target joins + +| Concern | Canonical owner and required join | +| --- | --- | +| Profile identity | `qualification.json.profile_id` and `source` own the selection. The ledger is validated against one explicit profile selection; no row may choose another repository, version label, tag, branch, or commit. | +| Source path and bytes | `qualification.json.selected_files` owns path-to-SHA-256 identity. Every ledger path and digest must join exactly. A newly consumed action, observation, reward, scheduler, or other source file must first be added to that qualified set and verified by the existing qualification driver. | +| Source selector | A selector must identify a YAML path, Python symbol, bounded line range, or document section unambiguously. Selector reachability is checked against the detached qualified checkout without importing or executing upstream code. Free-form descriptions are rationale, not selectors. | +| Portable target | Resolve SDL and contract references from `raes_contracts.contracts.schema_bundle()`, including `sdl-authoring-input-v1`; do not maintain a second list of SDL sections, contract ids, model names, properties, or enums. | +| Loss strength | Use ADR-069 §7's five tiers: authored-source, contract, execution-control, state/observation, and outcome/evaluation. These are issue-local claim labels, not a new RAES vocabulary. A loss may weaken more than one tier, and the ledger and disclosure must agree exactly. | +| Legal and attribution evidence | The pinned license files and `qualification.json.legal` own the source/legal facts. Ledger provenance rows cite those facts structurally. A downstream environment pack projects them into its own `environment-pack-provenance/v3` source row and runs `raes-pack-validate`; this repository neither copies that schema nor depends on the pack distribution. | + +`qualification.json` is backend evidence, not a published RAES semantic +surface. A legal or packaging fact can therefore be explicitly out of semantic +mapping scope while retaining a machine-resolvable evidence reference to the +qualification record. It must not be mislabeled as a RAES target or hidden in +`mapping_rule` prose. + +The currently qualified file set contains Scenario2, scenario images, +evaluation, the legacy wrapper, and selected simple agents. It does not contain +the complete action, observation, reward, admissibility, or turn-scheduling +implementation surface required by issue #13. The implementation must extend +the qualification-owned `selected_files` closure for every additionally +consumed file; placing unmatched digests only in the ledger would create a +competing pin authority. + +## Preserve two-dimensional coverage + +The accepted design has two independent completeness axes: + +- source families: Scenario2 and images, actions, observations, rewards, + wrappers, agents, evaluation, and provenance/licensing; and +- fact facets: topology, hosts, services, accounts, privileges, roles, + participants, initial knowledge, visibility, hidden truth, actions, + admissibility, turn order, trial lengths, termination, red variants, seeds, + stochastic controls, reward components, objectives, and derived measures. + +Validation must fail when either required axis has a gap. A broad row such as +“Scenario2 mapped” must not satisfy topology, hosts, services, accounts, +privileges, roles, visibility, and hidden truth simultaneously without +separately reviewable fact selectors. Keep each row atomic enough that one +source fact has one disposition and one reviewable transformation. Split a fact +that requires materially different targets or dispositions. + +The only dispositions are the issue's `mapped`, `out-of-scope`, and +`loss-disclosed`: + +- `mapped` requires a resolvable published RAES target and no loss reference; +- `out-of-scope` requires a bounded rationale and no RAES target or loss + reference; and +- `loss-disclosed` requires a referenced disclosure and a nonempty set of + recognized weakened tiers. A partial mapping must identify the valid target + separately from the lost portion rather than making one ambiguous row. + +Strict row validation must reject malformed JSONL, duplicate JSON keys, +non-object rows, non-finite values, unknown fields, blank required values, +duplicate row ids, unknown source families or fact facets, unknown +dispositions, unresolved selectors, source/profile/digest drift, unresolved +RAES targets, orphan disclosures, undisclosed losses, and disagreement between +row and disclosure tiers. + +Automation can prove the checked-in inventory is closed, classified, joined, +and drift-free. It cannot discover every semantic fact hidden in arbitrary +upstream Python or prose. Initial semantic completeness remains a review +obligation against the accepted two-axis checklist; after acceptance, immutable +file digests ensure changed source bytes force that review to run again. Do not +claim that category presence alone proves source exhaustiveness. + +## Reuse published semantic owners + +| Source concern | Published owner | +| --- | --- | +| Topology, hosts, services, accounts, roles, privileges, reachability, initial scenario knowledge, and objective truth | Existing `sdl-authoring-input-v1` sections and closed RAES SDL models, including infrastructure, nodes, relationships, entities, accounts, agents, propositions, assertions, objectives, action contracts, observation boundaries, behavior specifications, and evidence requirements as applicable | +| Participant-visible actions and admissibility | SDL action contracts plus published participant action/admission, execution, lifecycle, and joint-action contracts; native classes, ids, masks, and Gym spaces remain source-private | +| Visibility, observations, and hidden truth | SDL observation boundaries and published participant observation/shared-state/evidence contracts; hidden truth is classified and bounded, never copied into a portable payload | +| Turn order, logical steps, trial length, termination, participant/red-variant selection, seeds, and stochastic declarations | `experiment-authoring-input-v1` / `ExperimentSpecModel`, its run plan, episode control, allocation, conditions, and stochastic-control models | +| Reward components, evaluator policy, cumulative score, objectives, outcomes, and derived measures | `ExperimentTaskModel.evaluation_protocol`, evaluation-result, evidence-record, derived-measure, run, and study contracts; reward is not SDL objective truth | +| Source identity and checksums when a later portable artifact cites them | Published experiment artifact-reference and checksum contracts; source licensing remains pack-domain publication evidence | + +Do not use `SDLLineageLedgerModel`: it records the ancestry of the RAES +language and its normative artifacts, not the provenance of one +simulator-derived mapping. Do not add a CAGE-specific SDL section, metadata +convention, contract, vocabulary, manifest block, evidence type, or profile. + +## Compose existing repository validation + +The implementation must build on these incumbents: + +- `load_qualification()` and package-resource loading in + `raes_adapters.cyborg`; +- `tools/verify_cyborg_qualification.py` for the detached immutable checkout, + selected-file digest verification, path confinement, sanitized subprocess + environment, bounded errors, and source reachability; +- `tests/test_cyborg_qualification.py` and + `tests/test_cyborg_qualification_driver.py` for exact source/legal/runtime + joins and security regressions; +- `raes==2.0.0`, `ContractModel`'s closed-model behavior, and + `raes_contracts.contracts.schema_bundle()` for published target authority; +- the strict JSONL, coverage, loss-reference, and negative-test pattern in + `raes_adapters.cyberbattlesim.scenario_ledger`, without importing one backend + from the other or copying its incompatible category, disposition, or + equivalence vocabulary; +- `importlib.resources` and the existing Hatch wheel/sdist layout so the ledger + and disclosures remain installed package resources; and +- the existing `tests`, `distributions`, `docs`, and canonical `nox -s verify` + graph. No second workflow, conformance runner, or policy gate is needed. + +The CybORG row and loss rules remain module-local. Do not move them into +`raes_adapters.base`, create a shared ledger DTO/schema/registry, or import the +CyberBattleSim validator as a cross-backend dependency. The reusable authority +is RAES's published schema bundle and the qualification selection; the +backend-specific classifications intentionally remain separate. + +Parsing failures may use ordinary bounded `ValueError`/`RuntimeError` behavior, +and relational validation may return small row-id/field/reason problems as the +existing precedent does. Do not create an exception hierarchy, diagnostic +envelope, logger family, or error DTO. If a later portable runtime exposes a +failure, it must use RAES `Diagnostic`/`ApplyResult` boundaries. + +## Cross-cutting security and operational boundary + +| Layer | Required treatment | +| --- | --- | +| Authentication and authorization | None is introduced. These are static package resources and local validation, with no controller, caller, HTTP endpoint, or mutation API. A future service reuses RAES runtime strict-default authentication and authorization. | +| Secret handling | Consume public qualified sources only. Record account/credential semantics without real, learned, or cached credential values. Do not read a user secret store or include tokens, private material, hidden-state values, or environment contents in rows or disclosures. | +| Input shape and parsing | Decode UTF-8 JSONL strictly with duplicate-key, finite-number, closed-field, type, blank-value, and size/count checks. Resolve targets from the pinned RAES schema bundle. Parse upstream YAML/Python only for bounded selector reachability; never execute it. | +| Configuration and environment | The explicit qualification profile is the only selection input. No environment variable, CLI flag, working-directory discovery, floating ref, or user config may override source identity, target semantics, or loss tiers. | +| OS and process exposure | Offline ledger validation reads installed resources in process. Source-tree verification reuses the existing detached checkout, sanitized allowlisted environment, argument vectors, isolated temporary directory, timeouts, safe Python path, and path confinement. Do not add shell interpolation or put credentials/source payloads in argv. | +| Errors and observability | Report only bounded stage names or row id, field, and reason. Do not echo source payloads, rejected values, raw subprocess stdout/stderr, native object representations, environment/argument dumps, or full tracebacks. No new operational logger or portable log format is needed. | +| Persistence | Checked-in package resources and the existing qualification record are the only durable state. Temporary source checkouts are ephemeral. Do not add a database, cache authority, repository service, audit log, `ControlPlaneStore`, or evidence store. | +| Distribution and workflow | Keep resources under `src/raes_adapters/cyborg`; prove their installed presence through the existing build/clean-install boundary and run validation through the existing nox sessions. A new CI job would require a `PR Gate` change and is unjustified here. | + +## Extension seam + +The external parameter is one immutable module-local evidence selection: +qualification profile id plus ledger and loss-resource identities. The current +profile may be the default, but validation must receive that selection +explicitly. A future CAGE scenario, CybORG revision, or reviewed source closure +becomes another selection with its own qualified pins and resources, not an +environment override, central simulator registry entry, or branch in target +resolution. + +Source-family and fact-facet coverage are properties of the accepted CAGE-2 +design, while RAES target validity follows the pinned `schema_bundle()`. A new +source revision should therefore change selection data and ledger rows; a new +RAES release should change the pinned dependency and be reviewed against its +published bundle. Neither variation should require editing a hand-maintained +contract catalog. + +## Gotchas and anti-patterns + +- Do not conflate the smoke selection with the evaluation selection. The + qualification smoke uses seed `3`, two wrapper steps, an external blue Sleep + action, B-line red, and GreenAgent; the evaluation source declares 100 + episodes, trial lengths 30/50/100, `BlueLoadAgent`, and a three-policy red + set. +- Do not collapse wrapper `max_steps`, logical environment turns, participant + actions, evaluator trial cutoffs, source termination, Gym `done`, + termination, truncation, and cleanup into one fact. +- Do not treat `CybORG.set_seed(3)` as complete stochastic control. The + qualification already distinguishes simulator/Python random, Gym action + space, NumPy global state, and the external blue policy, and records the + evaluator seed as unbound. +- Do not merge backend, evaluator, blue/red/green participant, policy, and + control-plane caller identities. +- Do not merge authored Scenario2 truth, admitted/runtime state, + participant-visible observation, hidden truth, evidence, and derived + measures. +- Do not treat native action classes or ids as portable action identities, or + action availability as equivalent to admissibility. +- Do not treat reward components, reward vectors, cumulative score, objective + satisfaction, participant outcome, backend conformance, and equivalence as + synonyms. +- Do not silently repair the recorded Remove success misreport, Scenario2 port + mismatch, unbound evaluation seed, legacy Gym behavior, or packaging defect. + Any affected mapped fact must retain a source row and an appropriately tiered + loss. +- Do not let a mapped row, a complete ledger, a green validator, or a source + digest upgrade the qualification's `not-admissible` decision or imply an + installable extra, backend support, deterministic replay, or outcome + equivalence. +- Do not put semantic values into free-form metadata, `mapping_rule`, + `verification`, comments, logs, or Markdown because the intended RAES target + is inconvenient. + +## Non-goals and implementation boundaries + +- No CAGE-2 SDL scenario, experiment task/spec/study, backend manifest, + participant manifest, conformance profile, fixture, run evidence, derived + measure, environment pack, or equivalence result is authored. +- No CybORG adapter, evaluator, participant runtime, action/observation/reward + projector, simulator dependency, published patched artifact, or populated + `cyborg` extra is implemented. +- No RAES schema, SDL extension, contract, vocabulary, profile, diagnostic, + exception hierarchy, persistence service, or policy gate is introduced. +- No native simulator execution, hidden-state capture, raw log retention, + environment scan, privileged host operation, or runtime network service is + added for ledger validation. +- No existing qualification blocker, known defect, legal conclusion, source + revision, or equivalence claim is changed merely to make the mapping easier. diff --git a/docs/index.md b/docs/index.md index 85bc582..1633c33 100644 --- a/docs/index.md +++ b/docs/index.md @@ -17,6 +17,7 @@ semantic and protocol authority. - [Contribution guide](https://github.com/RAESystem/adapters/blob/dev/CONTRIBUTING.md) - [Architecture decisions](decisions/adrs/README.md) - [CybORG/CAGE-2 runtime qualification guardrails](decisions/cyborg-cage2-runtime-qualification-guardrails.md) +- [CybORG/CAGE-2 source-ledger guardrails](decisions/cyborg-cage2-source-ledger-guardrails.md) - [CyberBattleSim qualification guardrails](decisions/cyberbattlesim-qualification-guardrails.md) - [CyberBattleSim scenario and source-ledger guardrails](decisions/cyberbattlesim-scenario-ledger-guardrails.md) - [Project services](maintainers/project-services.md) diff --git a/mkdocs.yml b/mkdocs.yml index ff8fd62..f8f8334 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -37,6 +37,7 @@ nav: - Overview: decisions/adrs/README.md - Template: decisions/adrs/TEMPLATE.md - CybORG/CAGE-2 runtime qualification guardrails: decisions/cyborg-cage2-runtime-qualification-guardrails.md + - CybORG/CAGE-2 source-ledger guardrails: decisions/cyborg-cage2-source-ledger-guardrails.md - CyberBattleSim qualification guardrails: decisions/cyberbattlesim-qualification-guardrails.md - CyberBattleSim scenario and source-ledger guardrails: decisions/cyberbattlesim-scenario-ledger-guardrails.md - ADR 003 — Single distribution and Trusted Publishing: decisions/adrs/adr-003-single-distribution-and-trusted-publishing.md diff --git a/src/raes_adapters/cyborg/mapping/README.md b/src/raes_adapters/cyborg/mapping/README.md index f0edabc..8f379db 100644 --- a/src/raes_adapters/cyborg/mapping/README.md +++ b/src/raes_adapters/cyborg/mapping/README.md @@ -1,7 +1,9 @@ # CAGE-2 → RAES mapping ledger -**Authored under REP-003 (RAES issue for the CAGE-2 scenario + mapping).** This -directory is a placeholder standup under REP-002. +This backend-local evidence bridges the source closure qualified by issue #12 +to the published RAES surfaces that can carry each CAGE-2 fact. It does not +author an SDL scenario or experiment, implement a backend, change the +qualification's `not-admissible` result, or establish equivalence. Per RAES ADR-069 §2 and `docs/decisions/cage-2-replication-design.md`, the mapping from upstream CAGE-2/CybORG source facts to portable RAES artifacts is a @@ -10,13 +12,27 @@ mapping from upstream CAGE-2/CybORG source facts to portable RAES artifacts is a ## Files - `cage2-source-ledger.jsonl` — one JSON object per source fact. Each row records - `source_id`, `source_repo`, `source_version`, `source_path`, `source_selector`, - optional `source_digest`, `cage_fact_type`, `raes_target`, `mapping_rule`, - `loss_disclosure`, and `verification` (see the design record for field - definitions). Empty until REP-003. + a unique id, source family, fact facet, qualified repository/commit/path/digest, + verifiable selector, disposition, mapping rule, and verification. A mapped or + partially mapped fact cites a published schema-bundle id plus JSON pointer. + Qualification-owned legal evidence uses a `qualification.json` pointer rather + than pretending it is RAES semantics. - `cage2-loss-disclosures.md` — narrative loss disclosures for source facts RAES - cannot carry exactly. A disclosed gap weakens the replication claim; it is - never backfilled with raw CybORG logs or prose-only evidence. + cannot carry exactly. Each disclosure has a machine-parsed set of weakened + ADR-069 equivalence tiers that must exactly match its ledger rows. -Every upstream source fact must be mapped, explicitly declared out of scope, or -loss-disclosed. No claim may rest on CI success or a single cumulative score. +The module-local validator enforces both completeness axes from the accepted +design: source families (Scenario2/images, actions, observations, rewards, +wrappers, agents, evaluation, and provenance/licensing) and semantic facets +(topology through derived measures plus licensing/attribution). It rejects +malformed or open-ended rows, duplicate ids, coverage gaps, unclassified facts, +source-profile/digest drift, unresolved published targets, selector failures, +and missing/orphan/mismatched loss disclosures. The qualification reproducer +also checks every selector against a detached checkout without importing or +executing upstream code. + +Every fact is `mapped`, `out-of-scope`, or `loss-disclosed`. Native state, +observations, hidden truth, action ids, reward vectors, object representations, +raw logs, environment data, and tracebacks remain outside portable artifacts. +No claim may rest on a green validator, CI success, native-log similarity, or a +single cumulative score. diff --git a/src/raes_adapters/cyborg/mapping/cage2-loss-disclosures.md b/src/raes_adapters/cyborg/mapping/cage2-loss-disclosures.md index dde0563..5d00775 100644 --- a/src/raes_adapters/cyborg/mapping/cage2-loss-disclosures.md +++ b/src/raes_adapters/cyborg/mapping/cage2-loss-disclosures.md @@ -1,11 +1,68 @@ # CAGE-2 → RAES loss disclosures -**Authored under REP-003.** Placeholder standup under REP-002. +These disclosures accompany the source selection +`cage2-cyborg-2.1-source-26ce1c1`. Each heading is a stable ledger reference; +the machine-readable tier line binds the precise ADR-069 claim tiers weakened. +A disclosure permits a bounded claim with the stated weakness. It does not +upgrade the qualification's `not-admissible` decision or establish any +equivalence tier. -Each entry below records an upstream CAGE-2/CybORG source fact that the RAES -mapping cannot carry exactly, the reason, and the effect on the replication -claim (RAES ADR-069 §2, §7). A disclosed gap weakens or fails the relevant -equivalence tier explicitly; it is not filled by raw simulator logs, -fixture-local assertions, or prose-only evidence. +## loss-scenario-user3-port-mismatch -_No disclosures yet — the mapping ledger is authored under REP-003._ +**Equivalence tiers weakened:** authored-source, state/observation + +The pinned `linux_user_host_image1.yaml` uses local port 3389 for the User3 +SQL-injection path, while the maintained successor documents 3390. The ledger +preserves the selected bytes and flags the conflict. A later authored scenario +must choose and justify one value; it cannot claim exact source and +state/observation equivalence for both. + +## loss-native-observation-boundary + +**Equivalence tiers weakened:** contract, state/observation + +CybORG exposes mutable native observation and true-state dictionaries, +simulator enums, process/session records, and wrapper arrays. RAES can carry +bounded visibility projections, observation references, redaction policy, and +evidence references, but those native values and representations are +deliberately not portable. Exact native-payload identity is therefore not a +permitted contract or state/observation claim. + +## loss-remove-success-misreport + +**Equivalence tiers weakened:** state/observation, outcome/evaluation + +The selected `Remove` implementation initializes a successful observation even +when no suspicious process is removed. The mapping records the intended +defensive action and retains the defect; it does not silently reinterpret the +native success flag as an observed state transition or successful outcome. + +## loss-wrapper-cutoff-semantics + +**Equivalence tiers weakened:** execution-control + +`ChallengeWrapper.max_steps` forces its legacy `done` flag when the wrapper +counter reaches the bound. That cutoff is distinct from a source terminal +condition, evaluator trial length, Gym truncation, participant action count, +and cleanup. Until an experiment contract binds those facts separately, an +exact execution-control claim is not permitted. + +## loss-evaluation-seed-unbound + +**Equivalence tiers weakened:** execution-control, outcome/evaluation + +The pinned evaluator does not bind simulator, Python, NumPy, Gym action-space, +or blue-policy random streams. The qualification smoke's seed 3 is a different +two-step source-native probe and cannot fill this gap. Evaluation results +therefore cannot support deterministic execution-control or reproducible +outcome/evaluation claims. + +## loss-blue-policy-artifact-unbound + +**Equivalence tiers weakened:** authored-source, execution-control, outcome/evaluation + +The evaluator instantiates `BlueLoadAgent` without an immutable trained model +artifact. Its fallback creates a fresh PPO policy against Scenario1b, not the +selected Scenario2 evaluation condition. The blue implementation, model bytes, +training provenance, and stochastic state are unbound, so policy-dependent +source, execution, and outcome claims remain unsupported. diff --git a/src/raes_adapters/cyborg/mapping/cage2-source-ledger.jsonl b/src/raes_adapters/cyborg/mapping/cage2-source-ledger.jsonl index e69de29..3035b85 100644 --- a/src/raes_adapters/cyborg/mapping/cage2-source-ledger.jsonl +++ b/src/raes_adapters/cyborg/mapping/cage2-source-ledger.jsonl @@ -0,0 +1,40 @@ +{"source_id":"scenario-topology","source_family":"scenario","fact_facet":"topology","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/Scenario2.yaml","source_selector":"lines:388-425","source_digest":"c2f15ec1bd9ab94d2a44acadf6cb9a71acdcb4cbb8fc946129c832139b155b1b","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/infrastructure","mapping_rule":"The Enterprise, Operational, and User subnet declarations and access-control relationships map to authored SDL infrastructure and relationships; native subnet objects remain private.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-hosts","source_family":"scenario","fact_facet":"hosts","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/Scenario2.yaml","source_selector":"lines:286-387","source_digest":"c2f15ec1bd9ab94d2a44acadf6cb9a71acdcb4cbb8fc946129c832139b155b1b","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/nodes","mapping_rule":"The defender, enterprise, operational, server, and user host inventory and value labels map to authored SDL nodes; simulator host objects are not portable identities.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-services","source_family":"scenario","fact_facet":"services","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/images/Internal_image.yaml","source_selector":"lines:1-67","source_digest":"301c42ee3e9c410fc7df778d450d315aff1eedad873e01b8d79dfcac3a299ae8","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/nodes","mapping_rule":"Image-defined operating-system, process, service, port, and vulnerability facts map into the service and vulnerability members of authored SDL nodes.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-accounts","source_family":"scenario","fact_facet":"accounts","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/Scenario2.yaml","source_selector":"lines:78-151","source_digest":"c2f15ec1bd9ab94d2a44acadf6cb9a71acdcb4cbb8fc946129c832139b155b1b","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/accounts","mapping_rule":"Blue Velociraptor session usernames and host bindings map to synthetic authored account and agent intent without carrying learned credentials or native sessions.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-privileges","source_family":"scenario","fact_facet":"privileges","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/Scenario2.yaml","source_selector":"lines:109-151","source_digest":"c2f15ec1bd9ab94d2a44acadf6cb9a71acdcb4cbb8fc946129c832139b155b1b","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/accounts","mapping_rule":"The ubuntu and SYSTEM session principals map to authored account privilege intent; native privilege enums and session handles remain source-private.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-roles","source_family":"scenario","fact_facet":"roles","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/Scenario2.yaml","source_selector":"lines:1-285","source_digest":"c2f15ec1bd9ab94d2a44acadf6cb9a71acdcb4cbb8fc946129c832139b155b1b","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/agents","mapping_rule":"Blue, Red, and Green remain distinct participant roles and do not stand in for backend, evaluator, or caller identity.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-participants","source_family":"scenario","fact_facet":"participants","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/Scenario2.yaml","source_selector":"lines:1-285","source_digest":"c2f15ec1bd9ab94d2a44acadf6cb9a71acdcb4cbb8fc946129c832139b155b1b","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/agents","mapping_rule":"Scenario agent declarations map to authored participant agents while concrete implementation identity remains experiment apparatus context.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-initial-knowledge","source_family":"scenario","fact_facet":"initial-knowledge","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/Scenario2.yaml","source_selector":"lines:7-60","source_digest":"c2f15ec1bd9ab94d2a44acadf6cb9a71acdcb4cbb8fc946129c832139b155b1b","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/observation_boundaries","mapping_rule":"Role-specific INT host fields map to authored initial-knowledge and observation-boundary intent, never to portable native observation values.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-action-sets","source_family":"scenario","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/Scenario2.yaml","source_selector":"lines:62-75","source_digest":"c2f15ec1bd9ab94d2a44acadf6cb9a71acdcb4cbb8fc946129c832139b155b1b","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"The declared Blue action set is input to authored action contracts; native class names, ids, masks, and Gym spaces are not portable action identities.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"scenario-user3-port-mismatch","source_family":"scenario","fact_facet":"services","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Scenarios/images/linux_user_host_image1.yaml","source_selector":"lines:1-79","source_digest":"a474343f4cd6c922eeb9ec9f24c57bd2b93c01ab7a7bc18fa11f2b97bf1bb6f1","disposition":"loss-disclosed","raes_target":"sdl-authoring-input-v1#/properties/nodes","qualification_ref":"qualification.json#/known_defects/3","loss_disclosure":"loss-scenario-user3-port-mismatch","equivalence_tiers":["authored-source","state/observation"],"mapping_rule":"The pinned image's User3 SQL-injection path is retained as authored source; the qualification-owned defect evidence records the maintained successor's conflicting port and the ledger does not silently repair it.","verification":"Qualified digest and defect-reference joins; loss tiers agree with the disclosure."} +{"source_id":"observation-visibility","source_family":"observations","fact_facet":"visibility","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/EnvironmentController.py","source_selector":"python:EnvironmentController.__init__","source_digest":"7fcbd15064268d9bb45cd9e1774d6160f4f82dd563ac6e709556042eb8c98478","disposition":"mapped","raes_target":"participant-observation-envelope-v1#/properties/visibility_projection_ref","mapping_rule":"Role-specific INFO dictionaries and filtered observations map to an explicit participant visibility projection reference; their native dictionaries remain backend-local.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"observation-hidden-truth","source_family":"observations","fact_facet":"hidden-truth","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/EnvironmentController.py","source_selector":"lines:52-68","source_digest":"7fcbd15064268d9bb45cd9e1774d6160f4f82dd563ac6e709556042eb8c98478","disposition":"loss-disclosed","raes_target":"participant-observation-envelope-v1#/properties/hidden_state_refs","loss_disclosure":"loss-native-observation-boundary","equivalence_tiers":["contract","state/observation"],"mapping_rule":"The controller constructs a complete True information request separately from role-filtered observations. Portable evidence may reference that hidden-state source and projection policy, but its native dictionary and values never cross the RAES boundary.","verification":"Qualified digest join; loss tiers agree with the disclosure."} +{"source_id":"observation-native-shape","source_family":"observations","fact_facet":"visibility","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Observation.py","source_selector":"python:Observation","source_digest":"a20f9c3d46f5cfc9e4632ee3baf801041f20ff4077b829e70c1af5a1b68a1186","disposition":"out-of-scope","mapping_rule":"Native Observation dictionaries, enum values, process records, connections, and object representations are simulator-private and are not portable RAES payloads.","verification":"Qualified digest join; absence of a RAES target is required for this native representation."} +{"source_id":"control-admissibility","source_family":"observations","fact_facet":"admissibility","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/EnvironmentController.py","source_selector":"python:EnvironmentController.test_valid_action","source_digest":"7fcbd15064268d9bb45cd9e1774d6160f4f82dd563ac6e709556042eb8c98478","disposition":"mapped","raes_target":"participant-execution-control-v1#","mapping_rule":"Native action-class and parameter availability checks inform participant execution admission; the native action-space mapping itself is not exposed.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"control-turn-order","source_family":"observations","fact_facet":"turn-order","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/EnvironmentController.py","source_selector":"python:EnvironmentController.step","source_digest":"7fcbd15064268d9bb45cd9e1774d6160f4f82dd563ac6e709556042eb8c98478","disposition":"mapped","raes_target":"participant-joint-action-record-v1#/properties/realized_order","mapping_rule":"One logical environment turn iterates participant actions, computes truth, rewards and training updates, then applies the Blue monitor update; this realized order maps to the joint-action record.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"action-monitor","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/Monitor.py","source_selector":"python:Monitor","source_digest":"3a62df8f19cbb7461a17c00553914d8dd14f8f947d24d18c52b0dd93bdaad9a5","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"Monitor's participant intent, target basis, observation boundary, and reported effects map to an authored action contract, not to its native class identity.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"action-analyse","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/Analyse.py","source_selector":"python:Analyse","source_digest":"cc4bd1301e2ed47fc92bdce37bc2bd58e86aa7a9dd5f56eae3431c7e87749b03","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"Analyse's participant intent, host target, admission basis, and observation effect map to an authored action contract.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"action-remove","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/Remove.py","source_selector":"python:Remove","source_digest":"06cebaea236023425990d07f8b3119c8248d04c212525486a0b6fef81937b4af","disposition":"loss-disclosed","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","qualification_ref":"qualification.json#/known_defects/2","loss_disclosure":"loss-remove-success-misreport","equivalence_tiers":["state/observation","outcome/evaluation"],"mapping_rule":"Remove maps to an authored defensive action contract, while the selected source and qualification-owned defect evidence show its success observation can misreport that no suspicious process was removed.","verification":"Qualified digest and defect-reference joins; loss tiers agree with the disclosure."} +{"source_id":"action-restore","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/Restore.py","source_selector":"python:Restore","source_digest":"cf82d7517b762814f104bce8f135ee7109c9702c60748d1c8baccf7526d39978","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"Restore's participant intent, host target, admission basis, and state effect map to an authored action contract.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"action-discover-systems","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/DiscoverRemoteSystems.py","source_selector":"python:DiscoverRemoteSystems","source_digest":"2990da59387fd1d143bdb780b993db5b849b2a2e0461012caae8f465ba250a71","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"Remote-system discovery intent and bounded observation effects map to an authored action contract; discovered native addresses remain private.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"action-discover-services","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/DiscoverNetworkServices.py","source_selector":"python:DiscoverNetworkServices","source_digest":"10f6b5bfb82ca8f77c9f694820931e021244a4149512082f8046508cc4397bb0","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"Service-discovery intent, target grammar, admission basis, and bounded effects map to an authored action contract.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"action-exploit","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/ExploitRemoteService.py","source_selector":"python:ExploitRemoteService","source_digest":"55ca88b04d05cc5d50838af6101e108f97c3c09d38999ef7dbcce866e1d5f2dd","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"Remote exploitation intent, preconditions, target grammar, failure classes, and bounded effects map to an authored action contract.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"action-privilege-escalation","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/PrivilegeEscalate.py","source_selector":"python:PrivilegeEscalate","source_digest":"91953c3b21991397c0b312f6e50f826d49a003ffc01fa12f31436d11d7206318","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"Privilege-escalation intent, session preconditions, host target, failure classes, and bounded effects map to an authored action contract.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"action-impact","source_family":"actions","fact_facet":"actions","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/Actions/AbstractActions/Impact.py","source_selector":"python:Impact","source_digest":"249c28a3aa0730138f327f1f635bcbb5b4adc0979bea82146f2950b051cd489f","disposition":"mapped","raes_target":"sdl-authoring-input-v1#/properties/action_contracts","mapping_rule":"Impact intent, operational-service target, admission basis, and bounded availability effect map to an authored action contract.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"wrapper-termination-cutoff","source_family":"wrappers","fact_facet":"termination","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Agents/Wrappers/ChallengeWrapper.py","source_selector":"python:ChallengeWrapper.step","source_digest":"61ff99c786c63613c8db2a908c43a5e2835bff9c0fd760a6d1064cc5eb37102c","disposition":"loss-disclosed","raes_target":"experiment-authoring-input-v1#/properties/run_plan","loss_disclosure":"loss-wrapper-cutoff-semantics","equivalence_tiers":["execution-control"],"mapping_rule":"The wrapper's max_steps cutoff maps to episode-control intent but is not conflated with source terminal state, evaluator trial cutoff, truncation, or cleanup.","verification":"Qualified digest join; loss tiers agree with the disclosure."} +{"source_id":"wrapper-native-projection","source_family":"wrappers","fact_facet":"visibility","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Agents/Wrappers/ChallengeWrapper.py","source_selector":"python:ChallengeWrapper.__init__","source_digest":"61ff99c786c63613c8db2a908c43a5e2835bff9c0fd760a6d1064cc5eb37102c","disposition":"out-of-scope","mapping_rule":"The legacy table, enum-action, and Gym wrapper chain and its ndarray/four-tuple representations remain native implementation detail, not a RAES backend protocol.","verification":"Qualified digest join; absence of a RAES target is required for this native representation."} +{"source_id":"agents-meander-policy","source_family":"agents","fact_facet":"participants","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Agents/SimpleAgents/Meander.py","source_selector":"python:RedMeanderAgent","source_digest":"518b6e65c9a810dea434c83f1aa30d9662d5afd253d8c27c4d0375de42e0d173","disposition":"mapped","raes_target":"experiment-authoring-input-v1#/properties/apparatus_intent","mapping_rule":"RedMeanderAgent is a distinct red participant-policy implementation in experiment apparatus context, not a backend identity.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"agents-sleep-policy","source_family":"agents","fact_facet":"participants","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Agents/SimpleAgents/SleepAgent.py","source_selector":"python:SleepAgent","source_digest":"59407ee56bf1cf5cfa7a5d6929874e7bb444e06e052daa4456a7bf1cef051248","disposition":"mapped","raes_target":"experiment-authoring-input-v1#/properties/apparatus_intent","mapping_rule":"SleepAgent is a distinct inert participant-policy implementation in experiment apparatus context, not a backend identity.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"agents-green-participant","source_family":"agents","fact_facet":"participants","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Agents/SimpleAgents/GreenAgent.py","source_selector":"python:GreenAgent.get_action","source_digest":"eaaa58a79db113752a273030ff0cb346c2b1dbb59b7fbb35622081b4a59c271c","disposition":"mapped","raes_target":"experiment-authoring-input-v1#/properties/apparatus_intent","mapping_rule":"Green user behavior remains a distinct participant implementation in apparatus context and is not collapsed into backend state transitions.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"agents-stochastic-controls","source_family":"agents","fact_facet":"stochastic-controls","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Agents/SimpleAgents/B_line.py","source_selector":"python:B_lineAgent.get_action","source_digest":"d08d86225c62da0081c3c99bf63cee9bb32670e4d7d56382d3cd1d61e384ae2f","disposition":"mapped","raes_target":"experiment-authoring-input-v1#/properties/run_plan","qualification_ref":"qualification.json#/stochastic_sources","mapping_rule":"The policy's Python-random choice is a distinct stochastic-control source to declare in the run plan; the qualification inventory records the separate stream owners.","verification":"Qualified digest and stochastic-inventory joins; published target resolves through schema_bundle()."} +{"source_id":"evaluation-red-variants","source_family":"evaluation","fact_facet":"red-variants","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Evaluation/evaluation.py","source_selector":"lines:57-62","source_digest":"e4980dea8f1243ffda44bbae92521fc79ce89e74183623cce4d21715b940d51c","disposition":"mapped","raes_target":"experiment-authoring-input-v1#/properties/run_plan","mapping_rule":"The evaluator selects B_lineAgent, RedMeanderAgent, and SleepAgent as three distinct red-policy conditions in its run loop.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"evaluation-trial-lengths","source_family":"evaluation","fact_facet":"trial-lengths","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Evaluation/evaluation.py","source_selector":"lines:18-75","source_digest":"e4980dea8f1243ffda44bbae92521fc79ce89e74183623cce4d21715b940d51c","disposition":"mapped","raes_target":"experiment-authoring-input-v1#/properties/run_plan","mapping_rule":"The 100 episodes and 30, 50, and 100 logical-step trial conditions map to target run count, condition allocation, and episode-control limits.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"evaluation-unbound-seed","source_family":"evaluation","fact_facet":"seeds","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Evaluation/evaluation.py","source_selector":"lines:1-91","source_digest":"e4980dea8f1243ffda44bbae92521fc79ce89e74183623cce4d21715b940d51c","disposition":"loss-disclosed","raes_target":"experiment-authoring-input-v1#/properties/run_plan","qualification_ref":"qualification.json#/known_defects/4","loss_disclosure":"loss-evaluation-seed-unbound","equivalence_tiers":["execution-control","outcome/evaluation"],"mapping_rule":"The complete evaluator declares no binding for simulator, Python, NumPy, Gym, or policy random streams; qualification-owned defect evidence confirms the gap, so seed intent remains disclosed rather than executable.","verification":"Qualified digest and defect-reference joins; loss tiers agree with the disclosure."} +{"source_id":"evaluation-blue-policy-selection","source_family":"evaluation","fact_facet":"participants","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Evaluation/evaluation.py","source_selector":"lines:42-43","source_digest":"e4980dea8f1243ffda44bbae92521fc79ce89e74183623cce4d21715b940d51c","disposition":"loss-disclosed","raes_target":"experiment-authoring-input-v1#/properties/apparatus_intent","loss_disclosure":"loss-blue-policy-artifact-unbound","equivalence_tiers":["authored-source","execution-control","outcome/evaluation"],"mapping_rule":"The evaluator instantiates BlueLoadAgent without a model-file argument, leaving its participant-policy artifact unbound.","verification":"Qualified digest join; loss tiers agree with the disclosure."} +{"source_id":"agents-blue-policy-fallback","source_family":"agents","fact_facet":"participants","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Agents/SimpleAgents/BlueLoadAgent.py","source_selector":"python:BlueLoadAgent.get_action","source_digest":"86ee7e45b053bd66d4cbf42fed6bd915c8bde40fd9da0f8bc0230da8337ba453","disposition":"loss-disclosed","raes_target":"experiment-authoring-input-v1#/properties/apparatus_intent","loss_disclosure":"loss-blue-policy-artifact-unbound","equivalence_tiers":["authored-source","execution-control","outcome/evaluation"],"mapping_rule":"When no model is loaded, BlueLoadAgent.get_action constructs a fresh PPO policy against Scenario1b rather than binding immutable participant-policy bytes for the selected Scenario2 evaluation.","verification":"Qualified digest join; loss tiers agree with the disclosure."} +{"source_id":"evaluation-derived-measures","source_family":"evaluation","fact_facet":"derived-measures","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Evaluation/evaluation.py","source_selector":"lines:69-91","source_digest":"e4980dea8f1243ffda44bbae92521fc79ce89e74183623cce4d21715b940d51c","disposition":"mapped","raes_target":"experiment-derived-measure-v1#","mapping_rule":"Per-episode cumulative reward, condition mean, and sample standard deviation map to derived measures with provenance and limitations; they do not establish equivalence alone.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"reward-components","source_family":"rewards","fact_facet":"reward-components","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/BlueRewardCalculator.py","source_selector":"python:HybridAvailabilityConfidentialityRewardCalculator.calculate_reward","source_digest":"2377df1196db8dbaeff6cb3b03a5e3bb0e690c699c1a8abb1ebefa8ab569b61f","disposition":"mapped","raes_target":"experiment-task-v1#/properties/evaluation_protocol","mapping_rule":"Blue availability and confidentiality components map to named evaluation metrics and evidence requirements; native reward dictionaries and vectors are not portable outcomes.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"reward-objectives","source_family":"rewards","fact_facet":"objectives","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"CybORG/CybORG/Shared/RedRewardCalculator.py","source_selector":"python:HybridImpactPwnRewardCalculator.calculate_reward","source_digest":"965b523a60bf01a9c70a59834f5cc07cee5bdb76881df807d3e6d56c8ef6c062","disposition":"mapped","raes_target":"experiment-task-v1#/properties/evaluation_protocol","mapping_rule":"Red confidentiality and operational-availability rewards map to evaluation metric definitions; authored SDL objectives remain distinct truth propositions.","verification":"Qualified digest join; published target resolves through schema_bundle()."} +{"source_id":"legal-root-license","source_family":"provenance-licensing","fact_facet":"licensing","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"LICENSE","source_selector":"text:MIT License","source_digest":"feaf0fa4b07710228ac471388210bdfb442932af277938d73f29f12104bb29fb","disposition":"out-of-scope","qualification_ref":"qualification.json#/legal","mapping_rule":"Source redistribution and notice duties are qualification-owned provenance evidence, not RAES scenario semantics; environment-pack authors can project this structured reference into their published pack provenance.","verification":"Qualified digest join; qualification JSON pointer resolves."} +{"source_id":"source-attribution","source_family":"provenance-licensing","fact_facet":"attribution","source_repo":"https://github.com/cage-challenge/cage-challenge-2","source_version":"26ce1c1253fa9e2e73f25e6a7f2da32860c11257","source_path":"README.md","source_selector":"lines:462-472","source_digest":"f48bc9ec73a59b401da77806a68c18c06b86428d116bc226ece8d5d8831fb754","disposition":"out-of-scope","qualification_ref":"qualification.json#/legal","mapping_rule":"The README's challenge citation and attribution facts remain qualification-owned provenance evidence reusable by an environment pack, not free-form RAES metadata.","verification":"Qualified digest join; qualification JSON pointer resolves."} diff --git a/src/raes_adapters/cyborg/qualification.json b/src/raes_adapters/cyborg/qualification.json index 9d804b4..78f86a7 100644 --- a/src/raes_adapters/cyborg/qualification.json +++ b/src/raes_adapters/cyborg/qualification.json @@ -113,6 +113,58 @@ { "path": "CybORG/CybORG/Agents/SimpleAgents/BlueLoadAgent.py", "sha256": "86ee7e45b053bd66d4cbf42fed6bd915c8bde40fd9da0f8bc0230da8337ba453" + }, + { + "path": "CybORG/CybORG/Shared/EnvironmentController.py", + "sha256": "7fcbd15064268d9bb45cd9e1774d6160f4f82dd563ac6e709556042eb8c98478" + }, + { + "path": "CybORG/CybORG/Shared/Observation.py", + "sha256": "a20f9c3d46f5cfc9e4632ee3baf801041f20ff4077b829e70c1af5a1b68a1186" + }, + { + "path": "CybORG/CybORG/Shared/BlueRewardCalculator.py", + "sha256": "2377df1196db8dbaeff6cb3b03a5e3bb0e690c699c1a8abb1ebefa8ab569b61f" + }, + { + "path": "CybORG/CybORG/Shared/RedRewardCalculator.py", + "sha256": "965b523a60bf01a9c70a59834f5cc07cee5bdb76881df807d3e6d56c8ef6c062" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/Monitor.py", + "sha256": "3a62df8f19cbb7461a17c00553914d8dd14f8f947d24d18c52b0dd93bdaad9a5" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/Analyse.py", + "sha256": "cc4bd1301e2ed47fc92bdce37bc2bd58e86aa7a9dd5f56eae3431c7e87749b03" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/Remove.py", + "sha256": "06cebaea236023425990d07f8b3119c8248d04c212525486a0b6fef81937b4af" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/Restore.py", + "sha256": "cf82d7517b762814f104bce8f135ee7109c9702c60748d1c8baccf7526d39978" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/DiscoverRemoteSystems.py", + "sha256": "2990da59387fd1d143bdb780b993db5b849b2a2e0461012caae8f465ba250a71" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/DiscoverNetworkServices.py", + "sha256": "10f6b5bfb82ca8f77c9f694820931e021244a4149512082f8046508cc4397bb0" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/ExploitRemoteService.py", + "sha256": "55ca88b04d05cc5d50838af6101e108f97c3c09d38999ef7dbcce866e1d5f2dd" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/PrivilegeEscalate.py", + "sha256": "91953c3b21991397c0b312f6e50f826d49a003ffc01fa12f31436d11d7206318" + }, + { + "path": "CybORG/CybORG/Shared/Actions/AbstractActions/Impact.py", + "sha256": "249c28a3aa0730138f327f1f635bcbb5b4adc0979bea82146f2950b051cd489f" } ], "selection": { diff --git a/src/raes_adapters/cyborg/source_ledger.py b/src/raes_adapters/cyborg/source_ledger.py new file mode 100644 index 0000000..bfd3554 --- /dev/null +++ b/src/raes_adapters/cyborg/source_ledger.py @@ -0,0 +1,509 @@ +"""Load and validate the pinned CAGE-2 source-mapping evidence. + +Issue #13 owns these backend-local validation labels and resources. They are +not RAES schemas or vocabularies: portable target authority comes exclusively +from the published ``raes_contracts.contracts.schema_bundle()``. +""" + +from __future__ import annotations + +import ast +import hashlib +import json +import re +from importlib.resources import files +from pathlib import Path +from typing import Any, NamedTuple, cast + +from raes_contracts.contracts import schema_bundle # type: ignore[import-untyped] + +from . import load_qualification + +type JsonValue = None | bool | int | float | str | list[JsonValue] | dict[str, JsonValue] +type JsonObject = dict[str, JsonValue] + +_PACKAGE = __package__ or "raes_adapters.cyborg" +_MAX_LEDGER_BYTES = 1_000_000 +_MAX_LEDGER_ROWS = 1_000 +_MAX_LINE_BYTES = 64_000 +_LOSS_HEADING = re.compile(r"^## (?P[a-z0-9][a-z0-9-]*)\s*$") +_LOSS_TIERS = re.compile(r"^\*\*Equivalence tiers weakened:\*\*\s*(?P.+?)\s*$") +_LINE_SELECTOR = re.compile(r"^lines:(?P[1-9]\d*)-(?P[1-9]\d*)$") + + +class EvidenceSelection(NamedTuple): + """One immutable module-local CAGE-2 mapping evidence selection.""" + + qualification_profile_id: str + ledger_resource: tuple[str, ...] + losses_resource: tuple[str, ...] + + +CAGE2_SOURCE_26CE1C1 = EvidenceSelection( + qualification_profile_id="cage2-cyborg-2.1-source-26ce1c1", + ledger_resource=("mapping", "cage2-source-ledger.jsonl"), + losses_resource=("mapping", "cage2-loss-disclosures.md"), +) + +DISPOSITIONS = frozenset({"mapped", "out-of-scope", "loss-disclosed"}) +REQUIRED_SOURCE_FAMILIES = frozenset( + { + "scenario", + "actions", + "observations", + "rewards", + "wrappers", + "agents", + "evaluation", + "provenance-licensing", + } +) +REQUIRED_FACT_FACETS = frozenset( + { + "topology", + "hosts", + "services", + "accounts", + "privileges", + "roles", + "participants", + "initial-knowledge", + "visibility", + "hidden-truth", + "actions", + "admissibility", + "turn-order", + "trial-lengths", + "termination", + "red-variants", + "seeds", + "stochastic-controls", + "reward-components", + "objectives", + "derived-measures", + "licensing", + "attribution", + } +) +RECOGNIZED_EQUIVALENCE_TIERS = frozenset( + { + "authored-source", + "contract", + "execution-control", + "state/observation", + "outcome/evaluation", + } +) + +_REQUIRED_ROW_FIELDS = frozenset( + { + "source_id", + "source_family", + "fact_facet", + "source_repo", + "source_version", + "source_path", + "source_selector", + "source_digest", + "disposition", + "mapping_rule", + "verification", + } +) +_OPTIONAL_ROW_FIELDS = frozenset( + { + "raes_target", + "qualification_ref", + "loss_disclosure", + "equivalence_tiers", + } +) +_ALLOWED_ROW_FIELDS = _REQUIRED_ROW_FIELDS | _OPTIONAL_ROW_FIELDS + + +class LedgerProblem(NamedTuple): + """A bounded source-ledger validation problem.""" + + row_id: str + field: str + reason: str + + def __str__(self) -> str: + """Render only the bounded row, field, and reason.""" + return f"[{self.row_id}] {self.field}: {self.reason}" + + +def _read_text(*parts: str) -> str: + """Read an installed package resource as UTF-8.""" + return files(_PACKAGE).joinpath(*parts).read_text(encoding="utf-8") + + +def _string(row: JsonObject, field: str) -> str: + value = row.get(field) + return value if isinstance(value, str) else "" + + +def _string_list(row: JsonObject, field: str) -> list[str]: + value = row.get(field) + if not isinstance(value, list) or not all(isinstance(item, str) for item in value): + return [] + return cast(list[str], value) + + +def parse_ledger_rows(text: str) -> list[JsonObject]: + """Strictly parse bounded JSONL with duplicate-key and finite-number checks.""" + if len(text.encode("utf-8")) > _MAX_LEDGER_BYTES: + raise ValueError("ledger exceeds the bounded byte limit") + + def _no_duplicate_keys(pairs: list[tuple[str, JsonValue]]) -> JsonObject: + result: JsonObject = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate JSON key: {key!r}") + result[key] = value + return result + + def _reject_non_finite(value: str) -> float: + raise ValueError(f"non-finite number is not permitted: {value}") + + rows: list[JsonObject] = [] + for line_number, line in enumerate(text.splitlines(), start=1): + if not line.strip(): + continue + if len(line.encode("utf-8")) > _MAX_LINE_BYTES: + raise ValueError(f"ledger line {line_number} exceeds the bounded byte limit") + value = json.loads( + line, + object_pairs_hook=_no_duplicate_keys, + parse_constant=_reject_non_finite, + ) + if not isinstance(value, dict): + raise ValueError(f"ledger line {line_number} is not a JSON object") + rows.append(cast(JsonObject, value)) + if len(rows) > _MAX_LEDGER_ROWS: + raise ValueError("ledger exceeds the bounded row limit") + return rows + + +def load_source_ledger( + selection: EvidenceSelection = CAGE2_SOURCE_26CE1C1, +) -> list[JsonObject]: + """Load the selected CAGE-2 JSONL ledger.""" + return parse_ledger_rows(_read_text(*selection.ledger_resource)) + + +def parse_loss_disclosures(text: str) -> dict[str, set[str]]: + """Parse every disclosure heading and its exact weakened-tier set.""" + disclosures: dict[str, set[str]] = {} + current: str | None = None + for line in text.splitlines(): + heading = _LOSS_HEADING.match(line) + if heading: + if current is not None: + disclosures.setdefault(current, set()) + current = heading.group("loss_id") + continue + tier_line = _LOSS_TIERS.match(line) + if tier_line and current is not None: + disclosures[current] = { + tier.strip() for tier in tier_line.group("tiers").split(",") if tier.strip() + } + current = None + if current is not None: + disclosures.setdefault(current, set()) + return disclosures + + +def load_loss_disclosures( + selection: EvidenceSelection = CAGE2_SOURCE_26CE1C1, +) -> dict[str, set[str]]: + """Load the selected disclosure id-to-weakened-tiers mapping.""" + return parse_loss_disclosures(_read_text(*selection.losses_resource)) + + +def _resolve_json_pointer(root: JsonValue, pointer: str) -> bool: + if pointer == "": + return True + if not pointer.startswith("/"): + return False + current = root + for raw_part in pointer.removeprefix("/").split("/"): + part = raw_part.replace("~1", "/").replace("~0", "~") + if isinstance(current, dict): + if part not in current: + return False + current = current[part] + elif isinstance(current, list) and part.isdigit(): + index = int(part) + if index >= len(current): + return False + current = current[index] + else: + return False + return True + + +def resolve_raes_target(target: str, bundle: dict[str, JsonObject]) -> bool: + """Resolve ``#`` through ``schema_bundle``.""" + schema_id, separator, pointer = target.partition("#") + if not separator or schema_id not in bundle: + return False + return _resolve_json_pointer(bundle[schema_id], pointer) + + +def resolve_qualification_ref(reference: str, qualification: dict[str, Any]) -> bool: + """Resolve a qualification-owned evidence JSON pointer.""" + prefix = "qualification.json#" + if not reference.startswith(prefix): + return False + return _resolve_json_pointer(cast(JsonValue, qualification), reference.removeprefix(prefix)) + + +def _selector_shape_is_valid(selector: str) -> bool: + if selector.startswith("python:"): + return bool(selector.removeprefix("python:").strip()) + if selector.startswith("text:"): + return bool(selector.removeprefix("text:").strip()) + match = _LINE_SELECTOR.fullmatch(selector) + return bool(match and int(match.group("start")) <= int(match.group("end"))) + + +def _python_selector_exists(source: str, selector: str) -> bool: + try: + tree = ast.parse(source) + except SyntaxError: + return False + wanted = selector.removeprefix("python:") + symbols: set[str] = set() + for node in tree.body: + if isinstance(node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)): + symbols.add(node.name) + if isinstance(node, ast.ClassDef): + symbols.update( + f"{node.name}.{child.name}" + for child in node.body + if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)) + ) + return wanted in symbols + + +def _selector_exists(path: Path, selector: str) -> bool: + try: + source = path.read_text(encoding="utf-8") + except (OSError, UnicodeError): + return False + if selector.startswith("python:"): + return _python_selector_exists(source, selector) + if selector.startswith("text:"): + return selector.removeprefix("text:") in source + match = _LINE_SELECTOR.fullmatch(selector) + if not match: + return False + start = int(match.group("start")) + end = int(match.group("end")) + return start <= end <= len(source.splitlines()) + + +def validate_source_checkout( + source_root: Path, + rows: list[dict[str, Any]] | list[JsonObject], +) -> list[LedgerProblem]: + """Verify row paths, digests, and selectors against a detached source checkout.""" + root = source_root.resolve() + problems: list[LedgerProblem] = [] + for raw_row in rows: + row = cast(dict[str, Any], raw_row) + source_id = row.get("source_id") + row_id = source_id if isinstance(source_id, str) else "" + source_path = row.get("source_path") + source_digest = row.get("source_digest") + selector = row.get("source_selector") + if not isinstance(source_path, str): + problems.append(LedgerProblem(row_id, "source_path", "missing source path")) + continue + relative = Path(source_path) + candidate = (root / relative).resolve() + if relative.is_absolute() or not candidate.is_relative_to(root) or not candidate.is_file(): + problems.append(LedgerProblem(row_id, "source_path", "source path is not confined")) + continue + digest = hashlib.sha256(candidate.read_bytes()).hexdigest() + if not isinstance(source_digest, str) or digest != source_digest: + problems.append(LedgerProblem(row_id, "source_digest", "source digest drift")) + if not isinstance(selector, str) or not _selector_exists(candidate, selector): + problems.append(LedgerProblem(row_id, "source_selector", "selector does not resolve")) + return problems + + +def validate_source_ledger( + rows: list[JsonObject], + *, + qualification: dict[str, Any], + bundle: dict[str, JsonObject], + losses: dict[str, set[str]], +) -> list[LedgerProblem]: + """Validate row shape, coverage, source joins, targets, and disclosures.""" + problems: list[LedgerProblem] = [] + selected_files = { + item["path"]: item["sha256"] + for item in qualification.get("selected_files", []) + if isinstance(item, dict) + and isinstance(item.get("path"), str) + and isinstance(item.get("sha256"), str) + } + source = qualification.get("source", {}) + expected_repo = source.get("repository") if isinstance(source, dict) else None + expected_version = source.get("commit") if isinstance(source, dict) else None + seen_ids: set[str] = set() + source_families: set[str] = set() + fact_facets: set[str] = set() + referenced_losses: set[str] = set() + + for index, row in enumerate(rows, start=1): + row_id = _string(row, "source_id") or f"" + unknown_fields = sorted(set(row) - _ALLOWED_ROW_FIELDS) + problems.extend( + LedgerProblem(row_id, field, "unknown row field") for field in unknown_fields + ) + for field in sorted(_REQUIRED_ROW_FIELDS): + value = row.get(field) + if not isinstance(value, str) or not value.strip(): + problems.append(LedgerProblem(row_id, field, "missing or blank required value")) + for field in ("raes_target", "qualification_ref", "loss_disclosure"): + if field in row: + value = row[field] + if not isinstance(value, str) or not value.strip(): + problems.append( + LedgerProblem(row_id, field, "optional value must be a nonblank string") + ) + if "equivalence_tiers" in row: + raw_tiers = row["equivalence_tiers"] + if ( + not isinstance(raw_tiers, list) + or not raw_tiers + or not all(isinstance(tier, str) and tier.strip() for tier in raw_tiers) + or len(set(cast(list[str], raw_tiers))) != len(raw_tiers) + ): + problems.append( + LedgerProblem( + row_id, + "equivalence_tiers", + "tier set must be a nonempty unique string list", + ) + ) + + if row_id in seen_ids: + problems.append(LedgerProblem(row_id, "source_id", "duplicate source id")) + seen_ids.add(row_id) + + source_family = _string(row, "source_family") + fact_facet = _string(row, "fact_facet") + disposition = _string(row, "disposition") + if source_family not in REQUIRED_SOURCE_FAMILIES: + problems.append(LedgerProblem(row_id, "source_family", "unknown source family")) + else: + source_families.add(source_family) + if fact_facet not in REQUIRED_FACT_FACETS: + problems.append(LedgerProblem(row_id, "fact_facet", "unknown fact facet")) + else: + fact_facets.add(fact_facet) + if disposition not in DISPOSITIONS: + problems.append(LedgerProblem(row_id, "disposition", "unknown disposition")) + + if _string(row, "source_repo") != expected_repo: + problems.append(LedgerProblem(row_id, "source_repo", "source profile mismatch")) + if _string(row, "source_version") != expected_version: + problems.append(LedgerProblem(row_id, "source_version", "source profile mismatch")) + path = _string(row, "source_path") + expected_digest = selected_files.get(path) + if expected_digest is None: + problems.append(LedgerProblem(row_id, "source_path", "path is not qualified")) + elif _string(row, "source_digest") != expected_digest: + problems.append(LedgerProblem(row_id, "source_digest", "qualified digest drift")) + if not _selector_shape_is_valid(_string(row, "source_selector")): + problems.append(LedgerProblem(row_id, "source_selector", "invalid selector shape")) + + target = _string(row, "raes_target") + qualification_ref = _string(row, "qualification_ref") + loss_id = _string(row, "loss_disclosure") + tiers = set(_string_list(row, "equivalence_tiers")) + if target and not resolve_raes_target(target, bundle): + problems.append(LedgerProblem(row_id, "raes_target", "target does not resolve")) + if qualification_ref and not resolve_qualification_ref(qualification_ref, qualification): + problems.append( + LedgerProblem( + row_id, "qualification_ref", "qualification reference does not resolve" + ) + ) + + if disposition == "mapped": + if not target: + problems.append(LedgerProblem(row_id, "raes_target", "mapped row requires target")) + if loss_id or tiers: + problems.append( + LedgerProblem(row_id, "loss_disclosure", "mapped row cannot declare a loss") + ) + elif disposition == "out-of-scope": + if target: + problems.append( + LedgerProblem(row_id, "raes_target", "out-of-scope row cannot map a target") + ) + if loss_id or tiers: + problems.append( + LedgerProblem( + row_id, + "loss_disclosure", + "out-of-scope row cannot declare a loss", + ) + ) + elif disposition == "loss-disclosed": + if not loss_id: + problems.append( + LedgerProblem( + row_id, "loss_disclosure", "loss-disclosed row requires reference" + ) + ) + else: + referenced_losses.add(loss_id) + unknown_tiers = tiers - RECOGNIZED_EQUIVALENCE_TIERS + if not tiers or unknown_tiers: + problems.append( + LedgerProblem(row_id, "equivalence_tiers", "unrecognized or missing tier") + ) + if loss_id and losses.get(loss_id) != tiers: + problems.append( + LedgerProblem(row_id, "equivalence_tiers", "tier set disagrees with disclosure") + ) + + for missing in sorted(REQUIRED_SOURCE_FAMILIES - source_families): + problems.append(LedgerProblem("", "source_family", f"missing {missing}")) + for missing in sorted(REQUIRED_FACT_FACETS - fact_facets): + problems.append(LedgerProblem("", "fact_facet", f"missing {missing}")) + for loss_id, tiers in losses.items(): + if not tiers or tiers - RECOGNIZED_EQUIVALENCE_TIERS: + problems.append( + LedgerProblem(loss_id, "equivalence_tiers", "disclosure has invalid tier set") + ) + if loss_id not in referenced_losses: + problems.append(LedgerProblem(loss_id, "loss_disclosure", "orphan disclosure")) + return problems + + +def validate_all( + selection: EvidenceSelection = CAGE2_SOURCE_26CE1C1, +) -> list[LedgerProblem]: + """Validate the selected installed evidence set without network access.""" + qualification = load_qualification() + if qualification.get("profile_id") != selection.qualification_profile_id: + return [ + LedgerProblem( + "", + "qualification_profile_id", + "qualification profile mismatch", + ) + ] + return validate_source_ledger( + load_source_ledger(selection), + qualification=qualification, + bundle=cast(dict[str, JsonObject], schema_bundle()), + losses=load_loss_disclosures(selection), + ) diff --git a/tests/test_cyborg_source_ledger.py b/tests/test_cyborg_source_ledger.py new file mode 100644 index 0000000..7fc361b --- /dev/null +++ b/tests/test_cyborg_source_ledger.py @@ -0,0 +1,297 @@ +"""Pinned CAGE-2 source-ledger evidence and fail-closed validation.""" + +from __future__ import annotations + +import copy +import hashlib +from pathlib import Path +from typing import Any + +import pytest +from raes_contracts.contracts import schema_bundle + +import raes_adapters.cyborg.source_ledger as sl + + +def _context() -> dict[str, Any]: + return { + "qualification": sl.load_qualification(), + "bundle": schema_bundle(), + "losses": sl.load_loss_disclosures(), + } + + +def _validate(rows: list[dict[str, Any]]) -> list[sl.LedgerProblem]: + return sl.validate_source_ledger(rows, **_context()) + + +def _row(source_id: str) -> dict[str, Any]: + for row in sl.load_source_ledger(): + if row["source_id"] == source_id: + return copy.deepcopy(row) + raise AssertionError(f"no such row: {source_id}") + + +def test_checked_in_evidence_set_is_complete_and_valid() -> None: + assert sl.validate_all() == [] + + rows = sl.load_source_ledger() + assert {row["source_family"] for row in rows} == set(sl.REQUIRED_SOURCE_FAMILIES) + assert {row["fact_facet"] for row in rows} == set(sl.REQUIRED_FACT_FACETS) + assert {row["disposition"] for row in rows} <= sl.DISPOSITIONS + assert len({row["source_id"] for row in rows}) == len(rows) + + +def test_losses_bind_exactly_to_adr_069_equivalence_tiers() -> None: + losses = sl.load_loss_disclosures() + assert losses + assert set().union(*losses.values()) <= sl.RECOGNIZED_EQUIVALENCE_TIERS + + referenced = { + row["loss_disclosure"] + for row in sl.load_source_ledger() + if row["disposition"] == "loss-disclosed" + } + assert referenced == set(losses) + + +def test_legal_and_attribution_rows_are_reusable_qualification_evidence() -> None: + legal_rows = [ + row for row in sl.load_source_ledger() if row["source_family"] == "provenance-licensing" + ] + assert legal_rows + assert all(row["disposition"] == "out-of-scope" for row in legal_rows) + assert all( + row["qualification_ref"].startswith("qualification.json#/legal") for row in legal_rows + ) + assert all("raes_target" not in row for row in legal_rows) + + +def test_strict_jsonl_rejects_duplicate_keys_non_objects_and_non_finite_numbers() -> None: + with pytest.raises(ValueError, match="duplicate JSON key"): + sl.parse_ledger_rows('{"source_id": "a", "source_id": "b"}') + with pytest.raises(ValueError, match="not a JSON object"): + sl.parse_ledger_rows("[1, 2, 3]") + with pytest.raises(ValueError, match="non-finite"): + sl.parse_ledger_rows('{"n": NaN}') + + +def test_duplicate_id_unknown_field_and_blank_required_value_are_rejected() -> None: + rows = sl.load_source_ledger() + rows.append(copy.deepcopy(rows[0])) + assert any( + problem.field == "source_id" and "duplicate" in problem.reason + for problem in _validate(rows) + ) + + row = _row("scenario-topology") + row["metadata"] = {"hidden": "semantics"} + assert any( + problem.field == "metadata" and "unknown" in problem.reason for problem in _validate([row]) + ) + + row = _row("scenario-topology") + row["source_selector"] = " " + assert any(problem.field == "source_selector" for problem in _validate([row])) + + +@pytest.mark.parametrize( + ("field", "value"), + [ + ("source_family", "other-simulator"), + ("fact_facet", "vibes"), + ("disposition", "mostly-mapped"), + ], +) +def test_unknown_classification_is_rejected(field: str, value: str) -> None: + row = _row("scenario-topology") + row[field] = value + assert any(problem.field == field for problem in _validate([row])) + + +def test_missing_source_family_and_fact_facet_are_rejected() -> None: + rows = [row for row in sl.load_source_ledger() if row["source_family"] != "observations"] + assert any( + problem.field == "source_family" and "observations" in problem.reason + for problem in _validate(rows) + ) + + rows = [row for row in sl.load_source_ledger() if row["fact_facet"] != "admissibility"] + assert any( + problem.field == "fact_facet" and "admissibility" in problem.reason + for problem in _validate(rows) + ) + + +def test_source_profile_path_and_digest_drift_are_rejected() -> None: + row = _row("scenario-topology") + row["source_repo"] = "https://example.invalid/not-the-qualified-source" + assert any(problem.field == "source_repo" for problem in _validate([row])) + + row = _row("scenario-topology") + row["source_version"] = "deadbeef" * 5 + assert any(problem.field == "source_version" for problem in _validate([row])) + + row = _row("scenario-topology") + row["source_path"] = "not/qualified.py" + assert any(problem.field == "source_path" for problem in _validate([row])) + + row = _row("scenario-topology") + row["source_digest"] = "0" * 64 + assert any(problem.field == "source_digest" for problem in _validate([row])) + + +def test_raes_targets_resolve_only_through_the_published_schema_bundle() -> None: + bundle = schema_bundle() + assert sl.resolve_raes_target( + "sdl-authoring-input-v1#/properties/nodes", + bundle, + ) + assert sl.resolve_raes_target( + "experiment-authoring-input-v1#/properties/run_plan", + bundle, + ) + assert sl.resolve_raes_target("experiment-derived-measure-v1#", bundle) + assert not sl.resolve_raes_target("not-a-contract#/properties/nodes", bundle) + assert not sl.resolve_raes_target("sdl-authoring-input-v1#/properties/not_real", bundle) + assert not sl.resolve_raes_target("sdl-authoring-input-v1#not-a-pointer", bundle) + + +def test_mapped_out_of_scope_and_loss_disclosed_shapes_fail_closed() -> None: + row = _row("scenario-topology") + del row["raes_target"] + assert any(problem.field == "raes_target" for problem in _validate([row])) + + row = _row("scenario-topology") + row["loss_disclosure"] = "loss-native-observation-boundary" + row["equivalence_tiers"] = ["state/observation"] + assert any( + problem.field == "loss_disclosure" and "mapped row cannot declare a loss" in problem.reason + for problem in _validate([row]) + ) + + row = _row("legal-root-license") + row["raes_target"] = "sdl-authoring-input-v1#/properties/nodes" + assert any(problem.field == "raes_target" for problem in _validate([row])) + + row = _row("legal-root-license") + row["loss_disclosure"] = "loss-native-observation-boundary" + row["equivalence_tiers"] = ["state/observation"] + assert any( + problem.field == "loss_disclosure" + and "out-of-scope row cannot declare a loss" in problem.reason + for problem in _validate([row]) + ) + + row = _row("evaluation-unbound-seed") + del row["loss_disclosure"] + assert any(problem.field == "loss_disclosure" for problem in _validate([row])) + + row = _row("evaluation-unbound-seed") + row["equivalence_tiers"] = ["outcome/evaluation"] + assert any(problem.field == "equivalence_tiers" for problem in _validate([row])) + + +def test_optional_fields_are_closed_and_typed() -> None: + row = _row("legal-root-license") + row["qualification_ref"] = 42 + assert any(problem.field == "qualification_ref" for problem in _validate([row])) + + row = _row("evaluation-unbound-seed") + row["equivalence_tiers"] = "execution-control" + assert any(problem.field == "equivalence_tiers" for problem in _validate([row])) + + row = _row("scenario-topology") + row["loss_disclosure"] = [] + assert any(problem.field == "loss_disclosure" for problem in _validate([row])) + + +def test_loss_tier_disagreement_and_orphan_disclosure_are_rejected() -> None: + row = _row("evaluation-unbound-seed") + row["equivalence_tiers"] = ["outcome/evaluation"] + problems = _validate([row]) + assert any( + problem.field == "equivalence_tiers" and "disagrees" in problem.reason + for problem in problems + ) + + context = _context() + context["losses"] = {**context["losses"], "loss-orphan": {"contract"}} + problems = sl.validate_source_ledger(sl.load_source_ledger(), **context) + assert any( + problem.row_id == "loss-orphan" and problem.field == "loss_disclosure" + for problem in problems + ) + + +def test_qualification_json_pointers_are_machine_resolvable() -> None: + qualification = sl.load_qualification() + assert sl.resolve_qualification_ref("qualification.json#/legal", qualification) + assert sl.resolve_qualification_ref( + "qualification.json#/admissibility/decision", + qualification, + ) + assert not sl.resolve_qualification_ref("qualification.json#/not-there", qualification) + assert not sl.resolve_qualification_ref("../qualification.json#/legal", qualification) + + +def _checkout_row(path: Path, selector: str) -> dict[str, str]: + return { + "source_id": path.stem, + "source_path": path.name, + "source_digest": hashlib.sha256(path.read_bytes()).hexdigest(), + "source_selector": selector, + } + + +def test_source_checkout_verifies_python_line_and_text_selectors(tmp_path: Path) -> None: + python_source = tmp_path / "source.py" + python_source.write_text( + "class Example:\n def method(self) -> None:\n pass\n", + encoding="utf-8", + ) + yaml_source = tmp_path / "scenario.yaml" + yaml_source.write_text("Agents:\n Blue:\n actions:\n - Sleep\n", encoding="utf-8") + text_source = tmp_path / "README.md" + text_source.write_text("# CAGE Challenge 2\n\nAttribution facts.\n", encoding="utf-8") + + rows = [ + _checkout_row(python_source, "python:Example.method"), + _checkout_row(yaml_source, "lines:1-4"), + _checkout_row(text_source, "text:CAGE Challenge 2"), + ] + assert sl.validate_source_checkout(tmp_path, rows) == [] + + +def test_source_checkout_rejects_unreachable_selector_digest_drift_and_escape( + tmp_path: Path, +) -> None: + source = tmp_path / "source.py" + source.write_text("class Example:\n pass\n", encoding="utf-8") + + row = _checkout_row(source, "python:Missing") + assert any( + problem.field == "source_selector" + for problem in sl.validate_source_checkout(tmp_path, [row]) + ) + + row = _checkout_row(source, "python:Example") + row["source_digest"] = "0" * 64 + assert any( + problem.field == "source_digest" for problem in sl.validate_source_checkout(tmp_path, [row]) + ) + + row = _checkout_row(source, "python:Example") + row["source_path"] = "../source.py" + assert any( + problem.field == "source_path" for problem in sl.validate_source_checkout(tmp_path, [row]) + ) + + +def test_native_payload_fields_cannot_be_added_to_rows() -> None: + for field in ("native_state", "observation_values", "reward_vector", "action_id"): + row = _row("scenario-topology") + row[field] = "forbidden" + assert any( + problem.field == field and "unknown" in problem.reason for problem in _validate([row]) + ) diff --git a/tools/verify_cyborg_qualification.py b/tools/verify_cyborg_qualification.py index 2e59b3f..d9b6418 100644 --- a/tools/verify_cyborg_qualification.py +++ b/tools/verify_cyborg_qualification.py @@ -15,6 +15,7 @@ from urllib.request import ProxyHandler, Request, build_opener import raes_adapters.cyborg as cyborg +import raes_adapters.cyborg.source_ledger as source_ledger REPO_ROOT = Path(__file__).resolve().parent.parent _RUNTIME_FILES = { @@ -310,6 +311,12 @@ def verify_qualification(repo_root: Path = REPO_ROOT) -> None: if any(identities[key] != source_record[key] for key in identities): raise RuntimeError("CybORG qualification source identity mismatch") _verify_source_files(source, record["selected_files"]) + source_problems = source_ledger.validate_source_checkout( + source, + source_ledger.load_source_ledger(), + ) + if source_problems: + raise RuntimeError("CybORG source ledger source verification failed") archive_url = ( "https://codeload.github.com/cage-challenge/cage-challenge-2/tar.gz/" From c426b76a95814cfa0f1b2eb54c1b5220d92b70f6 Mon Sep 17 00:00:00 2001 From: Brad Edwards Date: Thu, 30 Jul 2026 05:36:20 +0200 Subject: [PATCH 2/3] Fix SonarCloud findings (cycle 1) --- src/raes_adapters/cyborg/source_ledger.py | 591 +++++++++++++++------- 1 file changed, 402 insertions(+), 189 deletions(-) diff --git a/src/raes_adapters/cyborg/source_ledger.py b/src/raes_adapters/cyborg/source_ledger.py index bfd3554..b74dc97 100644 --- a/src/raes_adapters/cyborg/source_ledger.py +++ b/src/raes_adapters/cyborg/source_ledger.py @@ -10,7 +10,6 @@ import ast import hashlib import json -import re from importlib.resources import files from pathlib import Path from typing import Any, NamedTuple, cast @@ -26,9 +25,11 @@ _MAX_LEDGER_BYTES = 1_000_000 _MAX_LEDGER_ROWS = 1_000 _MAX_LINE_BYTES = 64_000 -_LOSS_HEADING = re.compile(r"^## (?P[a-z0-9][a-z0-9-]*)\s*$") -_LOSS_TIERS = re.compile(r"^\*\*Equivalence tiers weakened:\*\*\s*(?P.+?)\s*$") -_LINE_SELECTOR = re.compile(r"^lines:(?P[1-9]\d*)-(?P[1-9]\d*)$") +_LOSS_HEADING_PREFIX = "## " +_LOSS_TIERS_PREFIX = "**Equivalence tiers weakened:**" +_PYTHON_SELECTOR_PREFIX = "python:" +_TEXT_SELECTOR_PREFIX = "text:" +_LINE_SELECTOR_PREFIX = "lines:" class EvidenceSelection(NamedTuple): @@ -139,11 +140,13 @@ def _read_text(*parts: str) -> str: def _string(row: JsonObject, field: str) -> str: + """Return a string field or a closed empty sentinel.""" value = row.get(field) return value if isinstance(value, str) else "" def _string_list(row: JsonObject, field: str) -> list[str]: + """Return a homogeneous string-list field or a closed empty sentinel.""" value = row.get(field) if not isinstance(value, list) or not all(isinstance(item, str) for item in value): return [] @@ -156,6 +159,7 @@ def parse_ledger_rows(text: str) -> list[JsonObject]: raise ValueError("ledger exceeds the bounded byte limit") def _no_duplicate_keys(pairs: list[tuple[str, JsonValue]]) -> JsonObject: + """Build an object while refusing duplicate JSON member names.""" result: JsonObject = {} for key, value in pairs: if key in result: @@ -164,6 +168,7 @@ def _no_duplicate_keys(pairs: list[tuple[str, JsonValue]]) -> JsonObject: return result def _reject_non_finite(value: str) -> float: + """Reject JSON parser extensions for NaN and infinities.""" raise ValueError(f"non-finite number is not permitted: {value}") rows: list[JsonObject] = [] @@ -192,22 +197,41 @@ def load_source_ledger( return parse_ledger_rows(_read_text(*selection.ledger_resource)) +def _loss_heading_id(line: str) -> str | None: + """Return a strictly bounded ASCII loss id from a Markdown heading.""" + if not line.startswith(_LOSS_HEADING_PREFIX): + return None + candidate = line.removeprefix(_LOSS_HEADING_PREFIX).strip() + allowed = frozenset("abcdefghijklmnopqrstuvwxyz0123456789-") + starts = frozenset("abcdefghijklmnopqrstuvwxyz0123456789") + valid = bool(candidate) and candidate[0] in starts and set(candidate) <= allowed + return candidate if valid else None + + +def _loss_tiers(line: str) -> set[str] | None: + """Return the comma-separated tiers from the canonical disclosure line.""" + if not line.startswith(_LOSS_TIERS_PREFIX): + return None + raw_tiers = line.removeprefix(_LOSS_TIERS_PREFIX).strip() + if not raw_tiers: + return None + return {tier.strip() for tier in raw_tiers.split(",") if tier.strip()} + + def parse_loss_disclosures(text: str) -> dict[str, set[str]]: """Parse every disclosure heading and its exact weakened-tier set.""" disclosures: dict[str, set[str]] = {} current: str | None = None for line in text.splitlines(): - heading = _LOSS_HEADING.match(line) - if heading: + heading = _loss_heading_id(line) + if heading is not None: if current is not None: disclosures.setdefault(current, set()) - current = heading.group("loss_id") + current = heading continue - tier_line = _LOSS_TIERS.match(line) - if tier_line and current is not None: - disclosures[current] = { - tier.strip() for tier in tier_line.group("tiers").split(",") if tier.strip() - } + tiers = _loss_tiers(line) + if tiers is not None and current is not None: + disclosures[current] = tiers current = None if current is not None: disclosures.setdefault(current, set()) @@ -222,25 +246,26 @@ def load_loss_disclosures( def _resolve_json_pointer(root: JsonValue, pointer: str) -> bool: - if pointer == "": - return True - if not pointer.startswith("/"): - return False + """Resolve a JSON pointer without exposing the referenced value.""" + valid = pointer == "" or pointer.startswith("/") current = root - for raw_part in pointer.removeprefix("/").split("/"): + parts = [] if pointer == "" else pointer.removeprefix("/").split("/") + for raw_part in parts: + if not valid: + break part = raw_part.replace("~1", "/").replace("~0", "~") if isinstance(current, dict): - if part not in current: - return False - current = current[part] + valid = part in current + if valid: + current = current[part] elif isinstance(current, list) and part.isdigit(): index = int(part) - if index >= len(current): - return False - current = current[index] + valid = index < len(current) + if valid: + current = current[index] else: - return False - return True + valid = False + return valid def resolve_raes_target(target: str, bundle: dict[str, JsonObject]) -> bool: @@ -260,20 +285,43 @@ def resolve_qualification_ref(reference: str, qualification: dict[str, Any]) -> def _selector_shape_is_valid(selector: str) -> bool: - if selector.startswith("python:"): - return bool(selector.removeprefix("python:").strip()) - if selector.startswith("text:"): - return bool(selector.removeprefix("text:").strip()) - match = _LINE_SELECTOR.fullmatch(selector) - return bool(match and int(match.group("start")) <= int(match.group("end"))) + """Validate a selector's bounded syntax without opening its source.""" + if selector.startswith(_PYTHON_SELECTOR_PREFIX): + valid = bool(selector.removeprefix(_PYTHON_SELECTOR_PREFIX).strip()) + elif selector.startswith(_TEXT_SELECTOR_PREFIX): + valid = bool(selector.removeprefix(_TEXT_SELECTOR_PREFIX).strip()) + else: + valid = _line_selector_bounds(selector) is not None + return valid + + +def _line_selector_bounds(selector: str) -> tuple[int, int] | None: + """Parse a positive inclusive line range without regex backtracking.""" + if not selector.startswith(_LINE_SELECTOR_PREFIX): + return None + start_text, separator, end_text = selector.removeprefix(_LINE_SELECTOR_PREFIX).partition("-") + valid = ( + bool(separator) + and start_text.isascii() + and end_text.isascii() + and start_text.isdigit() + and end_text.isdigit() + and not start_text.startswith("0") + and not end_text.startswith("0") + ) + if not valid: + return None + start, end = int(start_text), int(end_text) + return (start, end) if start <= end else None def _python_selector_exists(source: str, selector: str) -> bool: + """Resolve a top-level Python class/function or class method via the AST.""" try: tree = ast.parse(source) except SyntaxError: return False - wanted = selector.removeprefix("python:") + wanted = selector.removeprefix(_PYTHON_SELECTOR_PREFIX) symbols: set[str] = set() for node in tree.body: if isinstance(node, (ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)): @@ -288,20 +336,53 @@ def _python_selector_exists(source: str, selector: str) -> bool: def _selector_exists(path: Path, selector: str) -> bool: + """Resolve one supported selector against UTF-8 source text.""" try: source = path.read_text(encoding="utf-8") except (OSError, UnicodeError): - return False - if selector.startswith("python:"): - return _python_selector_exists(source, selector) - if selector.startswith("text:"): - return selector.removeprefix("text:") in source - match = _LINE_SELECTOR.fullmatch(selector) - if not match: - return False - start = int(match.group("start")) - end = int(match.group("end")) - return start <= end <= len(source.splitlines()) + exists = False + else: + bounds = _line_selector_bounds(selector) + if selector.startswith(_PYTHON_SELECTOR_PREFIX): + exists = _python_selector_exists(source, selector) + elif selector.startswith(_TEXT_SELECTOR_PREFIX): + exists = selector.removeprefix(_TEXT_SELECTOR_PREFIX) in source + elif bounds is not None: + _, end = bounds + exists = end <= len(source.splitlines()) + else: + exists = False + return exists + + +def _confined_source_path(root: Path, source_path: str) -> Path | None: + """Resolve a relative regular-file path confined beneath the source root.""" + relative = Path(source_path) + candidate = (root / relative).resolve() + confined = not relative.is_absolute() and candidate.is_relative_to(root) and candidate.is_file() + return candidate if confined else None + + +def _validate_checkout_row(root: Path, row: dict[str, Any]) -> list[LedgerProblem]: + """Validate one ledger row against a detached qualified checkout.""" + source_id = row.get("source_id") + row_id = source_id if isinstance(source_id, str) else "" + source_path = row.get("source_path") + if not isinstance(source_path, str): + return [LedgerProblem(row_id, "source_path", "missing source path")] + + candidate = _confined_source_path(root, source_path) + if candidate is None: + return [LedgerProblem(row_id, "source_path", "source path is not confined")] + + problems: list[LedgerProblem] = [] + digest = hashlib.sha256(candidate.read_bytes()).hexdigest() + if row.get("source_digest") != digest: + problems.append(LedgerProblem(row_id, "source_digest", "source digest drift")) + selector = row.get("source_selector") + if not isinstance(selector, str) or not _selector_exists(candidate, selector): + problems.append(LedgerProblem(row_id, "source_selector", "selector does not resolve")) + return problems def validate_source_checkout( @@ -312,37 +393,27 @@ def validate_source_checkout( root = source_root.resolve() problems: list[LedgerProblem] = [] for raw_row in rows: - row = cast(dict[str, Any], raw_row) - source_id = row.get("source_id") - row_id = source_id if isinstance(source_id, str) else "" - source_path = row.get("source_path") - source_digest = row.get("source_digest") - selector = row.get("source_selector") - if not isinstance(source_path, str): - problems.append(LedgerProblem(row_id, "source_path", "missing source path")) - continue - relative = Path(source_path) - candidate = (root / relative).resolve() - if relative.is_absolute() or not candidate.is_relative_to(root) or not candidate.is_file(): - problems.append(LedgerProblem(row_id, "source_path", "source path is not confined")) - continue - digest = hashlib.sha256(candidate.read_bytes()).hexdigest() - if not isinstance(source_digest, str) or digest != source_digest: - problems.append(LedgerProblem(row_id, "source_digest", "source digest drift")) - if not isinstance(selector, str) or not _selector_exists(candidate, selector): - problems.append(LedgerProblem(row_id, "source_selector", "selector does not resolve")) + problems.extend(_validate_checkout_row(root, cast(dict[str, Any], raw_row))) return problems -def validate_source_ledger( - rows: list[JsonObject], - *, +class _LedgerValidationContext(NamedTuple): + """Immutable dependencies shared by per-row ledger validation.""" + + selected_files: dict[str, str] + expected_repo: Any + expected_version: Any + qualification: dict[str, Any] + bundle: dict[str, JsonObject] + losses: dict[str, set[str]] + + +def _validation_context( qualification: dict[str, Any], bundle: dict[str, JsonObject], losses: dict[str, set[str]], -) -> list[LedgerProblem]: - """Validate row shape, coverage, source joins, targets, and disclosures.""" - problems: list[LedgerProblem] = [] +) -> _LedgerValidationContext: + """Build exact qualification and contract joins for ledger validation.""" selected_files = { item["path"]: item["sha256"] for item in qualification.get("selected_files", []) @@ -353,6 +424,262 @@ def validate_source_ledger( source = qualification.get("source", {}) expected_repo = source.get("repository") if isinstance(source, dict) else None expected_version = source.get("commit") if isinstance(source, dict) else None + return _LedgerValidationContext( + selected_files, + expected_repo, + expected_version, + qualification, + bundle, + losses, + ) + + +def _tiers_field_is_valid(raw_tiers: JsonValue) -> bool: + """Accept only a nonempty unique list of nonblank tier strings.""" + if not isinstance(raw_tiers, list) or not raw_tiers: + return False + strings = [tier for tier in raw_tiers if isinstance(tier, str)] + return ( + len(strings) == len(raw_tiers) + and all(tier.strip() for tier in strings) + and len(set(strings)) == len(strings) + ) + + +def _validate_row_shape(row: JsonObject, row_id: str) -> list[LedgerProblem]: + """Validate the row's closed field set and scalar/list value shapes.""" + problems = [ + LedgerProblem(row_id, field, "unknown row field") + for field in sorted(set(row) - _ALLOWED_ROW_FIELDS) + ] + problems.extend( + LedgerProblem(row_id, field, "missing or blank required value") + for field in sorted(_REQUIRED_ROW_FIELDS) + if not isinstance(row.get(field), str) or not cast(str, row[field]).strip() + ) + problems.extend( + LedgerProblem(row_id, field, "optional value must be a nonblank string") + for field in ("raes_target", "qualification_ref", "loss_disclosure") + if field in row and (not isinstance(row[field], str) or not cast(str, row[field]).strip()) + ) + if "equivalence_tiers" in row and not _tiers_field_is_valid(row["equivalence_tiers"]): + problems.append( + LedgerProblem( + row_id, + "equivalence_tiers", + "tier set must be a nonempty unique string list", + ) + ) + return problems + + +def _validate_classification( + row: JsonObject, + row_id: str, + source_families: set[str], + fact_facets: set[str], +) -> list[LedgerProblem]: + """Validate classification values and record their coverage.""" + problems: list[LedgerProblem] = [] + source_family = _string(row, "source_family") + fact_facet = _string(row, "fact_facet") + disposition = _string(row, "disposition") + if source_family in REQUIRED_SOURCE_FAMILIES: + source_families.add(source_family) + else: + problems.append(LedgerProblem(row_id, "source_family", "unknown source family")) + if fact_facet in REQUIRED_FACT_FACETS: + fact_facets.add(fact_facet) + else: + problems.append(LedgerProblem(row_id, "fact_facet", "unknown fact facet")) + if disposition not in DISPOSITIONS: + problems.append(LedgerProblem(row_id, "disposition", "unknown disposition")) + return problems + + +def _validate_source_join( + row: JsonObject, + row_id: str, + context: _LedgerValidationContext, +) -> list[LedgerProblem]: + """Validate profile identity, qualified file identity, and selector syntax.""" + problems: list[LedgerProblem] = [] + if _string(row, "source_repo") != context.expected_repo: + problems.append(LedgerProblem(row_id, "source_repo", "source profile mismatch")) + if _string(row, "source_version") != context.expected_version: + problems.append(LedgerProblem(row_id, "source_version", "source profile mismatch")) + path = _string(row, "source_path") + expected_digest = context.selected_files.get(path) + if expected_digest is None: + problems.append(LedgerProblem(row_id, "source_path", "path is not qualified")) + elif _string(row, "source_digest") != expected_digest: + problems.append(LedgerProblem(row_id, "source_digest", "qualified digest drift")) + if not _selector_shape_is_valid(_string(row, "source_selector")): + problems.append(LedgerProblem(row_id, "source_selector", "invalid selector shape")) + return problems + + +def _validate_references( + row: JsonObject, + row_id: str, + context: _LedgerValidationContext, +) -> list[LedgerProblem]: + """Resolve optional RAES and qualification evidence references.""" + problems: list[LedgerProblem] = [] + target = _string(row, "raes_target") + qualification_ref = _string(row, "qualification_ref") + if target and not resolve_raes_target(target, context.bundle): + problems.append(LedgerProblem(row_id, "raes_target", "target does not resolve")) + if qualification_ref and not resolve_qualification_ref( + qualification_ref, + context.qualification, + ): + problems.append( + LedgerProblem( + row_id, + "qualification_ref", + "qualification reference does not resolve", + ) + ) + return problems + + +def _validate_mapped_disposition( + row_id: str, + target: str, + loss_id: str, + tiers: set[str], +) -> list[LedgerProblem]: + """Enforce the target-only shape of a fully mapped row.""" + problems: list[LedgerProblem] = [] + if not target: + problems.append(LedgerProblem(row_id, "raes_target", "mapped row requires target")) + if loss_id or tiers: + problems.append( + LedgerProblem(row_id, "loss_disclosure", "mapped row cannot declare a loss") + ) + return problems + + +def _validate_out_of_scope_disposition( + row_id: str, + target: str, + loss_id: str, + tiers: set[str], +) -> list[LedgerProblem]: + """Enforce the no-target/no-loss shape of an out-of-scope row.""" + problems: list[LedgerProblem] = [] + if target: + problems.append( + LedgerProblem(row_id, "raes_target", "out-of-scope row cannot map a target") + ) + if loss_id or tiers: + problems.append( + LedgerProblem( + row_id, + "loss_disclosure", + "out-of-scope row cannot declare a loss", + ) + ) + return problems + + +def _validate_loss_disclosed_disposition( + row_id: str, + loss_id: str, + tiers: set[str], + losses: dict[str, set[str]], + referenced_losses: set[str], +) -> list[LedgerProblem]: + """Enforce exact disclosure references and recognized weakened tiers.""" + problems: list[LedgerProblem] = [] + if loss_id: + referenced_losses.add(loss_id) + else: + problems.append( + LedgerProblem(row_id, "loss_disclosure", "loss-disclosed row requires reference") + ) + if not tiers or tiers - RECOGNIZED_EQUIVALENCE_TIERS: + problems.append(LedgerProblem(row_id, "equivalence_tiers", "unrecognized or missing tier")) + if loss_id and losses.get(loss_id) != tiers: + problems.append( + LedgerProblem(row_id, "equivalence_tiers", "tier set disagrees with disclosure") + ) + return problems + + +def _validate_disposition( + row: JsonObject, + row_id: str, + context: _LedgerValidationContext, + referenced_losses: set[str], +) -> list[LedgerProblem]: + """Dispatch the row to its closed disposition validator.""" + disposition = _string(row, "disposition") + target = _string(row, "raes_target") + loss_id = _string(row, "loss_disclosure") + tiers = set(_string_list(row, "equivalence_tiers")) + if disposition == "mapped": + problems = _validate_mapped_disposition(row_id, target, loss_id, tiers) + elif disposition == "out-of-scope": + problems = _validate_out_of_scope_disposition(row_id, target, loss_id, tiers) + elif disposition == "loss-disclosed": + problems = _validate_loss_disclosed_disposition( + row_id, + loss_id, + tiers, + context.losses, + referenced_losses, + ) + else: + problems = [] + return problems + + +def _coverage_problems( + source_families: set[str], + fact_facets: set[str], +) -> list[LedgerProblem]: + """Report every missing source-family and fact-facet coverage label.""" + problems = [ + LedgerProblem("", "source_family", f"missing {missing}") + for missing in sorted(REQUIRED_SOURCE_FAMILIES - source_families) + ] + problems.extend( + LedgerProblem("", "fact_facet", f"missing {missing}") + for missing in sorted(REQUIRED_FACT_FACETS - fact_facets) + ) + return problems + + +def _disclosure_problems( + losses: dict[str, set[str]], + referenced_losses: set[str], +) -> list[LedgerProblem]: + """Report invalid tier sets and disclosures not referenced by ledger rows.""" + problems = [ + LedgerProblem(loss_id, "equivalence_tiers", "disclosure has invalid tier set") + for loss_id, tiers in losses.items() + if not tiers or tiers - RECOGNIZED_EQUIVALENCE_TIERS + ] + problems.extend( + LedgerProblem(loss_id, "loss_disclosure", "orphan disclosure") + for loss_id in losses + if loss_id not in referenced_losses + ) + return problems + + +def validate_source_ledger( + rows: list[JsonObject], + *, + qualification: dict[str, Any], + bundle: dict[str, JsonObject], + losses: dict[str, set[str]], +) -> list[LedgerProblem]: + """Validate row shape, coverage, source joins, targets, and disclosures.""" + context = _validation_context(qualification, bundle, losses) + problems: list[LedgerProblem] = [] seen_ids: set[str] = set() source_families: set[str] = set() fact_facets: set[str] = set() @@ -360,131 +687,17 @@ def validate_source_ledger( for index, row in enumerate(rows, start=1): row_id = _string(row, "source_id") or f"" - unknown_fields = sorted(set(row) - _ALLOWED_ROW_FIELDS) - problems.extend( - LedgerProblem(row_id, field, "unknown row field") for field in unknown_fields - ) - for field in sorted(_REQUIRED_ROW_FIELDS): - value = row.get(field) - if not isinstance(value, str) or not value.strip(): - problems.append(LedgerProblem(row_id, field, "missing or blank required value")) - for field in ("raes_target", "qualification_ref", "loss_disclosure"): - if field in row: - value = row[field] - if not isinstance(value, str) or not value.strip(): - problems.append( - LedgerProblem(row_id, field, "optional value must be a nonblank string") - ) - if "equivalence_tiers" in row: - raw_tiers = row["equivalence_tiers"] - if ( - not isinstance(raw_tiers, list) - or not raw_tiers - or not all(isinstance(tier, str) and tier.strip() for tier in raw_tiers) - or len(set(cast(list[str], raw_tiers))) != len(raw_tiers) - ): - problems.append( - LedgerProblem( - row_id, - "equivalence_tiers", - "tier set must be a nonempty unique string list", - ) - ) - + problems.extend(_validate_row_shape(row, row_id)) if row_id in seen_ids: problems.append(LedgerProblem(row_id, "source_id", "duplicate source id")) seen_ids.add(row_id) + problems.extend(_validate_classification(row, row_id, source_families, fact_facets)) + problems.extend(_validate_source_join(row, row_id, context)) + problems.extend(_validate_references(row, row_id, context)) + problems.extend(_validate_disposition(row, row_id, context, referenced_losses)) - source_family = _string(row, "source_family") - fact_facet = _string(row, "fact_facet") - disposition = _string(row, "disposition") - if source_family not in REQUIRED_SOURCE_FAMILIES: - problems.append(LedgerProblem(row_id, "source_family", "unknown source family")) - else: - source_families.add(source_family) - if fact_facet not in REQUIRED_FACT_FACETS: - problems.append(LedgerProblem(row_id, "fact_facet", "unknown fact facet")) - else: - fact_facets.add(fact_facet) - if disposition not in DISPOSITIONS: - problems.append(LedgerProblem(row_id, "disposition", "unknown disposition")) - - if _string(row, "source_repo") != expected_repo: - problems.append(LedgerProblem(row_id, "source_repo", "source profile mismatch")) - if _string(row, "source_version") != expected_version: - problems.append(LedgerProblem(row_id, "source_version", "source profile mismatch")) - path = _string(row, "source_path") - expected_digest = selected_files.get(path) - if expected_digest is None: - problems.append(LedgerProblem(row_id, "source_path", "path is not qualified")) - elif _string(row, "source_digest") != expected_digest: - problems.append(LedgerProblem(row_id, "source_digest", "qualified digest drift")) - if not _selector_shape_is_valid(_string(row, "source_selector")): - problems.append(LedgerProblem(row_id, "source_selector", "invalid selector shape")) - - target = _string(row, "raes_target") - qualification_ref = _string(row, "qualification_ref") - loss_id = _string(row, "loss_disclosure") - tiers = set(_string_list(row, "equivalence_tiers")) - if target and not resolve_raes_target(target, bundle): - problems.append(LedgerProblem(row_id, "raes_target", "target does not resolve")) - if qualification_ref and not resolve_qualification_ref(qualification_ref, qualification): - problems.append( - LedgerProblem( - row_id, "qualification_ref", "qualification reference does not resolve" - ) - ) - - if disposition == "mapped": - if not target: - problems.append(LedgerProblem(row_id, "raes_target", "mapped row requires target")) - if loss_id or tiers: - problems.append( - LedgerProblem(row_id, "loss_disclosure", "mapped row cannot declare a loss") - ) - elif disposition == "out-of-scope": - if target: - problems.append( - LedgerProblem(row_id, "raes_target", "out-of-scope row cannot map a target") - ) - if loss_id or tiers: - problems.append( - LedgerProblem( - row_id, - "loss_disclosure", - "out-of-scope row cannot declare a loss", - ) - ) - elif disposition == "loss-disclosed": - if not loss_id: - problems.append( - LedgerProblem( - row_id, "loss_disclosure", "loss-disclosed row requires reference" - ) - ) - else: - referenced_losses.add(loss_id) - unknown_tiers = tiers - RECOGNIZED_EQUIVALENCE_TIERS - if not tiers or unknown_tiers: - problems.append( - LedgerProblem(row_id, "equivalence_tiers", "unrecognized or missing tier") - ) - if loss_id and losses.get(loss_id) != tiers: - problems.append( - LedgerProblem(row_id, "equivalence_tiers", "tier set disagrees with disclosure") - ) - - for missing in sorted(REQUIRED_SOURCE_FAMILIES - source_families): - problems.append(LedgerProblem("", "source_family", f"missing {missing}")) - for missing in sorted(REQUIRED_FACT_FACETS - fact_facets): - problems.append(LedgerProblem("", "fact_facet", f"missing {missing}")) - for loss_id, tiers in losses.items(): - if not tiers or tiers - RECOGNIZED_EQUIVALENCE_TIERS: - problems.append( - LedgerProblem(loss_id, "equivalence_tiers", "disclosure has invalid tier set") - ) - if loss_id not in referenced_losses: - problems.append(LedgerProblem(loss_id, "loss_disclosure", "orphan disclosure")) + problems.extend(_coverage_problems(source_families, fact_facets)) + problems.extend(_disclosure_problems(losses, referenced_losses)) return problems From 9fc2894328e0b262808d65b4fa15362a4b1bb75b Mon Sep 17 00:00:00 2001 From: Brad Edwards Date: Thu, 30 Jul 2026 05:43:06 +0200 Subject: [PATCH 3/3] Fix SonarCloud findings (cycle 2) --- src/raes_adapters/cyborg/source_ledger.py | 45 ++++++++++++----------- 1 file changed, 24 insertions(+), 21 deletions(-) diff --git a/src/raes_adapters/cyborg/source_ledger.py b/src/raes_adapters/cyborg/source_ledger.py index b74dc97..1183bdc 100644 --- a/src/raes_adapters/cyborg/source_ledger.py +++ b/src/raes_adapters/cyborg/source_ledger.py @@ -245,27 +245,28 @@ def load_loss_disclosures( return parse_loss_disclosures(_read_text(*selection.losses_resource)) +def _json_pointer_step(current: JsonValue, part: str) -> tuple[bool, JsonValue]: + """Resolve one decoded JSON-pointer segment.""" + if isinstance(current, dict) and part in current: + return True, current[part] + if isinstance(current, list) and part.isdigit(): + index = int(part) + if index < len(current): + return True, current[index] + return False, current + + def _resolve_json_pointer(root: JsonValue, pointer: str) -> bool: """Resolve a JSON pointer without exposing the referenced value.""" - valid = pointer == "" or pointer.startswith("/") + if pointer != "" and not pointer.startswith("/"): + return False current = root - parts = [] if pointer == "" else pointer.removeprefix("/").split("/") - for raw_part in parts: - if not valid: - break + for raw_part in [] if pointer == "" else pointer.removeprefix("/").split("/"): part = raw_part.replace("~1", "/").replace("~0", "~") - if isinstance(current, dict): - valid = part in current - if valid: - current = current[part] - elif isinstance(current, list) and part.isdigit(): - index = int(part) - valid = index < len(current) - if valid: - current = current[index] - else: - valid = False - return valid + valid, current = _json_pointer_step(current, part) + if not valid: + return False + return True def resolve_raes_target(target: str, bundle: dict[str, JsonObject]) -> bool: @@ -401,8 +402,8 @@ class _LedgerValidationContext(NamedTuple): """Immutable dependencies shared by per-row ledger validation.""" selected_files: dict[str, str] - expected_repo: Any - expected_version: Any + expected_repo: str | None + expected_version: str | None qualification: dict[str, Any] bundle: dict[str, JsonObject] losses: dict[str, set[str]] @@ -422,8 +423,10 @@ def _validation_context( and isinstance(item.get("sha256"), str) } source = qualification.get("source", {}) - expected_repo = source.get("repository") if isinstance(source, dict) else None - expected_version = source.get("commit") if isinstance(source, dict) else None + source_repo = source.get("repository") if isinstance(source, dict) else None + source_version = source.get("commit") if isinstance(source, dict) else None + expected_repo = source_repo if isinstance(source_repo, str) else None + expected_version = source_version if isinstance(source_version, str) else None return _LedgerValidationContext( selected_files, expected_repo,