From 0435fcc58f0dce3aceaa57c1d2bf487634de16aa Mon Sep 17 00:00:00 2001 From: flashls1 Date: Thu, 10 Sep 2026 23:59:50 -0500 Subject: [PATCH 1/5] Plan post-merge OvernightLab continuity repair --- .../IMPLEMENTATION_PLAN.md | 29 ++++++++++ .../PREFLIGHT.md | 58 +++++++++++++++++++ 2 files changed, 87 insertions(+) create mode 100644 .clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/IMPLEMENTATION_PLAN.md create mode 100644 .clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/PREFLIGHT.md diff --git a/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/IMPLEMENTATION_PLAN.md b/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/IMPLEMENTATION_PLAN.md new file mode 100644 index 0000000..89a2c17 --- /dev/null +++ b/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/IMPLEMENTATION_PLAN.md @@ -0,0 +1,29 @@ +# Implementation Plan — finish the merged OvernightLab live layer and continuity + +Supersedes no performance plan and admits no new optimization candidate. It continues the merged Sept-10 results-first obligation only far enough to make the already-approved local control plane truthful, durable, and resumable. + +## Execution contract + +1. Add a minimal `OvernightLab/install.command` that installs only source/config/controller material plus the existing classifier source/build helper into `/Volumes/MAC MINI M4/TFTMAC`, preserves existing generated evidence, compiles the classifier, and performs static verification. It must not copy or alter Control, DEV app, SDK, emulator, AVD, system image, TFT package, credentials, or unrelated project files. +2. Add a minimal `OvernightLab/run-overnight-campaign.command`: `--self-test` runs controller self-test plus fault-test; ordinary arguments execute the existing `campaign` CLI under `caffeinate`. Do not duplicate campaign state logic in shell. +3. Update `project.md` current state so the active change is `b42fb30e-2f22-45f0-9d27-00d9e67a58bf`, PR #9 merge SHA is recorded, and the old RUNNING campaign is not presented as live after reconciliation. +4. Append `CHANGELOG.md` with the post-merge continuity/live-layer result. Keep `DEV-B8-WIN-01`; no performance version promotion occurs. +5. Validate source locally before live effects. +6. Install the small control plane into the external TFTMAC root. +7. Copy/merge only the ignored historical OvernightLab evidence from the closed `ff2f...` worktree into the live OvernightLab location, preserving source from the new install and preserving generated evidence byte-for-byte. Never delete the source evidence origin until durable preservation is proven. +8. Verify copied evidence identities/counts/hashes at a bounded representative level and preserve the database/campaign directories completely. +9. Reconcile `overnight-20260910T160337Z-501dd433` against actual host state using existing `reconcile_resume` without entering `run_campaign`. Expected result: stale RUNNING becomes recovered/interrupted with baseline restored; no candidate replay. +10. Regenerate the historical report after reconciliation and verify it remains available. +11. Run installed `verify-static`, `self-test`, and `fault-test`. Confirm no TFTMAC DEV/qemu campaign process was started by those validations. +12. Re-read current `facts.md` and `project.md`; resolve any new factual drift before finalization. +13. Run repository validation, review exact diff, checkpoint/publish, deliver PR, require exact-SHA CI green, merge, and reconcile clean source state. +14. Because this repository is SOURCE_ONLY, independently verify the local installed control-plane hashes/acceptance after merge. Do not invent a remote deployment. +15. Stop this repair at the current results-first decision boundary. The existing automatic queue remains `control` only. A future performance test requires one deliberately admitted, evidence-backed hypothesis from the then-current authority; resolved P1/P2/P3 and rejected transport candidates are not rerun automatically. + +## Exclusions + +No new app/runtime/SDK/AVD copy. No Control/LKG mutation. No storage cleanup. No new VM/builder. No generic telemetry expansion. No unbounded profiling. No gameplay/performance candidate during this repair. No resurrection of PBE, direct Vulkan, buffer-retention, queue-submit-inline, virtual-queue-off, fence-contexts-off, or historical global-sync as a current candidate. + +## Acceptance proof + +Green source validation + exact source diff/review + live install hash parity + installed static/self/fault tests + preserved historical evidence + non-running reconciled stale campaign + exact-SHA GitHub CI + merge + clean selected worktree/local post-merge live verification. diff --git a/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/PREFLIGHT.md b/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/PREFLIGHT.md new file mode 100644 index 0000000..b34c3c1 --- /dev/null +++ b/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/PREFLIGHT.md @@ -0,0 +1,58 @@ +# Preflight — post-merge TFTMAC OvernightLab continuity repair + +Date: 2026-09-10 America/Chicago +Change: `b42fb30e-2f22-45f0-9d27-00d9e67a58bf` +Completion class: IMPLEMENT_SHIP +Base: merged `master` SHA `bf61c21a723f2f132834acedda860efbb2223d42` + +## Controlling authority + +Read and adopted in this fresh Zoe session: current `facts.md`, `project.md`, `CHANGELOG.md`, and `.clara/plans/ff2f318b-245d-418a-b86f-e07d55b19826/RECOVERY_CONSTRAINTS_2026-09-10.md`. + +Current DEV authority is `DEV-B8-WIN-01`: official TFT `18.1-5423749` / `8423749`, 1920x1080 / 320 DPI / 60 Hz, effective 8 vCPU / 6144 MiB, host GPU/CoreAudio, selected game RHI `OPENGL_ES_ANGLE`, multifile cache ON, `preferSubmitAtFBOBoundary` disabled, and `syncMonolithicPipelinesToBlobCache` removed. Protected Control/LKG remains immutable. + +## Saved-point reconciliation + +1. Saved change `ff2f318b-245d-418a-b86f-e07d55b19826` published exact SHA `9674294dd557a8ed7250c34deb6e9ca3f8d05f86`. +2. Exact GitHub check `Validate TFTMAC` run `34559035858`, job `103137754606`, completed SUCCESS on that SHA. +3. PR #9 was then squash-merged as `bf61c21a723f2f132834acedda860efbb2223d42`; the prior managed change is correctly closed. +4. The ignored historical OvernightLab evidence still exists in the closed worktree, including the campaign/database data recorded by the saved checkpoint. It must not be lost merely because the source change merged. +5. The old campaign `overnight-20260910T160337Z-501dd433` has a stale `RUNNING` checkpoint, but current host reconciliation found no OvernightLab controller, TFTMAC DEV core, or owned qemu/5586 process. Its last run reached Tocker stage 1-5 but has no `result.json`; it is interrupted historical evidence, not a current promotion result. +6. `/Volumes/MAC MINI M4/TFTMAC/OvernightLab` does not currently exist. Therefore the plan's required local live layer is not actually installed despite README claims. +7. Merged source contains `OvernightLab/overnight_lab.py`, `README.md`, schema, authority, and manifest, but README references `install.command` and `run-overnight-campaign.command` that are absent. +8. `overnight_lab.py` derives `PROJECT_ROOT` as its parent and requires `tools/tft-screen-classifier.swift` plus `scripts/build-tft-screen-classifier.command`; a naive copy of only the OvernightLab directory would fail static acceptance. +9. Current manifest automatic queue is only `control`; all previously admitted fast-pass optimization families are already resolved in the current record books. This repair must not silently add or rerun a performance candidate. + +## Simplest correct mechanism + +Add only the missing source-controlled install and campaign wrapper seams. The installer copies the small OvernightLab source/config plus exactly the classifier source/build helper needed by the existing code into `/Volumes/MAC MINI M4/TFTMAC`, compiles the classifier, and runs static verification. It never copies an app, SDK, emulator, AVD, system image, game package, or credentials. + +Separately, before the closed worktree can be reclaimed, preserve its ignored runtime evidence into the new live OvernightLab location without deleting or rewriting it. Reconcile the stale campaign checkpoint against actual host state using the already-implemented `reconcile_resume` behavior, but do not run or resume a candidate as part of reconciliation. + +## Required source/document corrections + +- Add `OvernightLab/install.command`. +- Add `OvernightLab/run-overnight-campaign.command`. +- Correct `project.md` current-state text that still names the now-merged/closed `ff2f...` change as active; name this continuation change and record PR #9 merge/live-layer recovery state. +- Append a continuity entry to `CHANGELOG.md` for this tooling/live-layer repair. Do not create a new `DEV-B8-WIN-##` because no performance candidate is being tested. +- Update README command semantics only if required by the implemented wrapper behavior. +- `facts.md` changes only if a newly verified hard project fact requires it; current performance/runtime authority remains unchanged. + +## Acceptance + +- Protected Control and frozen LKG are never mutated. +- Closed-worktree ignored telemetry is durably preserved before any possible worktree cleanup. +- Live install exists at `/Volumes/MAC MINI M4/TFTMAC/OvernightLab` and contains no forbidden runtime/app copies. +- Installed source/config hashes match the selected managed source. +- Classifier builds and self-tests at the installed location. +- `verify-static`, `self-test`, and `fault-test` pass from the installed location. +- Stale old campaign is reconciled to a non-running interrupted/recovered historical state without replaying it; its evidence remains readable/reportable. +- No new gameplay/performance candidate runs during this repair. +- Source validator passes, diff/review is clean, and the selected worktree finishes clean. +- Source is delivered through PR/CI/merge. Project deployment profile is SOURCE_ONLY; the requested live layer for this change is the verified local install. + +## ZenGate / ZenMC qualification + +ZenGate basis: the missing live-install seam is a proven blocker to the already-approved live layer, and the proposed additions survive the removal test. No new architecture, service, scheduler, store, runtime clone, or experimental family is introduced. + +`ZENMC_NOT_REQUIRED` for the new source delta: the wrappers add no new lifecycle/state semantics. Campaign recovery, locking, rollback, quarantine, and resume behavior remain owned by the existing previously-tested Python controller. This change validates those existing recovery paths rather than redesigning them. From a64e9e5fa7dc7f0a02048bb06b0fbe1d2ac20162 Mon Sep 17 00:00:00 2001 From: flashls1 Date: Fri, 11 Sep 2026 00:05:33 -0500 Subject: [PATCH 2/5] Reconcile TFTMAC authority and evidence recovery route --- .../AMENDMENT_001.md | 33 +++++++++++++++++++ project.md | 6 ++-- 2 files changed, 36 insertions(+), 3 deletions(-) create mode 100644 .clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/AMENDMENT_001.md diff --git a/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/AMENDMENT_001.md b/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/AMENDMENT_001.md new file mode 100644 index 0000000..7a33e92 --- /dev/null +++ b/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/AMENDMENT_001.md @@ -0,0 +1,33 @@ +# Amendment 001 — post-close evidence recovery route + +Date: 2026-09-11 America/Chicago +Change: `b42fb30e-2f22-45f0-9d27-00d9e67a58bf` +Supersedes only the PRE/PLAN assumption that ignored OvernightLab generated evidence still exists in the closed `ff2f318b...` worktree and can be copied from there. All other scope, safety, and results-first constraints remain in force. + +## New direct evidence + +1. PR #9 merge closed change `ff2f318b-245d-418a-b86f-e07d55b19826`; its worktree is now absent. +2. Bounded searches found no old `overnight-20260910T160337Z-501dd433` directory, `TFTMAC_OVERNIGHT.sqlite`, or `active-campaign.txt` in remaining TFTMAC worktrees, `/Volumes/MAC MINI M4/TFTMAC`, user Trash, Clara durable areas searched, Spotlight results, or local Time Machine snapshots. +3. `tmutil listlocalsnapshots` reports no local snapshots for `/Volumes/MAC MINI M4` or `/`. +4. The authoritative DEV native capture root is `~/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures`, not the generic historical `~/Library/Application Support/TFTMAC/Captures` path used by an earlier bounded check. +5. Twelve Sept. 10 advanced-diagnostics capture directories remain in that native capture root for the relevant campaign window, each with `TFTMAC_NATIVE_RUNTIME.sqlite`. The final saved-run launch maps to capture `2026-09-10T22-43-10.664Z-a5718134-6211-4bcb-8bd6-c17b134e8a6f`, whose native DB remains present at 7,368,704 bytes. A later retained capture `2026-09-10T19-45-09.532Z-9d9d31f1-4359-4b57-bd0b-4ed818d72db8` remains present with a 20,217,856-byte native DB. + +## Classification + +This is a **BLOCKING DEPENDENCY for the original copy-from-closed-worktree preservation step**, not a reason to stop the project and not a reason to rerun resolved performance candidates. The derived OvernightLab campaign DB/screenshots/reports that lived only as ignored worktree files cannot be truthfully claimed recovered from current local storage. However, the authoritative native session telemetry survives externally, and merged source plus current record books preserve the verified decisions/winner. + +## Revised execution contract + +1. Keep the minimal wrapper/source repair already defined. +2. Install the small merged OvernightLab source/config/classifier helper into `/Volumes/MAC MINI M4/TFTMAC/OvernightLab` without touching Control/LKG/DEV runtime binaries. +3. Do **not** create a fake replacement for the lost old `TFTMAC_OVERNIGHT.sqlite`, campaign screenshots, or result files. Record their absence explicitly. +4. Inventory and seal the surviving relevant native capture directories/SQLite identities needed for continuity. These remain the primary raw performance evidence. +5. Initialize a fresh OvernightLab database only through normal current controller startup/self-test behavior; historical verified decisions remain sourced from `CHANGELOG.md`/`project.md` and surviving native captures, not synthesized rows. +6. Update `project.md` and `CHANGELOG.md` with the continuity fact: PR #9 merged; derived ignored worktree layer was removed with closed-worktree cleanup and has no local snapshot recovery route; authoritative native captures remain available and are the recovery evidence source. +7. Run installed `verify-static`, `self-test`, and `fault-test`. No gameplay/performance candidate is run in this continuity repair. +8. Preserve the current automatic queue as `control` only. Do not automatically resume the stale historical campaign identity, because its campaign DB/checkpoint no longer exists at the live layer. +9. Source validation/review/CI/merge and local live-install verification remain required. + +## Acceptance adjustment + +The old requirement "stale old campaign is reconciled in its original OvernightLab DB" is superseded because that DB is no longer locally recoverable. Replacement acceptance is: the loss is truthfully documented; surviving native captures are proven present; the new live OvernightLab starts from current authority without inventing historical rows; and no resolved candidate is replayed merely to reconstruct deleted derived telemetry. diff --git a/project.md b/project.md index 318f60d..73535c0 100644 --- a/project.md +++ b/project.md @@ -1,15 +1,15 @@ # TFTMAC Project Record -> **LIVING WIKI / CURRENT DEV STATE — updated 2026-09-10 America/Chicago.** This top section is the current mutable project state. Older dated sections below are preserved as project history and must not override this section when they conflict with newer verified evidence. +> **LIVING WIKI / CURRENT DEV STATE — updated 2026-09-11 America/Chicago.** This top section is the current mutable project state. Older dated sections below are preserved as project history and must not override this section when they conflict with newer verified evidence. **Project:** native macOS TFT client experience using the official Android TFT package -**Current development line:** `master` remains merged repository authority; active optimization work is isolated in Clara change `ff2f318b-245d-418a-b86f-e07d55b19826` against the DEV / `advanced_diagnostics` product. +**Current development line:** `master` remains merged repository authority. OvernightLab authority reconciliation change `ff2f318b-245d-418a-b86f-e07d55b19826` merged through PR #9 at `bf61c21a723f2f132834acedda860efbb2223d42` and is closed. Current post-merge continuity/live-layer recovery is isolated in Clara change `b42fb30e-2f22-45f0-9d27-00d9e67a58bf`; it does not admit a new performance candidate. **Protected release/LKG:** TFTMAC 2.3.0 build 8 Control/LKG remains separate, frozen, and available as historical rollback/comparison authority. **Current DEV application identity:** TFTMAC 2.3.0 build 8 DEV (`/Applications/TFTMAC DEV.app`, bundle `com.flashls1.tftmac.dev`). Test-ledger versions do not change the product release number. **Current official client:** `com.riotgames.league.teamfighttactics` `18.1-5423749`, versionCode `8423749`. **Current test series:** `DEV-B8-2026-09-10-A`. **Current verified working winner:** **`DEV-B8-WIN-01`**. -**Project record current through:** 2026-09-10 results-first verification pass. +**Project record current through:** 2026-09-11 post-PR-#9 authority rehydration; performance winner remains the 2026-09-10 results-first verification state. ## 0. Living wiki contract From 5803c879a2d7301520154c91978acee2f7f2010b Mon Sep 17 00:00:00 2001 From: flashls1 Date: Fri, 11 Sep 2026 00:13:39 -0500 Subject: [PATCH 3/5] Restore TFTMAC OvernightLab live continuity --- CHANGELOG.md | 12 ++ .../native-capture-recovery-2026-09-11.json | 106 ++++++++++++++++++ OvernightLab/install.command | 46 ++++++++ OvernightLab/run-overnight-campaign.command | 17 +++ project.md | 2 + ssot/AUTHORITY_INPUTS.sha256 | 2 +- ssot/STACK.lock.yaml | 4 +- 7 files changed, 186 insertions(+), 3 deletions(-) create mode 100644 OvernightLab/authority/native-capture-recovery-2026-09-11.json create mode 100755 OvernightLab/install.command create mode 100755 OvernightLab/run-overnight-campaign.command diff --git a/CHANGELOG.md b/CHANGELOG.md index b2b9cfa..8ddd3fb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -165,6 +165,18 @@ This is a verified **frame-pacing / tail-latency win**, not a claim that mean FP - Added the clean-workspace completion rule: the selected managed change cannot be called complete with accidental dirty Git state. - Historical Control values (6 vCPU / 5120 MiB / ports 5038/5582/8554) remain valid only when explicitly labeled protected Control/history and no longer describe the current DEV optimization baseline. +#### Post-merge OvernightLab continuity — 2026-09-11 + +**Outcome:** `SOURCE MERGED / LIVE-LAYER RECOVERY REQUIRED` +**Performance integration:** `NONE` — `DEV-B8-WIN-01` remains the current winner; no performance candidate was run or promoted. + +- PR #9 passed exact-SHA `Validate TFTMAC` CI on `9674294dd557a8ed7250c34deb6e9ca3f8d05f86` and squash-merged to `master` as `bf61c21a723f2f132834acedda860efbb2223d42`. +- The merged source retains the reconciled schema-2 OvernightLab authority, current-winner control manifest and telemetry policy. +- After Clara closed the merged `ff2f318b...` worktree, the ignored derived OvernightLab campaign/database/screenshots that existed only under that worktree were no longer locally present. Bounded recovery searches found no copy in the remaining TFTMAC worktrees, live TFTMAC root, Trash, Clara durable areas searched, Spotlight results, or local Time Machine snapshots. The project must not represent those derived files as recovered. +- The authoritative native DEV capture store remains present under `~/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures`. Twelve relevant Sept. 10 capture directories were directly observed for the campaign window, each with `TFTMAC_NATIVE_RUNTIME.sqlite`; the final 22:43 UTC session `2026-09-10T22-43-10.664Z-a5718134-6211-4bcb-8bd6-c17b134e8a6f` remains present with a 7,368,704-byte native database. +- Recovery therefore uses surviving native session evidence plus the synchronized record books. A fresh live OvernightLab may be initialized from current authority, but historical derived campaign rows/results must not be synthesized merely to replace deleted local output. +- This continuity repair does not reopen buffer retention, direct Vulkan, queue-submit-inline, virtual-queue-off, fence-contexts-off, historical global-sync, or any other resolved fast-pass candidate. + ### Documentation - Reconciled current Build 8 runtime, automatic-logging, and graphics-causality diff --git a/OvernightLab/authority/native-capture-recovery-2026-09-11.json b/OvernightLab/authority/native-capture-recovery-2026-09-11.json new file mode 100644 index 0000000..da322ee --- /dev/null +++ b/OvernightLab/authority/native-capture-recovery-2026-09-11.json @@ -0,0 +1,106 @@ +{ + "capture_count": 12, + "capture_root": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures", + "captures": [ + { + "capture_id": "2026-09-10T15-56-47.461Z-ecd07248-8fc5-44d9-818c-eb975ca1f3be", + "capture_mtime_utc": "2026-09-10T15:58:32.679582Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T15-56-47.461Z-ecd07248-8fc5-44d9-818c-eb975ca1f3be", + "native_db_bytes": 2297856, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T15-56-47.461Z-ecd07248-8fc5-44d9-818c-eb975ca1f3be/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "1c3f88f87bed70b9135c8cabf2c7a99630b55394b51c36a1ba29b96f4b7e3549" + }, + { + "capture_id": "2026-09-10T16-01-19.641Z-de4d51a4-cb88-4ea3-b887-501059bf9f33", + "capture_mtime_utc": "2026-09-10T16:02:29.306286Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T16-01-19.641Z-de4d51a4-cb88-4ea3-b887-501059bf9f33", + "native_db_bytes": 1228800, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T16-01-19.641Z-de4d51a4-cb88-4ea3-b887-501059bf9f33/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "a163c8e609c697adc6e21cdae85d8a7a9cea9e2f61fb275e2126fd98424415b5" + }, + { + "capture_id": "2026-09-10T16-03-38.498Z-75921468-c167-4020-813b-717a34edad98", + "capture_mtime_utc": "2026-09-10T16:04:06.060240Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T16-03-38.498Z-75921468-c167-4020-813b-717a34edad98", + "native_db_bytes": 3014656, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T16-03-38.498Z-75921468-c167-4020-813b-717a34edad98/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "c010939db47a3583f5cd3d09645ce6a2acad741a966351aee05469812880518d" + }, + { + "capture_id": "2026-09-10T19-02-30.967Z-25f5f0d2-9438-4bee-b18e-a0480f077ccc", + "capture_mtime_utc": "2026-09-10T19:03:37.801019Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-02-30.967Z-25f5f0d2-9438-4bee-b18e-a0480f077ccc", + "native_db_bytes": 970752, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-02-30.967Z-25f5f0d2-9438-4bee-b18e-a0480f077ccc/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "f24247d580ba2d1d05f0679134ac1b45150bba5e591b8b9fc6d263418eb18ad2" + }, + { + "capture_id": "2026-09-10T19-05-48.892Z-0190e735-918b-4313-9d21-6717621db0bd", + "capture_mtime_utc": "2026-09-10T19:08:21.868615Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-05-48.892Z-0190e735-918b-4313-9d21-6717621db0bd", + "native_db_bytes": 1892352, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-05-48.892Z-0190e735-918b-4313-9d21-6717621db0bd/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "3e75a395ad48c088a06e0f1e2641b0d18b00d9cfa45db5a32e6e565c23751a30" + }, + { + "capture_id": "2026-09-10T19-21-58.830Z-41bbd0db-143a-4f90-b885-21c09b84a52c", + "capture_mtime_utc": "2026-09-10T19:22:42.380242Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-21-58.830Z-41bbd0db-143a-4f90-b885-21c09b84a52c", + "native_db_bytes": 417792, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-21-58.830Z-41bbd0db-143a-4f90-b885-21c09b84a52c/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "99d78b98cf38aeae4ce63aa25a06fbbe84c12e502adfc2d7d132d7d3865e447f" + }, + { + "capture_id": "2026-09-10T19-33-36.532Z-f0a8e114-e572-4902-afa0-0ae7a534b1a9", + "capture_mtime_utc": "2026-09-10T19:36:07.970710Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-33-36.532Z-f0a8e114-e572-4902-afa0-0ae7a534b1a9", + "native_db_bytes": 491520, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-33-36.532Z-f0a8e114-e572-4902-afa0-0ae7a534b1a9/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "4517335b08d477186d1e8bbd2d45444f87e45bc58e83b3ed377f7b241ab10f8f" + }, + { + "capture_id": "2026-09-10T19-37-43.322Z-b2c259ea-4c69-449a-ac6e-d7664f937f02", + "capture_mtime_utc": "2026-09-10T19:37:54.695350Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-37-43.322Z-b2c259ea-4c69-449a-ac6e-d7664f937f02", + "native_db_bytes": 413696, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-37-43.322Z-b2c259ea-4c69-449a-ac6e-d7664f937f02/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "2d3d2f19d8c1c21e02ba59069b07c13799f76fc5345c2870334b87ea6eeaa9a7" + }, + { + "capture_id": "2026-09-10T19-39-23.992Z-18472c55-cc15-4fad-89f4-be355775114a", + "capture_mtime_utc": "2026-09-10T19:40:24.362009Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-39-23.992Z-18472c55-cc15-4fad-89f4-be355775114a", + "native_db_bytes": 413696, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-39-23.992Z-18472c55-cc15-4fad-89f4-be355775114a/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "90a6637a4d59d3f5b83fcf153e91bff278581ad9706a52138cb2549b1da81d9a" + }, + { + "capture_id": "2026-09-10T19-41-41.388Z-917d9ab1-4ef2-4e23-aa85-5d02a2cb1f09", + "capture_mtime_utc": "2026-09-10T19:42:04.041359Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-41-41.388Z-917d9ab1-4ef2-4e23-aa85-5d02a2cb1f09", + "native_db_bytes": 3158016, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-41-41.388Z-917d9ab1-4ef2-4e23-aa85-5d02a2cb1f09/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "0b98009b3fc2e32d176c14092c8b6a532a4864ca20aeffa9cf54b59b513d8aeb" + }, + { + "capture_id": "2026-09-10T19-45-09.532Z-9d9d31f1-4359-4b57-bd0b-4ed818d72db8", + "capture_mtime_utc": "2026-09-10T22:48:18.078823Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-45-09.532Z-9d9d31f1-4359-4b57-bd0b-4ed818d72db8", + "native_db_bytes": 20217856, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T19-45-09.532Z-9d9d31f1-4359-4b57-bd0b-4ed818d72db8/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "1eed4156c8115f0c9aeed107d9eafaa9adc5f2869b17ea35a95a857e797be46f" + }, + { + "capture_id": "2026-09-10T22-43-10.664Z-a5718134-6211-4bcb-8bd6-c17b134e8a6f", + "capture_mtime_utc": "2026-09-10T22:43:36.939990Z", + "capture_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T22-43-10.664Z-a5718134-6211-4bcb-8bd6-c17b134e8a6f", + "native_db_bytes": 7368704, + "native_db_path": "/Users/flash/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures/2026-09-10T22-43-10.664Z-a5718134-6211-4bcb-8bd6-c17b134e8a6f/TFTMAC_NATIVE_RUNTIME.sqlite", + "native_db_sha256": "57d02353d93523aced0f7993ae4dcf4be555c5486dd567c07714edd96846d3cd" + } + ], + "observed_utc": "2026-09-11T05:06:09Z", + "schema": 1, + "window_end_utc": "2026-09-11T00:00:00Z", + "window_start_utc": "2026-09-10T15:30:00Z" +} diff --git a/OvernightLab/install.command b/OvernightLab/install.command new file mode 100755 index 0000000..6e528e7 --- /dev/null +++ b/OvernightLab/install.command @@ -0,0 +1,46 @@ +#!/bin/zsh +set -euo pipefail + +readonly SOURCE_LAB="${0:A:h}" +readonly PROJECT_ROOT="${SOURCE_LAB:h}" +readonly LIVE_ROOT="/Volumes/MAC MINI M4/TFTMAC" +readonly LIVE_LAB="$LIVE_ROOT/OvernightLab" + +copy_file() { + local mode="$1" + local source="$2" + local destination="$3" + mkdir -p "${destination:h}" + if [[ "${source:A}" == "${destination:A}" ]]; then + chmod "$mode" "$destination" + return + fi + if [[ -f "$destination" ]] && cmp -s "$source" "$destination"; then + chmod "$mode" "$destination" + return + fi + /usr/bin/install -m "$mode" "$source" "$destination" +} + +[[ -d "$LIVE_ROOT" ]] || { print -u2 -- "TFTMAC live root is missing: $LIVE_ROOT"; exit 2; } + +# Source/config only. Generated campaigns/database/reports/bin remain in place. +copy_file 755 "$SOURCE_LAB/overnight_lab.py" "$LIVE_LAB/overnight_lab.py" +copy_file 755 "$SOURCE_LAB/install.command" "$LIVE_LAB/install.command" +copy_file 755 "$SOURCE_LAB/run-overnight-campaign.command" "$LIVE_LAB/run-overnight-campaign.command" +copy_file 644 "$SOURCE_LAB/README.md" "$LIVE_LAB/README.md" +copy_file 644 "$SOURCE_LAB/schema.sql" "$LIVE_LAB/schema.sql" +copy_file 644 "$SOURCE_LAB/authority/official-client-runtime.json" "$LIVE_LAB/authority/official-client-runtime.json" +copy_file 644 "$SOURCE_LAB/authority/native-capture-recovery-2026-09-11.json" "$LIVE_LAB/authority/native-capture-recovery-2026-09-11.json" +copy_file 644 "$SOURCE_LAB/manifests/official-candidates.json" "$LIVE_LAB/manifests/official-candidates.json" + +# Existing controller code resolves these two classifier inputs from ROOT.parent. +copy_file 644 "$PROJECT_ROOT/tools/tft-screen-classifier.swift" "$LIVE_ROOT/tools/tft-screen-classifier.swift" +copy_file 755 "$PROJECT_ROOT/scripts/build-tft-screen-classifier.command" "$LIVE_ROOT/scripts/build-tft-screen-classifier.command" + +TFT_SCREEN_CLASSIFIER_BINARY="$LIVE_LAB/bin/tft-screen-classifier" \ +TFT_SCREEN_CLASSIFIER_MODULE_CACHE="$LIVE_LAB/bin/swift-module-cache" \ + "$LIVE_ROOT/scripts/build-tft-screen-classifier.command" >/dev/null + +/usr/bin/python3 "$LIVE_LAB/overnight_lab.py" verify-static +print -r -- "$LIVE_LAB" diff --git a/OvernightLab/run-overnight-campaign.command b/OvernightLab/run-overnight-campaign.command new file mode 100755 index 0000000..c195b86 --- /dev/null +++ b/OvernightLab/run-overnight-campaign.command @@ -0,0 +1,17 @@ +#!/bin/zsh +set -euo pipefail + +readonly LAB_ROOT="${0:A:h}" +readonly CONTROLLER="$LAB_ROOT/overnight_lab.py" + +[[ -f "$CONTROLLER" ]] || { print -u2 -- "OvernightLab controller is missing: $CONTROLLER"; exit 2; } + +if [[ "${1:-}" == "--self-test" ]]; then + shift + (( $# == 0 )) || { print -u2 -- "--self-test accepts no additional arguments"; exit 2; } + /usr/bin/python3 "$CONTROLLER" self-test + /usr/bin/python3 "$CONTROLLER" fault-test + exit 0 +fi + +exec /usr/bin/caffeinate -d -i -s -m /usr/bin/python3 "$CONTROLLER" campaign "$@" diff --git a/project.md b/project.md index 73535c0..f80d290 100644 --- a/project.md +++ b/project.md @@ -78,6 +78,8 @@ OvernightLab is retained as a **data-preserving telemetry/provenance layer**, no - Root-only cache file inventory is opt-in; ordinary performance logging relies on property readback and native telemetry so the observer does not restart/disrupt ADB just to collect optional metadata. - The frozen LKG cache set with global pipeline sync remains historical comparator data. OvernightLab's normal control now applies the WIN-01 cache properties with global sync removed. - Generated campaigns, SQLite state, compiled caches/binaries and reports are runtime evidence, not repository source; they must remain locally retained/ignored rather than continually dirtying Git. +- **Post-merge continuity finding (2026-09-11):** after PR #9 merged and Clara closed the `ff2f318b...` worktree, that worktree's ignored OvernightLab campaign/database/screenshots were no longer present. Bounded searches found no copy in remaining TFTMAC worktrees, `/Volumes/MAC MINI M4/TFTMAC`, Trash, Clara durable areas searched, Spotlight results, or local Time Machine snapshots. Do not claim those derived files remain recoverable. +- **Raw evidence continuity remains intact:** the authoritative `~/Library/Application Support/TFTMAC/Modes/advanced_diagnostics/Captures` store still contains the relevant Sept. 10 DEV sessions and native SQLite telemetry, including `2026-09-10T22-43-10.664Z-a5718134-6211-4bcb-8bd6-c17b134e8a6f`. The live OvernightLab recovery must start fresh from current authority and may reference surviving native captures; it must not fabricate deleted historical campaign rows. ### Mandatory update rule after every test diff --git a/ssot/AUTHORITY_INPUTS.sha256 b/ssot/AUTHORITY_INPUTS.sha256 index 63faeab..0dfc24c 100644 --- a/ssot/AUTHORITY_INPUTS.sha256 +++ b/ssot/AUTHORITY_INPUTS.sha256 @@ -2,7 +2,7 @@ # Regenerate only after reviewing the complete authority change set. ee70562e850e82106ac3b07cb2e0ba00acbeef128f0c827edf003a3253aba654 facts.md -c6ae394f3721644737f433c2f667457ae70dee583ec2f9b1c33a7c69dec9756c project.md +fec86184eb4faf346f9aa81a704c9de1efeaea7c450eb5d73392edd1373345ec project.md 53e1416892348c2cd1145344b54f661134b59e1483891c9a3127a0270ce48f56 dev.md f5ae42d870a47565c52433f2d07d9db5b24dee3d914d7d896a160aa130800f5d benchmark.md 143af322d05e3ac8ccae92baf279ed8e6e8f62ce5bd2f757e045868e1f20c2c8 TFTMAC_GPU_RUNTIME_SSOT.md diff --git a/ssot/STACK.lock.yaml b/ssot/STACK.lock.yaml index 35bccc3..afc10bb 100644 --- a/ssot/STACK.lock.yaml +++ b/ssot/STACK.lock.yaml @@ -1,6 +1,6 @@ schema: 3 frozen_at: "2026-09-02T07:59:18Z" -reconciled_at: "2026-09-10 America/Chicago" +reconciled_at: "2026-09-11 America/Chicago" phase0_status: PASS authority: @@ -10,7 +10,7 @@ authority: documentation_authority: "facts.md > project.md > CHANGELOG.md > benchmark.md + dev.md + supporting references" authority_inputs: "ssot/AUTHORITY_INPUTS.sha256" runtime_authority_sha256: "3607fef47c7ca0b7ebd3ced416e0b3eb9e7dd37143624b97da2670ee283c7169" - authority_inputs_sha256: "3ab439c575e87ea9127cc939d7aed43148e1d3c135479a6edb89addf402d9e62" + authority_inputs_sha256: "1760a9bd96fac11c8dfb9927808de69eb3969ad69b438a9b3e9b0a2b6dafe634" zengate_version: "2.3" zengate_score: 95.8 zengate_result: "PASS" From 0fbf4c96af850a93c6619f418992606e793c9e60 Mon Sep 17 00:00:00 2001 From: flashls1 Date: Fri, 11 Sep 2026 00:29:37 -0500 Subject: [PATCH 4/5] Adopt incremental-gains optimization doctrine --- .../AMENDMENT_002_INCREMENTAL_GAINS.md | 62 +++++++++++++++++++ 1 file changed, 62 insertions(+) create mode 100644 .clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/AMENDMENT_002_INCREMENTAL_GAINS.md diff --git a/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/AMENDMENT_002_INCREMENTAL_GAINS.md b/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/AMENDMENT_002_INCREMENTAL_GAINS.md new file mode 100644 index 0000000..418976a --- /dev/null +++ b/.clara/plans/b42fb30e-2f22-45f0-9d27-00d9e67a58bf/AMENDMENT_002_INCREMENTAL_GAINS.md @@ -0,0 +1,62 @@ +# Amendment 002 — incremental-gains optimization doctrine + +Date: 2026-09-11 America/Chicago +Change: `b42fb30e-2f22-45f0-9d27-00d9e67a58bf` +Trigger: explicit Flash scope/goal clarification in the current bound TFTMAC conversation. + +## User outcome + +Continuous useful 60 FPS remains the ultimate product target, but it is **not** the minimum success threshold for each experiment. The active optimization strategy is cumulative: expose and test small existing/research-backed settings one factor at a time, retain every **verified repeatable net improvement**, layer the next candidate on top of the latest winner, and determine empirically whether those small gains compound toward continuous 60 FPS. + +If a candidate does not improve the game, is inconclusive, or regresses it, record the exact failure/result, restore the latest winner, and move on. Do not spend the pass building infrastructure merely to explain a loser. + +## Authority correction + +Current `facts.md`, `project.md`, `CHANGELOG.md`, and recovery constraints already describe cumulative verified-net-win promotion, but `facts.md`, `benchmark.md`, `dev.md`, the Swift benchmark decision engine, and its tests still contain a legacy **5% weighted-FPS promotion/rejection floor**. That floor conflicts with the clarified user goal because a legitimate 1–4% improvement could be rejected before it can compound. + +## Revised decision semantics + +1. `HOME_RUN` remains a label for a large/broad win. Its existing strong thresholds may remain as a standout classification. +2. `PROMISING` becomes the bounded-screen classification for an **incremental positive candidate**: there is at least one directly measured improvement signal and no material veto/regression that outweighs it. There is no fixed +5% weighted-FPS minimum. +3. `REJECT` is reserved for correctness/usability failure or material measured regression, not merely for failing to clear an arbitrary positive-gain percentage. +4. `INCONCLUSIVE` remains for invalid/mismatched evidence or a result that does not establish a directional net gain or regression. +5. A `PROMISING`/incremental win is not automatically permanent from one noisy sample. It requires the existing confirmation discipline. Once the improvement is repeatable and net-positive with no correctness/stability/compatibility/severe-tail veto, promote it as the next `DEV-B8-WIN-##` baseline. +6. 60 FPS remains the cumulative destination and full-run success condition. Until achieved, report the remaining deficit; do not reject a smaller verified step merely because it does not independently reach 60. +7. Existing historical decisions are not rewritten solely because the doctrine changed. Rejected candidates remain rejected unless new evidence/mechanism gives a specific reason to reopen them. + +## Implementation scope + +- Update `facts.md`, `project.md`, `CHANGELOG.md`, `benchmark.md`, and `dev.md` so this cumulative doctrine is explicit and no current authority says sub-5% gain alone is rejection. +- Update `tftmac/Runtime/CombatBenchmarkAnalysis.swift` to remove the legacy `<5% weighted FPS => REJECT` gate and classify small clean directional improvements as `PROMISING` while preserving correctness and material-tail vetoes. +- Update `Tests/TFTMACTests/CombatBenchmarkAnalysisTests.swift` with explicit sub-5% incremental-win coverage, neutral/no-signal inconclusive coverage, correctness rejection, and material-regression rejection. +- Refresh authority-input hashes/STACK lock as required by repository validation. +- Preserve the already-repaired OvernightLab live continuity work in this same change. +- Revalidate, republish PR #10 at the new exact SHA, require fresh exact-SHA CI, merge, and verify the local source-only live layer. + +## Simplest correct code rule + +Do not introduce a new scoring framework or new decision enum. Reuse the existing `PROMISING` classification. + +After validity/correctness and the existing 10% p95/p99 material-regression veto: + +- evaluate `HOME_RUN` first; +- classify `PROMISING` when one of these directly measured families improves: weighted FPS; 1% low; both p95 and p99 frame intervals; or smoothness (jank/severe/missed-vsync) without introducing a material conflicting regression; +- classify an exact/no-directional-change result as `INCONCLUSIVE`, not `REJECT`; +- keep `REJECT` for correctness failure or material regression. + +This deliberately removes the false 5% floor without constructing a synthetic weighted score whose weights would be arbitrary. + +## ZenGate + +PASS. The scope is an explicit user-goal correction and removes a contradictory false gate. The smallest viable mechanism is to revise the existing decision semantics and tests; no new service, datastore, framework, candidate matrix, runtime copy, or telemetry architecture is needed. + +`ZENMC_NOT_REQUIRED`: this changes deterministic comparison classification only; it does not add asynchronous lifecycle, recovery, concurrency, cutover, or external-effect state. + +## Acceptance + +- No current authority or executable decision path rejects a valid candidate solely because weighted FPS gain is below 5%. +- A small directly measured clean gain can reach `PROMISING` and, after existing confirmation, become the next working winner. +- Neutral/no-signal evidence remains `INCONCLUSIVE` rather than a fabricated win. +- Correctness failure and material p95/p99 regression still reject. +- Existing `HOME_RUN` large-win semantics remain available. +- Full source validation and fresh PR exact-SHA CI pass. From afd79fea4ef0e2fec0aadb012f19e2939cc65717 Mon Sep 17 00:00:00 2001 From: flashls1 Date: Fri, 11 Sep 2026 00:42:38 -0500 Subject: [PATCH 5/5] Align TFTMAC with cumulative small-gains optimization --- .../validate-source.mjs | 2 +- CHANGELOG.md | 14 ++++++- .../CombatBenchmarkAnalysisTests.swift | 41 +++++++++++++++---- benchmark.md | 19 ++++----- dev.md | 28 ++++++------- facts.md | 28 +++++-------- project.md | 17 ++++---- scripts/verify-tftmac.command | 2 +- ssot/AUTHORITY_INPUTS.sha256 | 8 ++-- ssot/STACK.lock.yaml | 4 +- tftmac/Runtime/CombatBenchmarkAnalysis.swift | 24 +++++++---- 11 files changed, 108 insertions(+), 79 deletions(-) diff --git a/.clara/plans/tftmac-causal-graphics-v1/diagnostic-first-boot-v1/validate-source.mjs b/.clara/plans/tftmac-causal-graphics-v1/diagnostic-first-boot-v1/validate-source.mjs index 77bacd8..410cf6b 100755 --- a/.clara/plans/tftmac-causal-graphics-v1/diagnostic-first-boot-v1/validate-source.mjs +++ b/.clara/plans/tftmac-causal-graphics-v1/diagnostic-first-boot-v1/validate-source.mjs @@ -106,7 +106,7 @@ assert(tests.includes('testRuntimeModeRegistrySelectsReceiptedDiagnosticsAndReje const testCount = fs.readdirSync(absolute('Tests/TFTMACTests')) .filter((name) => name.endsWith('.swift')) .reduce((total, name) => total + (readText(`Tests/TFTMACTests/${name}`).match(/^ func test/gm) ?? []).length, 0); -assert(testCount === 110, `expected 110 native tests, found ${testCount}`); +assert(testCount === 112, `expected 112 native tests, found ${testCount}`); console.log(JSON.stringify({ state: 'DIAGNOSTIC_FIRST_BOOT_SOURCE_PASS', diff --git a/CHANGELOG.md b/CHANGELOG.md index 8ddd3fb..0ed3fa7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -17,11 +17,21 @@ This section is the running test/version ledger for the current TFTMAC DEV optim - Accepted AVD baseline SHA-256: `b8cccc257dcc114ae5e6d24149514b7149b7e60a580fb74f79ca343823c28125`. - Test protocol: one hypothesis at a time; record `WIN`, `NO WIN`, or `INCONCLUSIVE`; keep proven wins; restore baseline after failures; do not create extra infrastructure merely to explain a loser. - Standing versioning policy: preserve the frozen LKG separately and never rewrite it to match the current experiment winner. Improvements advance the DEV working configuration version-by-version; each verified improvement becomes the next working winner, each failed or inconclusive test remains logged with its reason, and testing continues forward from the latest proven winner. Historical controls remain available for comparison and rollback. -- Promotion rule: **VERIFIED NET IMPROVEMENT = integrate and keep; NOT VERIFIED / INCONCLUSIVE / REGRESSION = log, do not integrate, restore the latest verified winner, and move to the next test.** A candidate does not need every metric to improve. Small regressions in secondary metrics are acceptable when the overall gameplay/system result is noticeably better and there is no correctness, stability, compatibility, or severe-tail regression that outweighs the gain. +- Promotion rule: **VERIFIED REPEATABLE NET IMPROVEMENT = integrate and keep; NOT VERIFIED / INCONCLUSIVE / REGRESSION = log, do not integrate, restore the latest verified winner, and move to the next test.** There is no fixed positive-gain percentage floor. A candidate does not need to reach 60 FPS or improve every metric. Small regressions in secondary metrics are acceptable when the overall gameplay/system result is better and there is no correctness, stability, compatibility, or severe-tail regression that outweighs the gain. - Evaluation weighting: FPS remains a heavily weighted metric but is not the sole optimization target. Frame pacing, p95/p99/worst-frame latency, jank, missed-vsync behavior, CPU/RHI efficiency, memory behavior, allocation/churn, stalls, stability, input responsiveness, and other measured system costs may establish a verified net win even when mean FPS is flat or slightly lower. -- Compounding rule: every new candidate is tested on top of the latest verified working winner, not repeatedly against the untouched LKG. The goal is cumulative efficiency: retain minimal proven wins so later changes can compound with them. A multiplicative/synergistic benefit is a hypothesis to verify, never an assumption; each combined working version must still pass its own bounded acceptance comparison before promotion. +- Compounding rule: every new candidate is tested on top of the latest verified working winner, not repeatedly against the untouched LKG. Continuous useful 60 FPS is the ultimate target, while each experiment asks only whether the tested delta produces a repeatable net improvement. Retain small proven wins so later changes can compound with them. A multiplicative/synergistic benefit is a hypothesis to verify, never an assumption; each combined working version must still pass its own bounded acceptance comparison before promotion. - Mandatory synchronized record-book rule: after every completed DEV optimization test, update this ledger before starting the next candidate. Also update `project.md` whenever current winner/configuration/test state changes, and update `facts.md` whenever the result changes a current hard fact, mandatory rule, runtime/client identity, protected boundary, or authoritative configuration. The record books must describe the current state before another test begins. +#### Optimization goal doctrine — 2026-09-11 + +**Outcome:** `CUMULATIVE SMALL-GAINS STRATEGY ADOPTED` + +- Continuous useful 60 FPS remains the ultimate target, but no individual candidate must independently reach 60 FPS. +- The prior executable +5% weighted-FPS floor conflicted with the cumulative strategy because it could discard legitimate small gains before they had a chance to compound. +- Small valid improvements are now eligible for `PROMISING`, must pass confirmation, and become the next `DEV-B8-WIN-##` only when the net improvement is repeatable and free of material correctness/stability/compatibility/tail regressions. +- Neutral, invalid, or losing candidates are logged and rolled back; the pass then moves to the next research-exposed setting rather than expanding into speculative infrastructure. +- `HOME_RUN` remains a useful label for unusually large/broad wins; it is not the only kind of improvement worth retaining. + #### Baseline — CONTROL_GREEN **Outcome:** `PASS / HISTORICAL CONTROL` diff --git a/Tests/TFTMACTests/CombatBenchmarkAnalysisTests.swift b/Tests/TFTMACTests/CombatBenchmarkAnalysisTests.swift index ba52fa2..8341285 100644 --- a/Tests/TFTMACTests/CombatBenchmarkAnalysisTests.swift +++ b/Tests/TFTMACTests/CombatBenchmarkAnalysisTests.swift @@ -55,11 +55,27 @@ final class CombatBenchmarkAnalysisTests: XCTestCase { XCTAssertEqual(analysis.decision, .reject) } - func testCompletedScreeningBelowFivePercentWeightedFPSImprovementRejects() { - let baseline = metrics() - let analysis = CombatBenchmarkAnalysis(baseline: baseline, candidate: baseline) + func testSmallSubFivePercentWeightedFPSImprovementIsPromising() { + let analysis = CombatBenchmarkAnalysis( + baseline: metrics(), + candidate: metrics(weightedFPS: 51) + ) - XCTAssertEqual(analysis.decision, .reject) + XCTAssertEqual(analysis.decision, .promising) + XCTAssertEqual(analysis.deltas.weightedFPSPercent, 2, accuracy: 0.001) + } + + func testTailOnlyIncrementalGainCanBePromisingWithSmallFPSTradeoff() { + let analysis = CombatBenchmarkAnalysis( + baseline: metrics(), + candidate: metrics( + weightedFPS: 49.5, + p95IntervalMilliseconds: 31, + p99IntervalMilliseconds: 49 + ) + ) + + XCTAssertEqual(analysis.decision, .promising) } func testInvalidCandidateIsInconclusiveAndReportsEveryFailure() { @@ -118,13 +134,24 @@ final class CombatBenchmarkAnalysisTests: XCTestCase { XCTAssertEqual(analysis.decision, .reject) } - func testThresholdGapIsInconclusive() { + func testNeutralComparisonIsInconclusive() { + let baseline = metrics() + let analysis = CombatBenchmarkAnalysis(baseline: baseline, candidate: baseline) + + XCTAssertEqual(analysis.decision, .inconclusive) + } + + func testMaterialWeightedFPSRegressionRejectsEvenWithTailImprovement() { let analysis = CombatBenchmarkAnalysis( baseline: metrics(), - candidate: metrics(weightedFPS: 53, onePercentLowFPS: 16) + candidate: metrics( + weightedFPS: 47, + p95IntervalMilliseconds: 30, + p99IntervalMilliseconds: 48 + ) ) - XCTAssertEqual(analysis.decision, .inconclusive) + XCTAssertEqual(analysis.decision, .reject) } func testOnePercentLowUsesMeanOfSlowestOnePercent() { diff --git a/benchmark.md b/benchmark.md index c772aa1..5ada22d 100644 --- a/benchmark.md +++ b/benchmark.md @@ -1,7 +1,7 @@ # TFTMAC Benchmark and Analysis Contract -**Authority date:** 2026-09-10 America/Chicago -**Formula version:** `tftmac-benchmark-v3` +**Authority date:** 2026-09-11 America/Chicago +**Formula version:** `tftmac-benchmark-v4` **Current DEV runtime:** TFTMAC DEV 2.3.0 build 8 / StockShadow on the M4 Mac mini; current verified optimization baseline `DEV-B8-WIN-01`, 1920×1080 / 60 Hz / **8 vCPU / 6144 MiB**, OpenGL ES through ANGLE. Historical Build 8 full-run captures remain benchmark evidence for their recorded configurations. **Purpose:** give a developer or AI agent one exact, reproducible process for turning TFTMAC session data into findings, comparisons, decisions, and explicit unknowns. @@ -30,7 +30,7 @@ Full runs remain valuable because they include the complete performance envelope A lobby, a reported `SRC 60`, an `OUT 60`, a successful launch, or an emulator process is never a gameplay benchmark. -The current optimization decision is **net-system efficiency**, with FPS heavily weighted but not exclusive. Direct graphics cadence, 1% low, p95/p99/worst-frame latency, jank, missed-vsync/severe behavior, source freshness and owned-boundary evidence remain primary. CPU/RHI efficiency, memory behavior, allocation/churn, stalls, responsiveness, thermal/power and audio may also be optimization or veto dimensions when directly measured. A small regression in one secondary metric does not automatically reject a candidate when the overall measured result is noticeably better and no correctness/stability/compatibility or severe-tail regression outweighs the gain. +The current optimization decision is **net-system efficiency**, with FPS heavily weighted but not exclusive. Direct graphics cadence, 1% low, p95/p99/worst-frame latency, jank, missed-vsync/severe behavior, source freshness and owned-boundary evidence remain primary. CPU/RHI efficiency, memory behavior, allocation/churn, stalls, responsiveness, thermal/power and audio may also be optimization or veto dimensions when directly measured. A small regression in one secondary metric does not automatically reject a candidate when the overall measured result is better and no correctness/stability/compatibility or severe-tail regression outweighs the gain. **There is no fixed positive-gain percentage floor:** a repeatable 1–4% improvement or a small pacing/tail win can be worth keeping because the active strategy is to compound verified gains toward continuous 60 FPS. ## 2. Evidence and claim discipline @@ -109,7 +109,7 @@ Before calculation, create an analysis manifest: ```json { "analysis_schema": "tftmac.benchmark-report.v1", - "formula_version": "tftmac-benchmark-v2", + "formula_version": "tftmac-benchmark-v4", "evidence_mode": "FULL_RUN|BOUNDED_AB|DIAGNOSTIC_ONLY|INVALID", "session_id": "", "session_database": "", @@ -458,10 +458,10 @@ compatibility fields rather than silently assume them. | Decision | Exact implemented rule | | --- | --- | -| `INCONCLUSIVE` | either run invalid; baseline correctness false; or valid values land between all resolving rules | -| `REJECT` | candidate correctness false; p95 or p99 interval is at least 10% worse; or weighted FPS gain is below 5% | -| `HOME_RUN` | after the 5% FPS guard: 1%-low gain at least 20%; jank and severe rates each fall at least 30% relative; and weighted FPS rises at least 10% **or** p95 interval falls at least 15% | -| `PROMISING` | weighted FPS rises at least 5%; 1%-low rises at least 10%; p95 and p99 intervals do not worsen | +| `INCONCLUSIVE` | either run invalid; baseline correctness false; or valid evidence has no directional improvement and no decisive material regression | +| `REJECT` | candidate correctness false; weighted FPS falls at least 5%; 1%-low falls at least 10%; or p95/p99 interval is at least 10% worse | +| `HOME_RUN` | 1%-low gain at least 20%; jank and severe rates each fall at least 30% relative; and weighted FPS rises at least 10% **or** p95 interval falls at least 15%, without a material veto | +| `PROMISING` | at least one directly measured improvement signal exists in weighted FPS, 1%-low FPS, both p95/p99 tails, or smoothness rates, and none of the material-regression/correctness vetoes fire; there is no positive +5% floor | Current code calculates/persists `observer_overhead_invalid` when trace-active versus trace-inactive FPS or p95 differs by more than 5% with at least ten @@ -473,8 +473,7 @@ code_decision: causal_interpretation: INVALID_OBSERVER_OVERHEAD | ELIGIBLE ``` -A `HOME_RUN` or `PROMISING` bounded result requires one cold confirmation and -one complete automatic full run before promotion to normal play. +A `HOME_RUN` or `PROMISING` bounded result requires one cold confirmation before it becomes the next DEV working winner. Promotion from DEV into normal-play/release authority still requires the stronger complete automatic full-run acceptance. This separation lets small verified gains compound without pretending the 60-FPS product target has already been met. These relative decisions select whether a change is worth retaining; they do not redefine the product goal. Every report must separately emit: diff --git a/dev.md b/dev.md index 29b2837..3c464d6 100644 --- a/dev.md +++ b/dev.md @@ -1,13 +1,13 @@ # TFTMAC Developer Record -> **CURRENT DEV ENGINEERING AUTHORITY — 2026-09-10.** Read `facts.md` first, then `project.md`, then `CHANGELOG.md` before using this engineering map. Older material below remains historical evidence only when it conflicts with those current records. +> **CURRENT DEV ENGINEERING AUTHORITY — 2026-09-11.** Read `facts.md` first, then `project.md`, then `CHANGELOG.md` before using this engineering map. Older material below remains historical evidence only when it conflicts with those current records. **Development baseline:** current verified working configuration `DEV-B8-WIN-01` on TFTMAC DEV 2.3.0 build 8 / StockShadow. **Effective DEV runtime:** 1920×1080 / 320 dpi / 60 Hz, **8 vCPU**, **6144 MiB (6 GiB)**, host GPU/CoreAudio, OpenGL ES through ANGLE, ADB/console/controller `5041/5586/8556`. **Protected Control:** separate immutable normal-play/LKG reference; historical 6-vCPU/5120-MiB Control values are not the DEV baseline. **Active campaign model:** results-first, one hypothesis at a time, integrating only verified net improvements and testing the next factor on the latest verified winner. **Current winner:** `DEV-B8-WIN-01` removes `syncMonolithicPipelinesToBlobCache` while retaining multifile cache and disabled `preferSubmitAtFBOBoundary`. -**Primary objective:** improve useful gameplay performance and total system efficiency toward continuous 60 FPS without correctness, login, audio, memory, launch, or cleanup regression. +**Primary objective:** accumulate repeatable small and large net improvements in useful gameplay performance/system efficiency, one exposed setting at a time, until the combined DEV line can hold continuous useful 60 FPS without correctness, login, audio, memory, launch, or cleanup regression. This is the engineering working file. It contains code ownership, measurement contracts, confirmed and rejected experiments, active hypotheses, and the next @@ -39,8 +39,8 @@ Rules: 9. Retain negative results so they are not recycled as “new” ideas. 10. A launch receipt proves setup, not performance. 11. Before finalizing any plan or change, re-read `facts.md` and `project.md`; if newer evidence conflicts, validate it and reconcile those authority files before finalization. -12. Promotion is based on verified **net** improvement, not mean FPS alone; p95/p99/worst-frame latency, jank, missed-vsync, CPU/RHI efficiency, memory behavior, stalls, responsiveness and stability all count. -13. A verified win becomes the next `DEV-B8-WIN-##` baseline; an unverified/inconclusive/regressing candidate is logged and not integrated. +12. Promotion is based on verified **repeatable net** improvement, not mean FPS alone; p95/p99/worst-frame latency, jank, missed-vsync, CPU/RHI efficiency, memory behavior, stalls, responsiveness and stability all count. There is no fixed positive-gain floor: small clean gains are intentionally eligible so they can compound. +13. A verified win becomes the next `DEV-B8-WIN-##` baseline even when the gain is small; an unverified/inconclusive/regressing candidate is logged, rolled back, and not integrated. 14. Do not declare a selected managed change complete with accidental dirty Git state. 15. OvernightLab telemetry is retained across minor configuration drift; comparability/promotion eligibility is separate from whether the data is worth keeping. Core client/RHI mismatch is data-only and cannot promote the current DEV line. 16. OvernightLab normal control always starts from the latest verified DEV winner; historical LKG/global-sync and rejected candidates remain cataloged evidence, not automatic queue entries. @@ -347,18 +347,14 @@ role—not the ephemeral token prefix/suffix. | Decision | Rule | | --- | --- | -| HOME_RUN | after the weighted-FPS +5% guard: 1% low +20%, jank and severe each -30% relative, and either weighted FPS +10% or p95 interval -15% | -| PROMISING | weighted FPS +5%, 1% low +10%, and p95/p99 intervals no worse | -| REJECT | weighted FPS gain below 5%, p95/p99 interval +10% worse, or candidate correctness/usability failure | -| INCONCLUSIVE | invalid/mismatched workload, coverage, clock, observer, or threshold gap | - -Any HOME_RUN/PROMISING result needs a five-minute cold confirmation. Rollback is -select Control and restart. A failed active candidate records correctness -rejection and saves Control automatically. - -Relative decisions select the better implementation; they do not lower the -goal. Report `TARGET_NOT_MET` until a complete automatic run holds at least 60 -useful FPS throughout with no missed-vsync equivalents or severe stalls. +| HOME_RUN | standout broad win: 1% low +20%, jank and severe each -30% relative, and either weighted FPS +10% or p95 interval -15%, without a material veto | +| PROMISING | any valid directional gain in weighted FPS, 1% low, both p95/p99 tails, or smoothness rates when no material regression/correctness veto fires; no +5% positive floor | +| REJECT | correctness/usability failure or material regression: weighted FPS -5%, 1% low -10%, p95/p99 interval +10% worse, or an equivalent operational veto | +| INCONCLUSIVE | invalid/mismatched workload, coverage, clock, observer, or valid evidence with no directional signal | + +Any HOME_RUN/PROMISING result needs a five-minute cold confirmation before becoming the next DEV winner. Rollback is select the latest verified winner and restart. A failed active candidate records the rejection/failure and restores that winner automatically. + +Relative decisions select the better implementation; they do not lower the goal. Small verified wins are intentionally retained and compounded. Report `TARGET_NOT_MET` until a complete automatic run holds at least 60 useful FPS throughout with no missed-vsync equivalents or severe stalls. ## 7. Retained results diff --git a/facts.md b/facts.md index 6d507c1..5ff9c5c 100644 --- a/facts.md +++ b/facts.md @@ -1,9 +1,9 @@ # TFTMAC Facts -> **CURRENT DEV AUTHORITY — 2026-09-10 America/Chicago.** The DEV optimization/versioning records in `project.md` and `CHANGELOG.md` supersede older mutable DEV build, campaign, performance, and readiness claims below. Historical measurements retain their original dates and remain evidence for their recorded configuration only. +> **CURRENT DEV AUTHORITY — 2026-09-11 America/Chicago.** The DEV optimization/versioning records in `project.md` and `CHANGELOG.md` supersede older mutable DEV build, campaign, performance, and readiness claims below. Historical measurements retain their original dates and remain evidence for their recorded configuration only. -**Authority date:** 2026-09-10 America/Chicago -**Current DEV evidence through:** 2026-09-10 results-first verification pass +**Authority date:** 2026-09-11 America/Chicago +**Current DEV evidence through:** 2026-09-11 incremental-gains optimization doctrine; latest performance winner remains `DEV-B8-WIN-01` **Purpose:** preserve current facts and hard boundaries that future TFTMAC work must not casually reinterpret. ## Mandatory current-authority reading and record-book rule @@ -18,7 +18,7 @@ For the active results-first pass, also obey `.clara/plans/ff2f318b-245d-418a-b8 **LOCKED RECORD-BOOK POLICY:** every completed DEV optimization test must update the current record books in the same work cycle. A test result is not considered fully recorded until `CHANGELOG.md` contains the exact result and integration decision and `project.md` reflects any change to current winner/state/version progression. Update `facts.md` whenever the result changes a current fact, hard boundary, mandatory rule, runtime/client identity, or authoritative current configuration. Do not copy transient measurements into `facts.md` merely because a test ran; facts stays concise and authoritative, while `CHANGELOG.md` retains the detailed test history. -**LOCKED PROMOTION POLICY:** VERIFIED NET IMPROVEMENT -> integrate it, assign/promote the next working `DEV-B8-WIN-##` identity, update `project.md` and `CHANGELOG.md`, and use that winner as the baseline for the next test. NOT VERIFIED / INCONCLUSIVE / REGRESSION -> log the exact result and reason in `CHANGELOG.md`, do not integrate it, keep/restore the latest verified winner in `project.md`, and move to the next candidate. A small regression in one secondary metric may be accepted when the overall measured system/gameplay result is noticeably better and no correctness, stability, compatibility, or severe-tail regression outweighs the gain. +**LOCKED PROMOTION POLICY:** continuous useful 60 FPS is the cumulative destination, not a per-candidate minimum. There is **no fixed positive-gain percentage floor** for retaining an optimization. VERIFIED, REPEATABLE NET IMPROVEMENT -> integrate it, assign/promote the next working `DEV-B8-WIN-##` identity, update `project.md` and `CHANGELOG.md`, and use that winner as the baseline for the next test. NOT VERIFIED / INCONCLUSIVE / REGRESSION -> log the exact result and reason in `CHANGELOG.md`, do not integrate it, keep/restore the latest verified winner in `project.md`, and move to the next candidate. A small regression in one secondary metric may be accepted when the overall measured system/gameplay result is better and no correctness, stability, compatibility, or severe-tail regression outweighs the gain. Small verified gains are intentionally retained so they can compound toward the 60-FPS target. **LOCKED LKG SEPARATION:** the frozen LKG remains separate and immutable. The evolving DEV working winner is never allowed to silently rewrite the historical LKG/control evidence. @@ -466,20 +466,12 @@ Privacy facts: Decision rules: -- **HOME_RUN:** after the weighted-FPS +5% guard, 1% low +20% or more, jank and - severe rates each -30% or more relative, and either weighted FPS +10% or p95 - interval -15%. -- **PROMISING:** weighted FPS +5% or more and 1% low +10% or more, with no - correctness or tail regression. -- **REJECT:** gain below 5%, p95/p99 worsens at least 10%, or any boot, render, - input, audio, login, memory, cleanup, or usability regression. -- **INCONCLUSIVE:** invalid workload/coverage/synchronization, incompatible - configuration identity, or result between thresholds. -- A winning candidate still requires a five-minute cold confirmation before - normal-use promotion. -- **LOCKED:** `HOME_RUN`/`PROMISING` are relative candidate decisions, not proof - that the product target is met. The separate full-run status remains - `TARGET_NOT_MET` until useful-frame cadence holds at least 60 FPS throughout. +- **HOME_RUN:** a standout broad win: 1% low +20% or more, jank and severe rates each -30% or more relative, and either weighted FPS +10% or p95 interval -15%, without a material veto. +- **PROMISING:** any valid directional improvement in weighted FPS, 1% low, both p95/p99 tails, or smoothness/jank/missed-vsync behavior, provided no material regression or correctness/usability veto outweighs it. There is no +5% FPS floor. +- **REJECT:** correctness/usability failure or material regression, including weighted FPS about 5% worse, 1% low about 10% worse, p95/p99 at least 10% worse, or any boot, render, input, audio, login, memory, cleanup, or usability regression. +- **INCONCLUSIVE:** invalid workload/coverage/synchronization, incompatible configuration identity, or valid evidence with no directional improvement or decisive regression. +- A `PROMISING` incremental candidate requires the existing cold confirmation before becoming the next DEV working winner; a later full-run check may still veto it if compounding exposes a material regression. +- **LOCKED:** `HOME_RUN`/`PROMISING` are relative candidate decisions, not proof that the product target is met. The separate full-run status remains `TARGET_NOT_MET` until useful-frame cadence holds at least 60 FPS throughout. ## 11. Verified results and decisions diff --git a/project.md b/project.md index f80d290..f743eb4 100644 --- a/project.md +++ b/project.md @@ -9,7 +9,7 @@ **Current official client:** `com.riotgames.league.teamfighttactics` `18.1-5423749`, versionCode `8423749`. **Current test series:** `DEV-B8-2026-09-10-A`. **Current verified working winner:** **`DEV-B8-WIN-01`**. -**Project record current through:** 2026-09-11 post-PR-#9 authority rehydration; performance winner remains the 2026-09-10 results-first verification state. +**Project record current through:** 2026-09-11 incremental-gains doctrine and post-PR-#9 continuity; performance winner remains `DEV-B8-WIN-01`. ## 0. Living wiki contract @@ -49,10 +49,11 @@ Mandatory companion records: 1. Start every new candidate from the latest verified DEV winner, currently `DEV-B8-WIN-01`, not from the frozen LKG unless a matched historical control is specifically required. 2. Change/test one primary hypothesis at a time. Add only a directly-related minimal blocker adjustment when concrete evidence says the intended candidate cannot otherwise execute. 3. Evaluate net system/gameplay improvement, not FPS alone. FPS is heavily weighted, alongside frame pacing, p95/p99/worst-frame latency, jank, missed-vsync, CPU/RHI efficiency, memory behavior, allocation/churn, stalls, input responsiveness, correctness, stability, and compatibility. -4. **VERIFIED NET IMPROVEMENT:** integrate it, assign the next `DEV-B8-WIN-##`, update this wiki and `CHANGELOG.md`, then test the next candidate on top of the new winner. -5. **NOT VERIFIED / INCONCLUSIVE / REGRESSION:** record it in `CHANGELOG.md`, do not integrate it, retain/restore the latest verified winner here, and move forward. -6. Compounding/synergy is desirable but must be measured. A prior verified win remains integrated while the next factor is tested; the combined configuration must itself verify before promotion. -7. Never rewrite the frozen LKG to match the DEV winner. LKG is historical control/rollback; DEV is the evolving optimization line. +4. **60 FPS is the cumulative destination, not a per-candidate gate.** A candidate does not need to reach 60 FPS or clear an arbitrary +5% threshold to be useful. +5. **VERIFIED REPEATABLE NET IMPROVEMENT:** even when small, integrate it, assign the next `DEV-B8-WIN-##`, update this wiki and `CHANGELOG.md`, then test the next candidate on top of the new winner. +6. **NOT VERIFIED / INCONCLUSIVE / REGRESSION:** record it in `CHANGELOG.md`, do not integrate it, retain/restore the latest verified winner here, and move forward without building an explanation project around the loser. +7. Compounding/synergy is the active strategy and must be measured. Each prior verified win remains integrated while the next factor is tested; the combined configuration must verify before promotion. The experiment program asks whether many small clean gains can add up to continuous useful 60 FPS. +8. Never rewrite the frozen LKG to match the DEV winner. LKG is historical control/rollback; DEV is the evolving optimization line. ### Current completed-test state @@ -102,11 +103,7 @@ the product UI. The application must also be an engineering laboratory that captures the complete runtime behavior well enough to make and reject graphics- pipeline changes based on evidence. -The completion standard is not “the emulator process exists” and not “the lobby -shows 60 FPS.” The user must be able to play through the native Mac window, and -the logger must preserve every under-target period across the complete run. The -graphics target is at least 60 useful FPS throughout, not only during selected -scenes. +The completion standard is not “the emulator process exists” and not “the lobby shows 60 FPS.” The user must be able to play through the native Mac window, and the logger must preserve every under-target period across the complete run. The ultimate graphics target is at least 60 useful FPS throughout, not only during selected scenes. The optimization path is intentionally incremental: test one exposed/research-backed setting at a time, keep every repeatable net improvement even when small, stack the next experiment on that winner, and measure whether the accumulated gains close the remaining gap to continuous 60 FPS. ## 2. Current architecture diff --git a/scripts/verify-tftmac.command b/scripts/verify-tftmac.command index 1d277cc..376a588 100755 --- a/scripts/verify-tftmac.command +++ b/scripts/verify-tftmac.command @@ -280,7 +280,7 @@ while read -r expected_hash authority_path; do done < ssot/AUTHORITY_INPUTS.sha256 readonly TEST_FUNCTION_COUNT="$(rg -n '^[[:space:]]*func test' Tests/TFTMACTests --glob '*.swift' | wc -l | tr -d '[:space:]')" -[[ "$TEST_FUNCTION_COUNT" == "110" ]] || fail "native test inventory drifted: expected 110, found $TEST_FUNCTION_COUNT" +[[ "$TEST_FUNCTION_COUNT" == "112" ]] || fail "native test inventory drifted: expected 112, found $TEST_FUNCTION_COUNT" [[ "$(plutil -extract LSSupportsGameMode raw "$INFO")" == "true" ]] \ || fail "native app is not eligible for macOS Game Mode" [[ "$(shasum -a 256 tftmac/Assets/TFTMAC-Official-Icon.png | awk '{print $1}')" == "d6ba9ceb76c4b1e44e87f059f775a0ed629f9bea29b0dd73245853d7dca3a016" ]] \ diff --git a/ssot/AUTHORITY_INPUTS.sha256 b/ssot/AUTHORITY_INPUTS.sha256 index 0dfc24c..f03fbd3 100644 --- a/ssot/AUTHORITY_INPUTS.sha256 +++ b/ssot/AUTHORITY_INPUTS.sha256 @@ -1,10 +1,10 @@ # TFTMAC Build 8 documentation and machine-readable authority inputs # Regenerate only after reviewing the complete authority change set. -ee70562e850e82106ac3b07cb2e0ba00acbeef128f0c827edf003a3253aba654 facts.md -fec86184eb4faf346f9aa81a704c9de1efeaea7c450eb5d73392edd1373345ec project.md -53e1416892348c2cd1145344b54f661134b59e1483891c9a3127a0270ce48f56 dev.md -f5ae42d870a47565c52433f2d07d9db5b24dee3d914d7d896a160aa130800f5d benchmark.md +3ba1fd9c4a946a4bc89a9f4994af1a27cd3fe061c79a3f28669b35dde2586fec facts.md +8d8cf83fea6d756fd069a8ffe5e9b4649c31503b7e73bea75b4fae39191a7e79 project.md +a91b5af9a15da1a519d336d5b78c6f7fb5d59de9970f99afb448c3a26d8ed54c dev.md +87eb1d0585e7c244a9f0a0eca74078f005ee2c8e7c6ab1ea737fc6c6be897720 benchmark.md 143af322d05e3ac8ccae92baf279ed8e6e8f62ce5bd2f757e045868e1f20c2c8 TFTMAC_GPU_RUNTIME_SSOT.md 3607fef47c7ca0b7ebd3ced416e0b3eb9e7dd37143624b97da2670ee283c7169 ssot/runtime-authority.json a4bdab6304b32c135fe00234564ef408e6049135945f025005c8286f0439e39a ssot/TFTMAC_ENGINEERING_MAP.sql diff --git a/ssot/STACK.lock.yaml b/ssot/STACK.lock.yaml index afc10bb..980de0e 100644 --- a/ssot/STACK.lock.yaml +++ b/ssot/STACK.lock.yaml @@ -10,7 +10,7 @@ authority: documentation_authority: "facts.md > project.md > CHANGELOG.md > benchmark.md + dev.md + supporting references" authority_inputs: "ssot/AUTHORITY_INPUTS.sha256" runtime_authority_sha256: "3607fef47c7ca0b7ebd3ced416e0b3eb9e7dd37143624b97da2670ee283c7169" - authority_inputs_sha256: "1760a9bd96fac11c8dfb9927808de69eb3969ad69b438a9b3e9b0a2b6dafe634" + authority_inputs_sha256: "a32d76a14a956aa0523b5c65f57654beab8a5b3107ac8eea372d034913948a63" zengate_version: "2.3" zengate_score: 95.8 zengate_result: "PASS" @@ -258,7 +258,7 @@ current_gameplay_capture: advanced_source_causal_logger: "partial_source_schema_and_cpp_abi_not_live_accepted" development_verification: - native_test_inventory: 110 + native_test_inventory: 112 causal_cpp_abi_test: PASS causal_event_bytes: 96 causal_ring_capacity: 256 diff --git a/tftmac/Runtime/CombatBenchmarkAnalysis.swift b/tftmac/Runtime/CombatBenchmarkAnalysis.swift index f39ba50..25a5c8a 100644 --- a/tftmac/Runtime/CombatBenchmarkAnalysis.swift +++ b/tftmac/Runtime/CombatBenchmarkAnalysis.swift @@ -187,10 +187,9 @@ struct CombatBenchmarkAnalysis: Equatable, Sendable { guard baselineValidity.isValid, candidateValidity.isValid else { return .inconclusive } guard baseline.correctnessPassed else { return .inconclusive } guard candidate.correctnessPassed else { return .reject } - if deltas.p95IntervalPercent >= 10 || deltas.p99IntervalPercent >= 10 { return .reject } - if deltas.weightedFPSPercent < 5 { return .reject } + if hasMaterialRegression(deltas) { return .reject } if isHomeRun(baseline: baseline, candidate: candidate, deltas: deltas) { return .homeRun } - if isPromising(deltas) { return .promising } + if hasDirectionalImprovement(deltas) { return .promising } return .inconclusive } @@ -205,11 +204,20 @@ struct CombatBenchmarkAnalysis: Equatable, Sendable { && (deltas.weightedFPSPercent >= 10 || deltas.p95IntervalPercent <= -15) } - private static func isPromising(_ deltas: CombatBenchmarkDeltas) -> Bool { - deltas.weightedFPSPercent >= 5 - && deltas.onePercentLowFPSPercent >= 10 - && deltas.p95IntervalPercent <= 0 - && deltas.p99IntervalPercent <= 0 + private static func hasMaterialRegression(_ deltas: CombatBenchmarkDeltas) -> Bool { + deltas.weightedFPSPercent <= -5 + || deltas.onePercentLowFPSPercent <= -10 + || deltas.p95IntervalPercent >= 10 + || deltas.p99IntervalPercent >= 10 + } + + private static func hasDirectionalImprovement(_ deltas: CombatBenchmarkDeltas) -> Bool { + deltas.weightedFPSPercent > 0 + || deltas.onePercentLowFPSPercent > 0 + || (deltas.p95IntervalPercent < 0 && deltas.p99IntervalPercent < 0) + || deltas.jankRatePercentagePoints < 0 + || deltas.severeRatePercentagePoints < 0 + || deltas.missedVsyncRatePercentagePoints < 0 } private static func relativeReductionIsAtLeast30Percent(from baseline: Double, to candidate: Double) -> Bool {