diff --git a/.akr/akr.lock b/.akr/akr.lock index 9f386709..a6dc9ea2 100644 --- a/.akr/akr.lock +++ b/.akr/akr.lock @@ -5,7 +5,7 @@ build { tool "akr 0.3.3" grammar "0.1" vocabulary "0.2" - source_graph "sha256:29947c3fadfc107253854896393926aa0698b43a301121176a032e4f7a7260f4" + source_graph "sha256:2e53454b4abc5a2e7803995f3fb00745b536d02f33e33c9607223cdf0dd3935b" } source ".akr/project.akr" { @@ -24,13 +24,13 @@ source ".akr/records/jpegxl-rs/constraints.akr" { } source ".akr/records/jpegxl-rs/decisions.akr" { - hash "sha256:a0a027029a3e5cca15f8b8da6c57bcd78beca92aa6be8e16329a84c652149570" - records 22 + hash "sha256:a7c07810b408069dab2854fdacf83ddb5d502da1a39974cef192ca016a15eb88" + records 23 } source ".akr/records/jpegxl-rs/evidence.akr" { - hash "sha256:05e3371b89c14b5329e59ee4650f9251e6666ee682beffbf3588af23a2562f11" - records 538 + hash "sha256:d04e0b11af4cc93647f60cb756ed981f01a093d4ab3969445c4eba15fd4f7aaf" + records 566 } source ".akr/records/jpegxl-rs/milestones.akr" { @@ -39,13 +39,13 @@ source ".akr/records/jpegxl-rs/milestones.akr" { } source ".akr/records/jpegxl-rs/observations.akr" { - hash "sha256:a4ac22959d397d2d3ccc7794c19793623a94b360565f0003b43a4e5722f3d121" - records 95 + hash "sha256:45603a70e136457f806013a2f0575d352fb4098e7ec538e3fdf6faad32bbe005" + records 102 } source ".akr/records/jpegxl-rs/papercuts.akr" { - hash "sha256:d10a9e7e42b1b89437934e9a3a4360d484c38094d1a44c91243ce8036f88b24d" - records 41 + hash "sha256:5ef7b252be9748a1505f9d11324b9fd25a192c507f59db7e4ce0011ea745cf5d" + records 46 } source ".akr/records/jpegxl-rs/policies.akr" { @@ -69,8 +69,8 @@ source ".akr/records/jpegxl-rs/tracks.akr" { } source ".akr/records/jpegxl-rs/work.akr" { - hash "sha256:fc2591b8fd0431514a35c1cfb4ab1ba44b639dbefbf7fe455ab85001d621362b" - records 177 + hash "sha256:52e69b5b86d843ee9caaa7a2fa2e7915a828b1c8a206aa8883b5ddb07f69cd6e" + records 188 } resolution @jpegxl-rs.decision.encoder-architecture-phases/1 { @@ -1423,6 +1423,36 @@ resolution @jpegxl-rs.observation.phase5-quality-policy-screen-2026-08-11/11 { hash "sha256:94cf4ec999e131054d664d2c758933fa05d0e47d880baa25751f51ac108ded1e" } +resolution @jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/1 { + slot derived_from + to @jpegxl-rs.work.pqc-one-shot-controller/3 + hash "sha256:b08e9ecd74467c21be5f661c5c94417d63576222c3713c1c9a9cacfd07a1773f" +} + +resolution @jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/1 { + slot verified_by + to @jpegxl-rs.evidence.pqc-one-shot-falsification-screen-2026-08-24/1 + hash "sha256:9ec7485c35348921be8ac2b3b07ae600325b4d9a13138ff0e6ea28abb6551730" +} + +resolution @jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/2 { + slot derived_from + to @jpegxl-rs.work.pqc-one-shot-controller/3 + hash "sha256:b08e9ecd74467c21be5f661c5c94417d63576222c3713c1c9a9cacfd07a1773f" +} + +resolution @jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/2 { + slot verified_by + to @jpegxl-rs.evidence.pqc-one-shot-falsification-screen-2026-08-24/1 + hash "sha256:9ec7485c35348921be8ac2b3b07ae600325b4d9a13138ff0e6ea28abb6551730" +} + +resolution @jpegxl-rs.observation.quality-corpus-critical-class-families-2026-08-25/1 { + slot derived_from + to @jpegxl-rs.work.pqc-one-shot-controller/3 + hash "sha256:b08e9ecd74467c21be5f661c5c94417d63576222c3713c1c9a9cacfd07a1773f" +} + resolution @jpegxl-rs.observation.quantizer-normalised-distortion-is-the-lever-2026-08-12/1 { slot derived_from to @jpegxl-rs.observation.distortion-currency-misprices-frequency-2026-08-12/1 @@ -3667,6 +3697,78 @@ resolution @jpegxl-rs.work.opt-v2-rate-work-sharing/2 { hash "sha256:f7733beff3d25a93a7ba717c5686463ca7c969d91eae48dbaacfeefae835d73d" } +resolution @jpegxl-rs.work.pipeline-metadata-seam/1 { + slot implements + to @jpegxl-rs.requirement.general-use-integration-surface/2 + hash "sha256:540f9f8226b27b8fc64b62502f583996d474e374cc34a10e79b6b77706b3b17f" +} + +resolution @jpegxl-rs.work.pqc-one-shot-controller/1 { + slot depends_on + to @jpegxl-rs.work.pqc-one-shot-pr0-hard-floor/1 + hash "sha256:b835fe82f41cf362b357e9b22d4f9cc5b6ed0c8ed7a256da9c4e51d18576c366" +} + +resolution @jpegxl-rs.work.pqc-one-shot-controller/2 { + slot depends_on + to @jpegxl-rs.work.pqc-one-shot-pr0-hard-floor/1 + hash "sha256:b835fe82f41cf362b357e9b22d4f9cc5b6ed0c8ed7a256da9c4e51d18576c366" +} + +resolution @jpegxl-rs.work.pqc-one-shot-controller/3 { + slot depends_on + to @jpegxl-rs.work.pqc-one-shot-pr0-hard-floor/1 + hash "sha256:b835fe82f41cf362b357e9b22d4f9cc5b6ed0c8ed7a256da9c4e51d18576c366" +} + +resolution @jpegxl-rs.work.pqc-one-shot-pr0-hard-floor/1 { + slot implements + to @jpegxl-rs.decision.quality-miss-fallback-semantics/1 + hash "sha256:52c2c39f382fd60234500876d6de823679e2b836bff13d4a5b2ec3a1c67bec12" +} + +resolution @jpegxl-rs.work.pqc-usable-efforts-cost/6 { + slot supported_by + to @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1 + hash "sha256:2b6173b98676727bede354bb3488928d0ec58c1efd7dcbbf4ef927ea0ec4c51f" +} + +resolution @jpegxl-rs.work.pqc-usable-efforts-cost/7 { + slot supported_by + to @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1 + hash "sha256:2b6173b98676727bede354bb3488928d0ec58c1efd7dcbbf4ef927ea0ec4c51f" +} + +resolution @jpegxl-rs.work.pqc-usable-efforts-cost/7 { + slot supported_by + to @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25/1 + hash "sha256:7cc5c06e34773ef0692dacc6763c8e677730dd1172aede7d96d744eceec361ec" +} + +resolution @jpegxl-rs.work.pqc-usable-efforts-cost/8 { + slot supported_by + to @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1 + hash "sha256:2b6173b98676727bede354bb3488928d0ec58c1efd7dcbbf4ef927ea0ec4c51f" +} + +resolution @jpegxl-rs.work.pqc-usable-efforts-cost/8 { + slot supported_by + to @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25/1 + hash "sha256:7cc5c06e34773ef0692dacc6763c8e677730dd1172aede7d96d744eceec361ec" +} + +resolution @jpegxl-rs.work.pqc-usable-efforts-cost/9 { + slot supported_by + to @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1 + hash "sha256:2b6173b98676727bede354bb3488928d0ec58c1efd7dcbbf4ef927ea0ec4c51f" +} + +resolution @jpegxl-rs.work.pqc-usable-efforts-cost/9 { + slot supported_by + to @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25/1 + hash "sha256:7cc5c06e34773ef0692dacc6763c8e677730dd1172aede7d96d744eceec361ec" +} + resolution @jpegxl-rs.work.quality-q0-harness-attribution/1 { slot part_of to @jpegxl-rs.track.encoder-optimization/1 @@ -3775,6 +3877,16 @@ seal @jpegxl-rs.decision.ssimulacra2-is-the-primary-promotion-metric/1 { hash "sha256:9aa847d339b8d732eb49a5f6bbd48a6c4e41819fd3824400c8c4681d7b4f8549" } +seal @jpegxl-rs.evidence.avx2-pooling-bit-identity-2026-08-26/1 { + state verified + hash "sha256:3cdec63fc9c4af853f3e0245c9ad125c5ba23ce1b0fdf612aca20fa4f0d8d42b" +} + +seal @jpegxl-rs.evidence.bench-vs-libjxl-reproducible-2026-08-25/1 { + state verified + hash "sha256:a24dfed6aab83df6837b4ebe83c0a2bf6aa28a646df346c2b6f84175fe267b3b" +} + seal @jpegxl-rs.evidence.cli-netpbm-subtypes-2026-08-21/1 { state verified hash "sha256:b2c54a89b5174895fb5955a8fedbec20f197f9fb970db914dffd48f3ef55525f" @@ -4015,6 +4127,11 @@ seal @jpegxl-rs.evidence.history-backfilled/1 { hash "sha256:4aa4bba4295cce3e93c48dff164c981a82ecf1ad1206355da100dfc2d81ed0f1" } +seal @jpegxl-rs.evidence.jpeg-phase-a-roundtrip-2026-08-25/1 { + state verified + hash "sha256:758a044c3d01f0c235ff65a06fdb5928e83d2388a5be7be9d95ebacdbbf09d0f" +} + seal @jpegxl-rs.evidence.m1-plan-ir/1 { state verified hash "sha256:c89bc7184d9b204761769600cac70406bef53a0cd84e263ba7e450e393ed222c" @@ -4065,6 +4182,11 @@ seal @jpegxl-rs.evidence.markdown-cutover/2 { hash "sha256:f6d62a57f797272bde901c7362291f30e581372263914c099cbbea2ce3648610" } +seal @jpegxl-rs.evidence.mit-only-licensing-audit-2026-08-25/1 { + state verified + hash "sha256:657d25081fa678e05ee0a58ae07779836eb24161441877d7b8078ec0f6ff03cc" +} + seal @jpegxl-rs.evidence.modular-effort-ramp-landed-2026-08-10/1 { state verified hash "sha256:026ded0ca69253f0932ef49abcb56de636c98dd28d9a0021f6a2adb002348524" @@ -5855,6 +5977,31 @@ seal @jpegxl-rs.evidence.phase9-quant-scheduling-speed-2026-08-15/1 { hash "sha256:cb264f0ebd981e140e9b4a3dac7845d50ea72489e8117e6198a957e436099375" } +seal @jpegxl-rs.evidence.pipeline-metadata-seam-2026-08-26/1 { + state verified + hash "sha256:9ef1d7916672b5d8b3592497a750aabda8126af95eeb51d7b4c6fd506d114edf" +} + +seal @jpegxl-rs.evidence.pqc-fused-render-memory-12mp-2026-08-25/1 { + state verified + hash "sha256:53234a275fac57eb5ceb155e3b7a7aa1ffb3a402518f3c84fc2862f0e9bd5c24" +} + +seal @jpegxl-rs.evidence.pqc-fused-render-production-identity-2026-08-25/1 { + state verified + hash "sha256:ece4d88bb5c22dc47060ff753fb9f765df8872806e563ee05d4b22fdf3fa0ef7" +} + +seal @jpegxl-rs.evidence.pqc-fused-render-wall-anchors-2026-08-25/1 { + state verified + hash "sha256:43c93edb9ceaed9c2d1494a3bca171017cbdc3a03deb70e5638cefedbbc91655" +} + +seal @jpegxl-rs.evidence.pqc-fused-render-workspace-gates-2026-08-25/1 { + state verified + hash "sha256:996ba0b5fb50020d07a6631f973453e5a18ae135dfd6ded1e5257426e1506b99" +} + seal @jpegxl-rs.evidence.pqc-low-memory-12mp-2026-08-23/1 { state verified hash "sha256:53e8a0eba08edfa89e096ddc9606f88132a7905aa4cc96c8a0c224c3b1359ed6" @@ -5880,11 +6027,56 @@ seal @jpegxl-rs.evidence.pqc-low-memory-xlarge-identity-2026-08-23/1 { hash "sha256:79a3f38d1ea158a2cf0d801e7183885188f146d970d654988c55de059ca11d39" } +seal @jpegxl-rs.evidence.pqc-memory-12mp-2026-08-25/1 { + state verified + hash "sha256:214d5a4410bf29ecec7b1250fcf741e3d23891c3dbaeb124c0000ed36eb5c40c" +} + +seal @jpegxl-rs.evidence.pqc-metric-cuts-memory-12mp-2026-08-25/1 { + state verified + hash "sha256:16e6e15888f020722bce4e4a0802def17b4ae1c69f3ed76f1076bf37ea1f72b6" +} + +seal @jpegxl-rs.evidence.pqc-metric-cuts-production-identity-2026-08-25/1 { + state verified + hash "sha256:c67b712f513538cacb4844962869a153bd83b751141860e725d26acee4157d30" +} + +seal @jpegxl-rs.evidence.pqc-metric-cuts-wall-anchors-2026-08-25/1 { + state verified + hash "sha256:5a6d21fd31d84490f75b0c46fba08abf06704594ebbebf82d627b65dc3c33f46" +} + +seal @jpegxl-rs.evidence.pqc-metric-cuts-workspace-gates-2026-08-25/1 { + state verified + hash "sha256:166facf3550075dd7f9f882fa6d057193a81efc2cb3e8cb488e76b32edab71cb" +} + +seal @jpegxl-rs.evidence.pqc-one-shot-falsification-screen-2026-08-24/1 { + state verified + hash "sha256:9ec7485c35348921be8ac2b3b07ae600325b4d9a13138ff0e6ea28abb6551730" +} + +seal @jpegxl-rs.evidence.pqc-one-shot-pr1-5-gates-2026-08-24/1 { + state verified + hash "sha256:b419d8755a391df8409ca076112e2aa557841ad9dc9fa7666b7949bffc3614db" +} + +seal @jpegxl-rs.evidence.pqc-one-shot-promotion-ab-2026-08-25/1 { + state verified + hash "sha256:a6abe403b04e1ff0e28d938ac91d68e24042b6211256ca2706f68219484603dd" +} + seal @jpegxl-rs.evidence.pqc-pr0-corpus-manifest-2026-08-22/1 { state verified hash "sha256:7e2025d7b999667cc087b45cd550f86dbbc68a07de58532eada59a38898e857f" } +seal @jpegxl-rs.evidence.pqc-pr0-hard-floor-tests-2026-08-24/1 { + state verified + hash "sha256:05ccad4755a3b8fae64850e88f830a53c9424175ddec1200d9bc3a646e788bdb" +} + seal @jpegxl-rs.evidence.pqc-pr0-predictor-table-2026-08-22/1 { state verified hash "sha256:240c39820db7a47f61f4ad8c3ab109dd51cb48fc7ed6bee2de98c6c704d22501" @@ -5895,6 +6087,11 @@ seal @jpegxl-rs.evidence.pqc-pr0-sources-registered-2026-08-22/1 { hash "sha256:a225dddef75f255ed5115cfb36782388278c7e3897f40a3cacb9e30eb21fc143" } +seal @jpegxl-rs.evidence.pqc-pr0-workspace-gates-2026-08-24/1 { + state verified + hash "sha256:a1d17ff10b705658a94f18f7d468b629cd3a48cc35bd71687e96c36eabb7c602" +} + seal @jpegxl-rs.evidence.pqc-pr1-api-semantics-2026-08-22/1 { state verified hash "sha256:454da8cf4e2f11f3a2347547973189179866e3fc7c13501f476b4d44fea4bcef" @@ -6020,11 +6217,36 @@ seal @jpegxl-rs.evidence.pqc-pr7-balanced-single-shot-2026-08-22/1 { hash "sha256:91eddafebcc680bde1d197530f6aeb86bb3ef9d24eb4cfe8623e539dd493b622" } +seal @jpegxl-rs.evidence.pqc-pr7-closure-workspace-gates-2026-08-24/1 { + state verified + hash "sha256:41c0eb5952e24d5bb2a72a6eaca2f9cd2d5019d4044112671839901bc64fdb83" +} + seal @jpegxl-rs.evidence.pqc-pr7-dev-split-2026-08-22/1 { state verified hash "sha256:98914411a1b245d74cc49faab93ae8a79cc0594122b842bb78eb39490933d35d" } +seal @jpegxl-rs.evidence.pqc-pr7-holdout-complete-2026-08-24/1 { + state verified + hash "sha256:0d86d7c3470b6cd109ac2200256479601f0a5dcfe2c394d9816dea71153884d8" +} + +seal @jpegxl-rs.evidence.pqc-pr7-policy-gating-test-2026-08-24/1 { + state verified + hash "sha256:24c29e34a49849206ca8f873c4abc657c945875d2d8e380d3c57e7482d3d0bf2" +} + +seal @jpegxl-rs.evidence.pqc-pr7-promotion-disposition-2026-08-24/1 { + state verified + hash "sha256:607f815f2c55ee821124b57e89128b733e6826f3f838b7d0a000d6adb184130a" +} + +seal @jpegxl-rs.evidence.pqc-pr7-quality-promotion-full-2026-08-24/1 { + state verified + hash "sha256:40063888bc1e845191c0dff5e77c70d344adeeaaeebf361758909677e98b35e9" +} + seal @jpegxl-rs.evidence.pqc-pr7-quality-promotion-rejected-2026-08-22/1 { state verified hash "sha256:4c42aeb2d62fb81496e253780d2f4c586d7274abf1a1d6a6e04504050c03cae5" @@ -6035,6 +6257,11 @@ seal @jpegxl-rs.evidence.pqc-pr7-reducer-holdout-partial-2026-08-22/1 { hash "sha256:a3d9cc6819973d8f86a583b502c3aaf79848765afd538a54430885068eb54774" } +seal @jpegxl-rs.evidence.pqc-qpv2-expanded-corpus-gate-2026-08-25/1 { + state verified + hash "sha256:f2bcdb8fe27e27f07ecdbdc02473d08f900730068580a5d7010dfbac15e777e2" +} + seal @jpegxl-rs.evidence.pqc-quality-rate-curve-exact-10of13-2026-08-23/1 { state verified hash "sha256:1d790e523631ee2a2718fdd8d36e9f771abcc6e87a6a6b19d2f9574a227079a7" @@ -6065,6 +6292,11 @@ seal @jpegxl-rs.evidence.pqc-quality-rate-workspace-gates-2026-08-23/1 { hash "sha256:0b79b56ce0add2a7dfcfb65e2fe21ffd19c63d29c58eef060fe7ae8aabbd2e30" } +seal @jpegxl-rs.evidence.pqc-scatter-production-identity-2026-08-25/1 { + state verified + hash "sha256:cf0e291df7b6b964933db25872fdaedcd256bea565ebb15dc0af43b7edf9a906" +} + seal @jpegxl-rs.evidence.pqc-workspace-gates-2026-08-22/1 { state verified hash "sha256:2ec904ea78689b5b6d6b13bf089b1ec9216440c0fe53933a7adfcc624bae23eb" @@ -6375,6 +6607,11 @@ seal @jpegxl-rs.evidence.quality-q9-workspace-gates-2026-08-19/1 { hash "sha256:1354865c8a1ed5c86b4115e25fce5f36ba058ca2ca0cda994a6819ce5b55394a" } +seal @jpegxl-rs.evidence.readme-current-state-2026-08-25/1 { + state verified + hash "sha256:989ae9559b849bf5403b1190578cd14817311aeecda969877ffb3c41aff8c427" +} + seal @jpegxl-rs.evidence.release-workspace-tests-2026-08-21/1 { state verified hash "sha256:0adca8daccad59afa3f6628047cfa6c0987dc2f28d25dbf22f85df15ccb46a49" @@ -6465,6 +6702,11 @@ seal @jpegxl-rs.evidence.spine-conventions/1 { hash "sha256:5793d09f02a73315418bdbcd130b2116fba1c79ae1b09148cc2deb782032c0d3" } +seal @jpegxl-rs.evidence.workspace-gates-2026-08-25/1 { + state verified + hash "sha256:153332cecba0e30394a6f437136d4b713462bec9a737dab6abf96acffa6dabe6" +} + seal @jpegxl-rs.milestone.durable-rulings/1 { state completed hash "sha256:5b51c57197a52f692c0735aabb0dec45b946cf834767e8ffa53bb1bb22d98d35" @@ -6780,6 +7022,11 @@ seal @jpegxl-rs.observation.patches-k3/1 { hash "sha256:bca247170ce59df488ca3e70fa061470da5f680454af775bd9bdbb99364ff5f7" } +seal @jpegxl-rs.observation.pgo-gain-remeasured-2026-08-26/1 { + state verified + hash "sha256:a52cc8346a8a844e0a336c7d0502b6d08803fa7c8b89d2e6e78f2fdadfb9d467" +} + seal @jpegxl-rs.observation.phase0-diag-12mp-2026-08-06/1 { state verified hash "sha256:0ad130b5940213668dae12b110ad8430b528f5cec580b2f37a352916e0b8209e" @@ -6870,11 +7117,26 @@ seal @jpegxl-rs.observation.phase5-quality-policy-screen-2026-08-11/11 { hash "sha256:c311117e5b24ac418aadfdabdf0ad660b0adc23ceb0b008014d0fb3583e681dc" } +seal @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1 { + state verified + hash "sha256:2b6173b98676727bede354bb3488928d0ec58c1efd7dcbbf4ef927ea0ec4c51f" +} + seal @jpegxl-rs.observation.pqc-memory-50mp-oom-2026-08-22/1 { state verified hash "sha256:cad2fc5acdba36cfe38becc4b019e94721e78eaafbab1bb3c2cfe4ac46af0295" } +seal @jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/1 { + state superseded + hash "sha256:e4bb7745b41276fc39ac8e64510ed636a2b72af5d566527f6cd1285ceb29f014" +} + +seal @jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/2 { + state verified + hash "sha256:ca161c3f5e33a8423510798f8f02a42c9ff8a249e0253a4b30746b80dc823195" +} + seal @jpegxl-rs.observation.pqc-pr4-development-split-2026-08-22/1 { state verified hash "sha256:551aa85444ac169493e69682d6e76d070126b8b8489b847ed95c918029ac3302" @@ -6885,6 +7147,16 @@ seal @jpegxl-rs.observation.pqc-pr4-rate-curve-axis-audit-2026-08-23/1 { hash "sha256:b5cc2fc639a5ff68cf86ff841e24eb7cb8b57268addaaa0386c62fa29815569f" } +seal @jpegxl-rs.observation.pqc-pr7-full-holdout-quality-rejected-2026-08-24/1 { + state verified + hash "sha256:1109634526a1ddc0fa6bb5b91274107c3d16b335ed5c6b8e6c1bbd72cd89b882" +} + +seal @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25/1 { + state verified + hash "sha256:7cc5c06e34773ef0692dacc6763c8e677730dd1172aede7d96d744eceec361ec" +} + seal @jpegxl-rs.observation.preopt-encode-baseline-2026-08-06/1 { state superseded hash "sha256:237d838644e7329dac763d29c12a89e23016f8eea889b92bb29faba53b3cc089" @@ -6980,6 +7252,11 @@ seal @jpegxl-rs.observation.q6-zero-run-term-explains-dct8-residual-only-2026-08 hash "sha256:b8f2dbcfa64821d55adec25c2e7bcacda934a401eefbbd7723fb364caa4f259a" } +seal @jpegxl-rs.observation.quality-corpus-critical-class-families-2026-08-25/1 { + state verified + hash "sha256:576bd479e06fb2d5d69373a7ba5f73bf941f179bede18cf38b6208c3ae362421" +} + seal @jpegxl-rs.observation.quantizer-normalised-distortion-is-the-lever-2026-08-12/1 { state verified hash "sha256:e56c8f017d1e65f60d136491437082485b23dc243579d744580724d60c5376d8" @@ -7055,6 +7332,11 @@ seal @jpegxl-rs.papercut.a-harmless-scratch-setup-command-was-rejected/1 { hash "sha256:02b2818aed32aa725a64082233cdf17902c2e392312d3bf0efc844f9c7a82fca" } +seal @jpegxl-rs.papercut.a-work-revision-that-cites-already-committed/1 { + state verified + hash "sha256:a9a46c3accb9e5662018b6c51e6963c46227779915fccdfbad4ed2760c7cb5b9" +} + seal @jpegxl-rs.papercut.after-a-temporary-comparison-worktree-restored/1 { state verified hash "sha256:1c15195f3309872a0066c124388b0c7067fd750b408fde501d053993771b3ad7" @@ -7095,6 +7377,11 @@ seal @jpegxl-rs.papercut.akr-view-active-work-current-state-etc-returned/1 { hash "sha256:2a61604b0658be410cee41308798d1ebbd09eaddc359920070ef288f28094216" } +seal @jpegxl-rs.papercut.amending-an-akr-generated-commit-message-to/1 { + state verified + hash "sha256:324b2d0be9193c85fd78d61b16b0b192e5018869696403eb3e83deaaa0e0719b" +} + seal @jpegxl-rs.papercut.cargo-fmt-all-check-fails-on-code-committed-at/1 { state verified hash "sha256:71d820d4bdb6ed7e69feba0fc64163724fa0206d5d81f06df6edc39b8e5f4f7a" @@ -7150,6 +7437,11 @@ seal @jpegxl-rs.papercut.knowledge-complete-s-cited-evidence-gets-its/1 { hash "sha256:a6ec1e870934aaba989ff74eb2303653d98e95bbcd0fd31ee51ce413de0e2a46" } +seal @jpegxl-rs.papercut.knowledge-get-with-detail-canonical-on-jpegxl/1 { + state verified + hash "sha256:963859e4ca865daa4b566ce5dcee87f04cd6329e871cf488e0b82b09d58fa0ed" +} + seal @jpegxl-rs.papercut.knowledge-propose-documents-observation/1 { state verified hash "sha256:52ae41822ad3434cc6063f6406d9468358b6efff701e4c4e92813d0510f30e6e" @@ -7165,6 +7457,11 @@ seal @jpegxl-rs.papercut.knowledge-propose-exposes-topic-for-every/1 { hash "sha256:b99598934c10fe506bb7e86efdd5010c579d466bb4f4f315ad679a1de6b81e28" } +seal @jpegxl-rs.papercut.knowledge-propose-s-generic-slots-schema-does/1 { + state verified + hash "sha256:1fef80d0caae9853d9d3e19e4690b5461e69a0bc0c860b1dbe9175fc7ac1444a" +} + seal @jpegxl-rs.papercut.knowledge-propose-says-observation-requires/1 { state verified hash "sha256:b33e9338149f0f018960d14444e79d81216c226c2abe324f84c777b21b8c00c2" @@ -7205,6 +7502,11 @@ seal @jpegxl-rs.papercut.the-frozen-pqc-holdout-reducer-scripts-and/1 { hash "sha256:e738634f68ea43c20be6b63cae06d85bd456f4d92da73b8d1df463f37b1a62ca" } +seal @jpegxl-rs.papercut.the-frozen-pr7-holdout-parser-assumed/1 { + state verified + hash "sha256:8a26bf882b01e30ce17f1b0fe4814bed149f6d7bd3a92e70e94f1f92c59abba4" +} + seal @jpegxl-rs.papercut.the-installed-akr-0-3-1-rejected-four-existing/1 { state verified hash "sha256:afd89dd7ff8b6d1849dd88765b833898038ff37d09b17d51edb174065b593850" @@ -7950,6 +8252,11 @@ seal @jpegxl-rs.work.general-use-api-cli/2 { hash "sha256:5a06768e00971924a9a0b85b80406c86d2c9f1fe5f1d0deaad629c016bdd6d3e" } +seal @jpegxl-rs.work.jpeg-bitstream-recompression/1 { + state active + hash "sha256:ce3dba766d1838cacfa52c7f95faf9c996bebf87a5013c532a4b798d7a41bce2" +} + seal @jpegxl-rs.work.native-windows-quality-tooling/1 { state completed hash "sha256:32381a6237cf21c76162f2b8c666262d29a970603310d53ac7f5116173ee1095" @@ -7990,6 +8297,26 @@ seal @jpegxl-rs.work.opt-v2-rate-work-sharing/2 { hash "sha256:93522c79a39a3f4c6d63f5859638303fd543d1dc0511601607377870e2c75224" } +seal @jpegxl-rs.work.pqc-one-shot-controller/1 { + state superseded + hash "sha256:01b5735443d7e1701fca2610f7b14de046ef700387824ffb4e7a8f58ca252e68" +} + +seal @jpegxl-rs.work.pqc-one-shot-controller/2 { + state superseded + hash "sha256:cb0014971ecf67042acc47d0a458c8143961be1258d98133f4f9e99038818206" +} + +seal @jpegxl-rs.work.pqc-one-shot-controller/3 { + state active + hash "sha256:b08e9ecd74467c21be5f661c5c94417d63576222c3713c1c9a9cacfd07a1773f" +} + +seal @jpegxl-rs.work.pqc-one-shot-pr0-hard-floor/1 { + state completed + hash "sha256:b835fe82f41cf362b357e9b22d4f9cc5b6ed0c8ed7a256da9c4e51d18576c366" +} + seal @jpegxl-rs.work.pqc-pr0-provenance-corpus-calibration/1 { state completed hash "sha256:4a9c3697ecf25305f271d44a852cb36969c221b9e24ffd3e9d72169b7d930a70" @@ -8020,6 +8347,11 @@ seal @jpegxl-rs.work.pqc-pr5-policy-bank/1 { hash "sha256:ad3fd1c19cd3e3e0f4072a699a19bf89319a89ac7416708a14f2b83e5287a9e4" } +seal @jpegxl-rs.work.pqc-pr7-holdout-closure/1 { + state completed + hash "sha256:18aac74973a69a91a06178e90647d485892f52bb74f71604424c374a6ac90567" +} + seal @jpegxl-rs.work.pqc-pr7-terminal-reducer/1 { state abandoned hash "sha256:635adcd9a015d6c13ad9ba16f7e9780acf274ff714c5d5cf05464dc992c401f2" @@ -8071,8 +8403,33 @@ seal @jpegxl-rs.work.pqc-usable-efforts-cost/4 { } seal @jpegxl-rs.work.pqc-usable-efforts-cost/5 { + state superseded + hash "sha256:6b5fa483e0774275d95fb55701356b4b712a1e8e3bbc8e3ba65387ab3ed15799" +} + +seal @jpegxl-rs.work.pqc-usable-efforts-cost/6 { + state superseded + hash "sha256:281900161facaedb1d325cb274dfec47a1a48830edde29c310686ec05de15865" +} + +seal @jpegxl-rs.work.pqc-usable-efforts-cost/7 { + state superseded + hash "sha256:33268ef1d1bbb0dfcc10f00996e905d82bb20e07021dc496a0564e80121af9b9" +} + +seal @jpegxl-rs.work.pqc-usable-efforts-cost/8 { + state superseded + hash "sha256:7ce16e490dba480b9e9f117b466a7566c0af391b305812081973379f0e7ebb9c" +} + +seal @jpegxl-rs.work.pqc-usable-efforts-cost/9 { state active - hash "sha256:7f9bd01e4a9b9cfed55fd58d5e5b6c561a87856ad1ec256ed96c7aad6a08dc86" + hash "sha256:a0dd993d34e6e6b53af2c95644e83482dbd8aa80efa966d60dcc808f8b775fb3" +} + +seal @jpegxl-rs.work.publish-readme-benchmark-mit/1 { + state completed + hash "sha256:99149d1ef7358b3e24f53ec0fd705b7c8093b979333c973bdabbb1a223de48d3" } seal @jpegxl-rs.work.quality-q0-harness-attribution/1 { diff --git a/.akr/records/jpegxl-rs/decisions.akr b/.akr/records/jpegxl-rs/decisions.akr index 7218f20a..8e53b5e5 100644 --- a/.akr/records/jpegxl-rs/decisions.akr +++ b/.akr/records/jpegxl-rs/decisions.akr @@ -435,6 +435,35 @@ record jpegxl-rs.decision.perceptual-quality-contract/1 : decision { } } +record jpegxl-rs.decision.quality-miss-fallback-semantics/1 : decision { + title "An under-target perceptual result is refused at the facade by default; lossless and best-effort are explicit opt-in fallbacks" + state proposed + scope [ + path "JPXL/crates/jpxl-cli/**", + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl/**" + ] + decision """ + An under-target perceptual result is never an ordinary success at the public facade. When the bounded controller stops at SaturatedTop or UnderTargetWorkCap, the default is to refuse: Encoder returns Error::TargetNotMet carrying a QualityMiss (kind LadderSaturated | WorkBudgetExhausted, requested and best verified scores, probe/price counts, metric version, quality trace), and the CLI exits 1 without creating the output file. Two explicit fallbacks exist via Encoder::with_quality_fallback / --quality-fallback: Lossless emits a mathematically lossless stream reported as PerceptualStatus::FallbackLossless with achieved score 100, and BestEffort emits the finest canonically verified under-target stream under its true status and score. SaturatedFloor remains a success. The policy layer (search_frame_perceptual) still returns every terminal status with bytes; the facade owns the refusal. The rescue probe stays one probe beyond the navigation cap and is documented as an observable per-solve maximum of pixel_probes + 1; folding it inside the cap is a deliberate search-policy change that would require the standing Contract B screen. + """ + context """ + The 2026-08-24 one-shot advisor memo's first code finding, verified against quality.rs, jpxl/src/lib.rs and jpxl-cli/src/main.rs: the facade wrapped SaturatedTop and UnderTargetWorkCap outcomes as Ok and the CLI wrote the output file and exited 0, so a scripted caller checking only the exit code received a silently under-target stream — incompatible with the hard-floor reading of the perceptual-quality contract. The completed 91-cell locked holdout contains no under-target cells, so refusing changes no canonical-sweep behavior. + """ + consequences """ + Met, MetAdjacentRungs, MetWorkCap, SaturatedFloor and RescuedFreshStructure outputs stay byte-identical. The public API adds Error::TargetNotMet, QualityMiss, QualityMissKind, QualityFallback, PerceptualStatus::FallbackLossless and Encoder::with_quality_fallback; the CLI adds --quality-fallback lossless|best-effort. A refused encode still appends its jpxl.quality-trace/1 record to JPXL_QUALITY_TRACE so failed searches remain calibration input. Facade, policy-layer and CLI tests now pin the miss path, which previously had no coverage anywhere. + """ + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + author "GitHub Uploader" + source { + kind external + role origin + document "jpxl-one-shot-quality-controller-memo-2026-08-24" + use """ + Adopted the memo's PR 0 contract repair: refusal by default, explicit lossless/best-effort fallback modes, structured failure payload, CLI exit-1 with no output file, and public miss-path tests. + """ + } +} + record jpegxl-rs.decision.quant-bias-defaults/1 : decision { title "quant_bias defaults are 1 minus x" state proposed diff --git a/.akr/records/jpegxl-rs/evidence.akr b/.akr/records/jpegxl-rs/evidence.akr index c1fed104..1ed843bb 100644 --- a/.akr/records/jpegxl-rs/evidence.akr +++ b/.akr/records/jpegxl-rs/evidence.akr @@ -1,6 +1,34 @@ akr 0.1 project jpegxl-rs +record jpegxl-rs.evidence.avx2-pooling-bit-identity-2026-08-26/1 : evidence { + title "jpegxl-rs.evidence.avx2-pooling-bit-identity-2026-08-26" + state verified + result pass + method command + observed_at git:d2a54b5486b90fd4808e9bb77c67eea6bbd9f010 + command "cargo test -p jpxl-perceptual; cmp on full A/B encodes vs HEAD-baseline build" + summary """ + accumulate_band's avx2+fma clone is byte-identical to the scalar + walk: the lane test covers remainder lengths 0,1,7,8,9,64,250, + and full quality-path encodes of the canonical images compare + equal with cmp against a pre-change baseline binary. Wall-clock + on the 12 MP quality path improved about 5-8% at 4 threads. + """ +} + +record jpegxl-rs.evidence.bench-vs-libjxl-reproducible-2026-08-25/1 : evidence { + title "One documented command compares JPXL and native cjxl/djxl on the same inputs with recorded provenance" + state verified + result pass + method command + observed_at git:a523d0157ac223678b9d02342d0a8a7fbc34d642 + command "JPXL/tools/bench_vs_libjxl.sh --jpxl target/release/jpxl --quality \"70 85\" --runs 1 --jsonl bench-vs-libjxl.jsonl " + summary """ + After tools/setup-oracles.sh produced native Linux cjxl/djxl v0.13.0 (196a43d9, pinned in oracle-bin/PINNED_REVISIONS.txt), bench_vs_libjxl.sh emitted a size/bpp/SSIMULACRA2/wall table for jpxl --quality 70/85 and cjxl -d 3.0/1.5/1.0 on three corpus images (2400x1800 photo, 1600x900 text-screenshot, 700x700 radial gradient), every stream decoded and scored by the same in-tree metric, with a provenance header (UTC date, host, binary versions and sha256, flags, per-input sha256/dims) and machine-readable jpxl.bench-vs-libjxl/1 JSONL kept at pqc-scatter-20260825/bench-vs-libjxl.jsonl. Sample: on the photo, jpxl q85 emitted 796,985 bytes at score 85.45 versus cjxl -d1.0 at 804,397 bytes and score 83.41; on text-screenshot cjxl -d3.0 was 4.7x smaller at a higher score, consistent with the recorded VarDCT-on-synthetic gap. + """ +} + record jpegxl-rs.evidence.cli-netpbm-subtypes-2026-08-21/1 : evidence { title "Focused CLI tests passed: explicit PPM/PGM output uses P6/P5 and incompatible channel layouts error." state verified @@ -583,6 +611,18 @@ record jpegxl-rs.evidence.history-backfilled/1 : evidence { author "GitHub Uploader" } +record jpegxl-rs.evidence.jpeg-phase-a-roundtrip-2026-08-25/1 : evidence { + title "jpxl-jpeg roundtrips the full 5,218-file Pol Art archive byte-identically except one provably corrupt file, with typed refusals proven" + state verified + result pass + method command + observed_at git:c1cdbf582689914c46af67c1b3e59cacc88b920b + command "cd JPXL && cargo run --release -p jpxl-jpeg --example roundtrip_sweep -- \"/mnt/Samsung980_1TB/Pol Desktop/Pol Art Folder 08-24-2017\"" + summary """ + The deterministic sweep (sorted paths, stride 1 — a superset of the >=500-sample requirement) parsed and re-serialized every JPEG in the archive: 5,217 of 5,218 byte-identical with zero mismatches and zero refusals; coverage of the identical set spans baseline 3,166, progressive 2,049, extended sequential 2, chroma 4:4:4 2,985 / 4:2:0 2,081 / 4:2:2 140 / 4:4:0 6, grayscale, restart intervals 684, Exif 2,419 / ICC 1,558 / XMP 1,396, and 41 files with post-EOI tail data. The single non-roundtripping file is ~15 KB of real data zero-padded to 3.18 MB, rejected by libjpeg's djpeg as premature EOF and refused here with a typed malformed-stream error. All 81 earlier mismatches shared one root cause — libjpeg's non-derivable progressive AC EOB-run splits — fixed by recording the exact run lengths at decode and replaying them flush-on-match at encode; the prog_large.jpg regression fixture proves the recorded runs are load-bearing. Typed refusal of arithmetic, hierarchical, lossless, and 12-bit inputs is proven by 10 dedicated tests; 34 crate tests pass in release. + """ +} + record jpegxl-rs.evidence.m1-plan-ir/1 : evidence { title "M1 - VarDCT plan IR and structural split" state verified @@ -763,6 +803,19 @@ record jpegxl-rs.evidence.markdown-cutover/2 : evidence { author "GitHub Uploader" } +record jpegxl-rs.evidence.mit-only-licensing-audit-2026-08-25/1 : evidence { + title "Workspace licensing surface is MIT-only with no copyleft anywhere in the dependency graph" + state verified + result pass + method command + observed_at git:a523d0157ac223678b9d02342d0a8a7fbc34d642 + command "cd JPXL && cargo tree --workspace (dependency/license enumeration; full table in JPXL/docs/LICENSING-AUDIT.md)" + artifact "JPXL/docs/LICENSING-AUDIT.md" + summary """ + All 12 workspace crates declare MIT via license.workspace; LICENSE and JPXL/LICENSE-MIT carry consistent MIT text; a scan of the full dependency graph (default ~41 external crates plus ~26 behind off-by-default measurement features) found no GPL/LGPL/AGPL/MPL/CDDL/EUPL/SSPL license anywhere; the only non-MIT arms are permissive (moxcms and pxfm BSD-3-Clause/Apache-2.0 on the default path via the image raster adapter; ssimulacra2 BSD-2, butteraugli BSD-3, v_frame BSD-2, imgref CC0/Apache behind feature gates). No ISO text or AGPL-derived content is present in any shipped crate. Full table in JPXL/docs/LICENSING-AUDIT.md. + """ +} + record jpegxl-rs.evidence.modular-effort-ramp-landed-2026-08-10/1 : evidence { title "At 70688ac the lean default (Effort::DEFAULT = 1) reproduces the pre-change default's bytes exactly on all three photo-corpus sizes: small_0p8MP 1,075,465 B sha16 C3C60CEE4D0741D9, mid_4MP 5,157,974 B sha16 5C49AB9020706F0D, large_12MP 9,795,599 B sha16 CA27799D3E7B441D -- all IDENTICAL to the pre-ramp fingerprints. Wall time 267 / 619 / 1165 ms for the default versus 1185 / 5324 / 16919 ms for --effort 7, which reproduces the same three fingerprints and so remains the byte-identical density anchor. Speedup on this corpus is 4.4x / 8.6x / 14.5x for identical output." state verified @@ -5035,6 +5088,75 @@ record jpegxl-rs.evidence.phase9-quant-scheduling-speed-2026-08-15/1 : evidence """ } +record jpegxl-rs.evidence.pipeline-metadata-seam-2026-08-26/1 : evidence { + title "jpegxl-rs.evidence.pipeline-metadata-seam-2026-08-26" + state verified + result pass + method command + observed_at git:05733e9b32cec8e460e552fea5f01c20c87ca941 + command "cargo test --workspace (jpxl facade + jpxl-encode suites)" + summary """ + The metadata seam is verified end to end at this commit: an + appended Exif box round-trips through the decoder's box walk + with the codestream untouched; with_exif rejects payloads + without a TIFF byte-order header; all four ColourSpace values + round-trip through the decoder's ColourEncoding parse; a + greyscale image rejects non-sRGB signalling; lossy VarDCT + returns a typed Unsupported for non-sRGB input. Cross-repo, a + real 25 MB Samsung DNG's 574-byte TIFF blob survived + raw-autotune develop, JPXL encode, openarc archive and extract + byte-identically (recorded on the openarc ledger). + """ +} + +record jpegxl-rs.evidence.pqc-fused-render-memory-12mp-2026-08-25/1 : evidence { + title "Fused-render 12 MP memory: peak 2,008,408 KiB under the 2 GiB ceiling" + state verified + result pass + method command + observed_at git:c23bc009ae6cd1a288e1922a60fe534dc711324c + command "python3 .agent/scratch/pqc-metric-cuts-20260825/rss_12mp.py" + summary """ + Balanced q85 on the locked 12 MP anchor after the fused render: peak RSS 2,007,060-2,008,408 KiB over three serialized capped runs, under the 2,097,152 KiB ceiling; output bytes unchanged (1,315,649). Report in kept scratch pqc-metric-cuts-20260825/rss-12mp.json. + """ +} + +record jpegxl-rs.evidence.pqc-fused-render-production-identity-2026-08-25/1 : evidence { + title "Fused-render production identity: 52/52 matrix and 6/6 pre-change A/B byte-identical" + state verified + result pass + method command + observed_at git:c23bc009ae6cd1a288e1922a60fe534dc711324c + command "python3 .agent/scratch/pqc-metric-cuts-20260825/identity_scatter.py" + summary """ + After the fused depth-quantized probe render, the full locked production-identity matrix is 52/52 byte-identical (threads 1/4, AVX2 on/off) and the 6-cell A/B against the pre-change 1fc3ca7 reference binary is byte-identical across threads 1/4/8 and AVX2 off; the classifier's per-sample equivalence with encode-then-round-trip is additionally pinned by the fused_depth_levels_match_the_srgb_round_trip unit test. Reports in kept scratch pqc-metric-cuts-20260825 (round 2; round-1 reports preserved as *-round1). + """ +} + +record jpegxl-rs.evidence.pqc-fused-render-wall-anchors-2026-08-25/1 : evidence { + title "Fused-render wall anchors: 4.3 MP at 2.50x (8t), 2.0x target still unmet" + state verified + result fail + method command + observed_at git:c23bc009ae6cd1a288e1922a60fe534dc711324c + command "python3 .agent/scratch/pqc-metric-cuts-20260825/wall_scatter.py" + summary """ + Interleaved matched-score timing after the fused render: quality/rate 3.06x at 4 threads (load_1m 1.6-1.9) and 2.50x at 8 (load 4.0-4.2) on the 4.3 MP anchor; 4.65x at 4 threads on 12 MP (load 1.9-3.7); the 12 MP 8-thread schedule cell read 3.54x under ambient load 4.0-5.4 and is not representative — quiet manual 12 MP 8-thread quality runs measured 2.39-2.53 s against the schedule's 2.79 s median. Down from 3.40x/2.76x and 5.07x/3.50x before the fusion the same day. The 2.0x wall-anchors target remains unmet on both anchors. Report in kept scratch pqc-metric-cuts-20260825/wall-scatter.json (round 1 preserved as wall-scatter-round1.json). + """ +} + +record jpegxl-rs.evidence.pqc-fused-render-workspace-gates-2026-08-25/1 : evidence { + title "Fused-render workspace gates: fmt, clippy -D warnings, release tests all pass" + state verified + result pass + method command + observed_at git:c23bc009ae6cd1a288e1922a60fe534dc711324c + command "cd JPXL && set -o pipefail && cargo fmt --all --check && cargo clippy --workspace --all-targets -- -D warnings && cargo test --workspace --release" + summary """ + One strictly chained run at the fused-render tree: cargo fmt --all --check, clippy --workspace --all-targets -D warnings, and the release workspace test suite (76 test groups) all pass, exit 0. + """ +} + record jpegxl-rs.evidence.pqc-low-memory-12mp-2026-08-23/1 : evidence { title "Low-memory Balanced q85 path meets the 12 MP two-GiB RSS ceiling" state verified @@ -5093,6 +5215,105 @@ record jpegxl-rs.evidence.pqc-low-memory-xlarge-identity-2026-08-23/1 : evidence """ } +record jpegxl-rs.evidence.pqc-memory-12mp-2026-08-25/1 : evidence { + title "Balanced q85 on the locked 12 MP anchor peaks at 1.95 GiB RSS, under the 2.0 GiB ceiling" + state verified + result pass + method command + observed_at git:b0c128c785d7524c6002eb34ca7ebd990b24bf1e + command "python3 .agent/scratch/pqc-scatter-20260825/rss_12mp.py" + summary """ + Three serialized release encodes of the locked 4000x3000 anchor (Balanced q85, 4 threads, RLIMIT_AS 9 GiB, ru_maxrss via wait4) peaked at 2,036,248-2,043,056 KiB, all under the 2,097,152 KiB (2.0 GiB) ceiling, with identical 1,315,649-byte output each run; rows in the kept scratch entry pqc-scatter-20260825/rss-12mp.json. + """ +} + +record jpegxl-rs.evidence.pqc-metric-cuts-memory-12mp-2026-08-25/1 : evidence { + title "Metric-cuts 12 MP memory: peak 2,008,188 KiB under the 2 GiB ceiling" + state verified + result pass + method command + observed_at git:c0253b3134674f9223100ccd992fca59815161a0 + command "python3 .agent/scratch/pqc-metric-cuts-20260825/rss_12mp.py" + summary """ + Balanced q85 on the locked 12 MP anchor after the exact metric-path cost cuts: peak RSS 2,007,068-2,008,188 KiB over three serialized capped runs, under the 2,097,152 KiB ceiling and down from 2,043,056 before the cuts; output bytes unchanged (1,315,649). Report in kept scratch pqc-metric-cuts-20260825/rss-12mp.json. + """ +} + +record jpegxl-rs.evidence.pqc-metric-cuts-production-identity-2026-08-25/1 : evidence { + title "Metric-cuts production identity: 52/52 matrix and 6/6 pre-change A/B byte-identical" + state verified + result pass + method command + observed_at git:c0253b3134674f9223100ccd992fca59815161a0 + command "python3 .agent/scratch/pqc-metric-cuts-20260825/identity_scatter.py" + summary """ + After the exact metric-path cost cuts, the full locked production-identity matrix is 52/52 byte-identical (threads 1/4, AVX2 on/off), and a 6-cell A/B against the pre-change reference binary (worktree build of 1fc3ca7) is byte-identical across threads 1/4/8 and AVX2 off; reports in kept scratch pqc-metric-cuts-20260825 (identity-scatter-summary.json, identity-ab.json). + """ +} + +record jpegxl-rs.evidence.pqc-metric-cuts-wall-anchors-2026-08-25/1 : evidence { + title "Metric-cuts wall anchors: ratios cut ~9-19 percent, 2.0x target still unmet" + state verified + result fail + method command + observed_at git:c0253b3134674f9223100ccd992fca59815161a0 + command "python3 .agent/scratch/pqc-metric-cuts-20260825/wall_scatter.py" + summary """ + Interleaved matched-score timing after the exact metric-path cost cuts (standing recipe, load_1m 1.8-4.5): quality/rate 3.40x at 4 threads and 2.76x at 8 on the 4.3 MP anchor, 5.07x and 3.50x on 12 MP, from 3.87x/3.22x and 5.42x/4.25x before the cuts (that run at load 5.2-7.7; the ratio columns are the load-robust comparison). Quality medians 1.33/1.17 s (4.3 MP) and 3.65/2.62 s (12 MP). The 2.0x wall-anchors target remains unmet on both anchors. Report in kept scratch pqc-metric-cuts-20260825/wall-scatter.json. + """ +} + +record jpegxl-rs.evidence.pqc-metric-cuts-workspace-gates-2026-08-25/1 : evidence { + title "Metric-cuts workspace gates: fmt, clippy -D warnings, release tests all pass" + state verified + result pass + method command + observed_at git:c0253b3134674f9223100ccd992fca59815161a0 + command "cd JPXL && set -o pipefail && cargo fmt --all --check && cargo clippy --workspace --all-targets -- -D warnings && cargo test --workspace --release" + summary """ + One strictly chained run at the metric-cuts tree: cargo fmt --all --check, clippy --workspace --all-targets -D warnings, and the release workspace test suite (76 test groups) all pass, exit 0. + """ +} + +record jpegxl-rs.evidence.pqc-one-shot-falsification-screen-2026-08-24/1 : evidence { + title "One-shot falsification screen: fails on corpus coverage, transform features validated" + state verified + result fail + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "python JPXL/tools/quality_predictor_v2.py train --labels oracle-labels.jsonl --feature-sets table2 source source+transform --transform-features transform-features.json --sweep-dir oracle-sweeps --ridge-grid 0.03 0.3 3.0" + summary """ + The memo's pre-PR5 falsification screen fails on the expanded 38-family corpus: best model (pooled quantile regression, source+transform features, ridge 3.0) reaches leave-one-family-out median |ln err| 0.220 and p90 0.805 vs the quick-screen p90<=0.45 and production median<=0.10/p90<=0.30 bounds. First-plan success 0.88 and simulated common-case bytes 1.0016 pass their checks, and first-or-one-correction success is 1.00, but the honest uncertainty intervals route 99% of requests to the exact controller (expected reconstructions 3.98, no wall win). Tail is concentrated in saturated (p90 3.58) and sky-noise gradient (p90 4.07) classes - single-content-family coverage - while photo/scene/text/noise/line-art/grayscale sit at p90 0.36-0.67. Standalone transform-summary cost on the 12 MP anchor is 21.9% of a matched-rate encode (fails the 5% bolt-on gate; in-search shared-cache marginal cost unmeasured). Report kept in .agent/scratch/one-shot-qpv2-20260824/qpv2-report.json. + """ + author "GitHub Uploader" +} + +record jpegxl-rs.evidence.pqc-one-shot-pr1-5-gates-2026-08-24/1 : evidence { + title "One-shot PR 1-5 implementation: all workspace gates pass" + state verified + result pass + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "cargo test --workspace && cargo test -p jpxl-encode-policy --features one-shot-controller && cargo clippy --workspace --all-targets [--features one-shot-controller] && cargo fmt --all --check && python -m unittest discover -s JPXL/tools/tests && python JPXL/tools/make-quality-guard-fixtures.py build --check" + summary """ + The one-shot program's PR 1-5 implementation passes every workspace gate: full test suite green (71 suites) in the default build and 155 policy tests under the one-shot-controller feature (including the PR 5 routing test and the trace/2 assertions), clippy clean in both configurations, formatting clean, 75 Python tool tests green (trainer, oracle-label, fixture-generator suites), and the fixture generator's check mode reproduces all 56 corpus PPMs byte-identically with the family split-hygiene guard passing. Default-build behavior is unchanged: the shadow predictor only logs, and the one-shot start is compiled out. + """ + author "GitHub Uploader" +} + +record jpegxl-rs.evidence.pqc-one-shot-promotion-ab-2026-08-25/1 : evidence { + title "One-shot promotion A/B: floor held, bytes and work strictly better in aggregate" + state verified + result pass + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "python JPXL/tools/one_shot_promotion_ab.py --manifest test-set/quality-corpus.json --splits holdout ... ; python JPXL/tools/one_shot_promotion_ab.py --manifest test-set/ext-polart-manifest.json --splits ext-holdout ..." + summary """ + Promotion A/B of the one-shot-controller build against the default controller, interleaved on the same machine. Locked 13-image holdout (91 cells): zero floor violations both arms, byte geomean 0.99744 (bound 1.007), wall geomean 0.989, reconstructions 348 vs 339; worst cells are three sub-kilobyte saturated fixtures (max ratio 1.51 = +131 bytes absolute). Never-tuned ext-holdout paintings (50 families, 350 cells): zero floor violations, byte geomean 0.99018, reconstructions 1287 to 1056 (-18%), wall geomean 0.838, worst cell 1.117 at target 95 with a higher achieved score. Achieved-score dips above the floor exist in both arms (1 default, 2 one-shot) - a pre-existing bounded-search artifact, no hard inversion below target anywhere. Wall anchors at q85 vs matched-rate: 4.3 MP 1.36x both arms; 12 MP 8.02x default vs 6.31x one-shot (metric/render remains the dominant residual, as the memo predicted). Raw rows kept in .agent/scratch/one-shot-qpv2-20260824/ (ab-locked-holdout.json, ab-ext-holdout.json). + """ + author "GitHub Uploader" +} + record jpegxl-rs.evidence.pqc-pr0-corpus-manifest-2026-08-22/1 : evidence { title "The generator reproduces 47 fixtures (47 PPM + 29 PNG + 47 JSON provenance sidecars, idempotent sha256) and the jpxl.codec-corpus/1 manifest test-set/quality-corpus.json (gitignored) validates; splits are calibration 19, development 15, holdout 13 with no source family in two splits; classes text-screenshot 5, line-art 5, gradient 6, saturated 4, tiny 6, noise-lowlight 1, grayscale 2, photo-scene 7, photo 11. 14 generator unit tests pass." state verified @@ -5106,6 +5327,19 @@ record jpegxl-rs.evidence.pqc-pr0-corpus-manifest-2026-08-22/1 : evidence { """ } +record jpegxl-rs.evidence.pqc-pr0-hard-floor-tests-2026-08-24/1 : evidence { + title "PR0 hard floor: targeted miss-path tests pass" + state verified + result pass + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "cargo test -p jpxl-encode-policy running_out_of_probes && cargo test -p jpxl --test quality_encode && cargo test -p jpxl-cli --test cli_quality" + summary """ + New miss-path tests all pass: policy-layer UnderTargetWorkCap with rescue accounting (pixel_probes 1 + 1 rescue), facade refuse/lossless/best-effort on the deterministic 64x64 gradient miss at 99.5 Fast (TargetNotMet with LadderSaturated, FallbackLossless bytes equal to the lossless encoder's, best-effort re-scored independently), and 8/8 CLI quality tests including exit-1-with-no-output-file refusal and --quality-fallback validation. + """ + author "GitHub Uploader" +} + record jpegxl-rs.evidence.pqc-pr0-predictor-table-2026-08-22/1 : evidence { title "256 fixed-quantizer points (16 calibration images x 16 global_scale rungs 400..73728) scored with the reference SSIMULACRA2 in 604 s produced quality_predictor.rs: 56 table cells over 5 luma-variance x 3 flat-fraction buckets plus a fallback log fit [15.62, -1.60, 0.0099, -2.57]. Leave-one-image-out |ln(pred/actual)| median 0.30, p90 2.68 (cells hold 1-4 images); loss-vs-scale slope median -2.34; saturation at 73728 is 0 up to target 70 and 0.06/0.19/0.81/0.94 at 80/85/90/95." state verified @@ -5130,6 +5364,19 @@ record jpegxl-rs.evidence.pqc-pr0-sources-registered-2026-08-22/1 : evidence { """ } +record jpegxl-rs.evidence.pqc-pr0-workspace-gates-2026-08-24/1 : evidence { + title "PR0 hard floor: workspace and tools gates pass" + state verified + result pass + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "cargo test --workspace && cargo test --workspace --profile fast-debug && cargo clippy --workspace --all-targets && cargo fmt --all --check && python -m unittest discover -s JPXL/tools/tests" + summary """ + Workspace gates pass after the hard-floor change: full test suite green in both debug and fast-debug profiles (zero failures), clippy emits no warnings, formatting is clean, and the 46 Python tool tests pass (codec_compare now opts into --quality-fallback best-effort so curve points on an unmet target stay measurable). + """ + author "GitHub Uploader" +} + record jpegxl-rs.evidence.pqc-pr1-api-semantics-2026-08-22/1 : evidence { title "6/6 facade target tests pass: score 100 is byte-identical to the lossless path, scores outside 0..=100 are rejected, conflicting targets (quality+bpp, bpp+global_scale) are rejected, a perceptual score below 100 runs the quality controller, --global-scale encodes and decodes, and the rate report matches the bytes." state verified @@ -5430,6 +5677,18 @@ record jpegxl-rs.evidence.pqc-pr7-balanced-single-shot-2026-08-22/1 : evidence { """ } +record jpegxl-rs.evidence.pqc-pr7-closure-workspace-gates-2026-08-24/1 : evidence { + title "PR7 holdout closure workspace and feature gates pass" + state verified + result pass + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "cd JPXL && cargo build --workspace; cargo test --workspace --release; cargo clippy --workspace --all-targets -- -D warnings; cargo fmt --all --check; cargo test -p jpxl-cli --release --features quality-effort,perceptual" + summary """ + Workspace build, full release suite, strict Clippy, formatting, and the quality-effort+perceptual feature-enabled CLI release tests all pass after the policy note and regression-test update. + """ +} + record jpegxl-rs.evidence.pqc-pr7-dev-split-2026-08-22/1 : evidence { title "Development split, Balanced with the reducer on versus off: bytes geomean 0.9825 at target 70 (photos 0.9903, min 0.888) and 0.9912 at 85 (photos 0.9886, min 0.9835); zero floor violations in 30 cells (the reduced stream is kept only when its exact size is smaller and every accepted batch is re-scored canonically); at most 6 evaluations; wall geomean 1.53x / 1.63x. Bytes down at matched-or-better score with bounded work, but the wall exceeds Balanced's +25% budget, so the reducer ships default-off on Balanced (BALANCED_DEFAULT_REDUCER = None) and on for the feature-gated Quality effort; the locked-holdout gate is still to be measured." state verified @@ -5442,6 +5701,52 @@ record jpegxl-rs.evidence.pqc-pr7-dev-split-2026-08-22/1 : evidence { """ } +record jpegxl-rs.evidence.pqc-pr7-holdout-complete-2026-08-24/1 : evidence { + title "Full PR7 locked-holdout Quality/reducer sweep completes after low-memory changes" + state verified + result pass + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "python3 .agent/scratch/pr7-holdout-closure-20260824/resume_holdout.py && python3 .agent/scratch/pr7-holdout-closure-20260824/analyze.py" + summary """ + Combined historical checkpoints with 53 resumed rows to close 91/91 Quality and 91/91 Balanced cells on all 13 locked holdout images. All 182 successful rows had zero canonical-floor and decoder failures; reducer work stayed at <=6 evaluations, saved 263,266 exact bytes total, and produced 0.991552 final/before geomean. The formerly blocked 50 MP Quality rows completed serially under a 12 GiB cap; worst measured peak was 11,163,372 KiB. Kept report: .agent/scratch/pr7-holdout-closure-20260824/report.json (SHA-256 fd78df93...f282e). + """ +} + +record jpegxl-rs.evidence.pqc-pr7-policy-gating-test-2026-08-24/1 : evidence { + title "Production budgets keep rejected Quality-only work disabled" + state verified + result pass + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "cd JPXL && cargo test -p jpxl-encode-policy --profile fast-debug production_budgets_keep_quality_only_work_disabled" + summary """ + The new policy regression test passes and proves Fast/Balanced default budgets retain zero policy trials and no reducer while the feature-gated Quality budget retains its bank and terminal reducer. + """ +} + +record jpegxl-rs.evidence.pqc-pr7-promotion-disposition-2026-08-24/1 : evidence { + title "Full Contract B result is dispositioned without weakening the public gate" + state verified + result pass + method observation + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + summary """ + The full 79-cell common-score report was evaluated against the standing Quality promotion conditions. Overall matched bytes pass at 0.958292, but Butteraugli mean 1.047881 and worst pnorm3 1.868818 fail their 1.02/1.05 limits, so the existing quality-effort feature gate remains in force. No search-policy change is justified; the source note now records the completed gate and a regression test pins Quality-only work out of Fast/Balanced. + """ +} + +record jpegxl-rs.evidence.pqc-pr7-quality-promotion-full-2026-08-24/1 : evidence { + title "Full locked holdout rejects public Quality promotion under Contract B" + state verified + result fail + method observation + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + summary """ + Across 79 common-score cells, Quality/Balanced bytes pass at 0.958292 geomean, but Butteraugli mean ratio is 1.047881 (>1.02) and worst pnorm3 ratio is 1.868818 (>1.05), so Contract B fails. Photos are nearly byte-neutral at 0.991367 with BA mean 1.000790 and worst pnorm3 1.042495; non-photo savings drive the aggregate while saturated, tiny, and text cells drive the guard failure. Quality remains feature-gated. + """ +} + record jpegxl-rs.evidence.pqc-pr7-quality-promotion-rejected-2026-08-22/1 : evidence { title "Partial locked holdout definitively rejects public Quality promotion" state verified @@ -5468,6 +5773,19 @@ record jpegxl-rs.evidence.pqc-pr7-reducer-holdout-partial-2026-08-22/1 : evidenc author "GitHub Uploader" } +record jpegxl-rs.evidence.pqc-qpv2-expanded-corpus-gate-2026-08-25/1 : evidence { + title "Expanded-corpus qpv2 model: blind holdout hits production p90; transform features win" + state verified + result pass + method command + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + command "python JPXL/tools/quality_corpus_extend.py --source-dir \"D:\\Pol Desktop\\Pol Art Folder 08-24-2017\" --name ext-polart --max-images 220; quality_oracle_labels.py sweep (2h budget); quality_predictor_v2.py train --feature-sets table2 source source+transform" + summary """ + Corpus expanded with 211 independent painting families from the user's collection (104 calibration / 57 development / 50 never-tuned ext-holdout, deterministic split by family hash, originals untouched); 103 swept within the 2-hour budget for 994 total label rows (zero censored). Retrained pooled model, 10-fold family-grouped CV: source+transform wins (CV median 0.121 / p90 0.461 vs source-only 0.176/0.595); blind never-tuned ext-holdout: median 0.113, p90 0.258 (production p90 bound met), p99 0.478, first-plan 0.925, first-or-one-correction 1.00. Remaining CV tail is entirely the saturated (p90 3.58) and sky-noise gradient classes, still effectively class-held-out. Emitted runtime model qpv2-st-1 (19 features: 9 source + 10 DCT8-summary); the in-search transform summary measured -3.9% wall at 4.3 MP (shares the cover's cache; standalone 7.8% after parallelizing the fill from 21.9% serial). Artifacts in .agent/scratch/one-shot-qpv2-20260824/. + """ + author "GitHub Uploader" +} + record jpegxl-rs.evidence.pqc-quality-rate-curve-exact-10of13-2026-08-23/1 : evidence { title "Corrected common-axis rate curve is exact on 10/13 holdout images; xlarge quality remains memory-blocked" state verified @@ -5537,6 +5855,18 @@ record jpegxl-rs.evidence.pqc-quality-rate-workspace-gates-2026-08-23/1 : eviden """ } +record jpegxl-rs.evidence.pqc-scatter-production-identity-2026-08-25/1 : evidence { + title "Production identity holds after the band-parallel scatter: 52/52 locked cells identical across threads 1/4 and AVX2 on/off" + state verified + result pass + method command + observed_at git:b0c128c785d7524c6002eb34ca7ebd990b24bf1e + command "python3 .agent/scratch/pqc-scatter-20260825/identity_scatter.py" + summary """ + The full locked 13-image matrix (both production efforts, q70 and q85, serialized under a 9 GiB address-space cap) produced 52/52 byte-identical cells across threads 1, threads 4, and threads 4 with JPXL_DISABLE_AVX2=1 on the band-parallel-scatter binary; separately, six image/target cells (both anchors at q85/q70, Fast and Balanced, text-screenshot and gradient) hashed identically between this binary and a reference binary built from commit 1fc3ca7 in an isolated worktree, and within the new binary at threads 1/4/8 and AVX2 off. Rows and summaries are in the kept scratch entry pqc-scatter-20260825 (identity-scatter-summary.json, identity-ab.json). + """ +} + record jpegxl-rs.evidence.pqc-workspace-gates-2026-08-22/1 : evidence { title "Release test suite green across the workspace (the one failure seen in the background run was the PR 1 placeholder CLI test, rewritten in the same tree), clippy clean under -D warnings with default and extended feature sets, fmt clean. Rate-mode production streams on mid.ppm (--bpp 1.0, 4 threads) hash 05bae79d4c96f77b2bb6bd3b1ad6a794331323d11e7c55903db4c4359798701b (balanced) and 07d71108de1fc69fd4fa5cb9e0e71917ec0887510bd21bc7db5eab4a0bf6bc6b (fast), identical to the pre-change binary." state verified @@ -6298,6 +6628,18 @@ record jpegxl-rs.evidence.quality-q9-workspace-gates-2026-08-19/1 : evidence { """ } +record jpegxl-rs.evidence.readme-current-state-2026-08-25/1 : evidence { + title "README describes the current supported scope, exclusions, clean-room policy, build steps, and the recorded benchmark" + state verified + result pass + method manual + observed_at git:a523d0157ac223678b9d02342d0a8a7fbc34d642 + artifact "README.md" + summary """ + The README's claims were verified against JPXL/docs/CONFORMANCE.md and the actual CLI contracts (jpxl --help, encode, compare); the quality-search section now correctly describes the promoted one-shot seed (qpv2-st-1 pooled quantile regression behind the default one-shot-controller feature, with the calibrated-table fallback for out-of-envelope inputs) and the full-resolution canonical verification before every emission, replacing the stale claim that the calibrated table picks the first probe; the benchmark section documents the reproducible bench_vs_libjxl.sh entry point, and the license section states the no-copyleft position with a pointer to the audit. + """ +} + record jpegxl-rs.evidence.release-workspace-tests-2026-08-21/1 : evidence { title "Release-mode workspace tests pass" state verified @@ -6515,3 +6857,15 @@ record jpegxl-rs.evidence.spine-conventions/1 : evidence { """ author "GitHub Uploader" } + +record jpegxl-rs.evidence.workspace-gates-2026-08-25/1 : evidence { + title "Workspace release gates pass with jpxl-jpeg, the scatter change, and the publish surface in the tree" + state verified + result pass + method command + observed_at git:c1cdbf582689914c46af67c1b3e59cacc88b920b + command "cd JPXL && cargo test --workspace --release && cargo clippy --workspace --all-targets -- -D warnings && cargo fmt --all --check" + summary """ + Under pipefail in one chained run: cargo test --workspace --release reported 76 test-result groups, all ok with zero failures (including the new jpxl-jpeg suite); cargo clippy --workspace --all-targets -D warnings finished clean; cargo fmt --all --check passed. + """ +} diff --git a/.akr/records/jpegxl-rs/observations.akr b/.akr/records/jpegxl-rs/observations.akr index 452785bc..a2a7b1f3 100644 --- a/.akr/records/jpegxl-rs/observations.akr +++ b/.akr/records/jpegxl-rs/observations.akr @@ -1091,6 +1091,22 @@ record jpegxl-rs.observation.patches-k3/1 : observation { } } +record jpegxl-rs.observation.pgo-gain-remeasured-2026-08-26/1 : observation { + title "PGO gain re-measured at 2-8%; Phase 8.5 magnitudes are historical" + state verified + statement """ + Re-measured at this commit, the combined release-final (fat LTO) + plus PGO cycle gains only 2-8% over the thin-LTO release profile on + the canonical training and unseen images. The Phase 8.5 PGO + magnitudes recorded around 2026-08-14 (36-53%) predate the + Opt-F..P rounds and the 2026-08-25 fused probe-render work and no + longer describe HEAD; treat them as historical, and re-verify + before citing PGO as a major lever. + """ + observed_at git:b1f7196c325cce789620b87a0907bf594c82a73b + method command +} + record jpegxl-rs.observation.phase0-diag-12mp-2026-08-06/1 : observation { title "Phase-0 measured multiplicity at 4000x3000 (threads 1 vs auto)" state verified @@ -1759,6 +1775,26 @@ record jpegxl-rs.observation.phase5-quality-policy-screen-2026-08-11/11 : observ ] } +record jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1 : observation { + title "Parallelizing the serial varblock render and linearization cut 12 MP quality wall 21%; ratio 3.11x -> 2.46x on the measuring host" + state verified + scope [ + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl-perceptual/src/evaluator.rs", + path "JPXL/crates/jpxl-plan-render/**" + ] + statement """ + Per-probe phase attribution on the locked 12 MP anchor (fast-debug, comparing 1-thread vs 8-thread walls to identify serial stages) showed the reference side of SSIMULACRA2 already precomputed once per search (PrecomputedReference with ReferenceRetention::Moments at <=24 MP); the serial per-probe stages were varblock reconstruction (500-730 ms; a 12 MP frame is one LF group, so group-level parallelism contributed nothing) and transfer-curve linearization (105-180 ms). After banding both over the EncodeExecutor (map_ordered varblock chunks of 2048 with serial disjoint scatter; banded LUT lookup), varblocks measure ~235 ms and linearize ~18 ms; per-probe render+metric fell ~1.3 s to ~0.77 s. Release wall on the anchor at 8 threads: 4.923 s -> 3.906 s (-21%), matched-rate ratio 3.11x -> 2.46x on this host (the recorded 6.31x/8.02x baselines were another host/state; the relative move is the reliable signal). Byte-identical streams on six image/target cells, identical canonical scores, cross-thread determinism at threads 1/4/8 in both profiles, peak working set unchanged at 1,899 MB, all touched-crate gates green. Remaining per-probe residual is co-dominated by the still-serial varblock sample scatter (~part of 235 ms) and the already-parallel metric (~380 ms); the largest further headroom is score-changing (navigate on a reduced pyramid or downscaled reconstruction, full-resolution only for final verification), which requires a Contract B promotion A/B. + """ + observed_at git:9c6e64a2f146e2805af9124a9bd061d7bebce763 + method command + watches [ + "JPXL/crates/jpxl-plan-render/src/lib.rs", + "JPXL/crates/jpxl-perceptual/src/evaluator.rs" + ] + author "GitHub Uploader" +} + record jpegxl-rs.observation.pqc-memory-50mp-oom-2026-08-22/1 : observation { title "Perceptual encodes of 50 MP sources reach ~12 GB RSS; two concurrent measurement runs OOM-killed a 31 GB host twice" state verified @@ -1784,6 +1820,48 @@ record jpegxl-rs.observation.pqc-memory-50mp-oom-2026-08-22/1 : observation { ] } +record jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/1 : observation { + title "The one-shot falsification screen fails on corpus coverage, not on the design: transform features halve the error, the tail is two under-represented classes" + state superseded + scope [ + path "JPXL/crates/jpxl-encode-policy/src/quality_prediction.rs", + path "JPXL/crates/jpxl-encode-policy/src/quality_predictor_v2.rs", + path "JPXL/tools/quality_oracle_labels.py", + path "JPXL/tools/quality_predictor_v2.py" + ] + statement """ + On the expanded 56-image / 38-family corpus with production-endpoint oracle labels (fresh Balanced pixel plans over the complete effective ladder; zero censored rows even at target 95, so the old predictor's 81-94% high-target saturation was an artifact of its HfMul=1 ladder cap), the memo's pre-PR5 falsification screen fails: best leave-one-family-out median |ln(predicted/oracle crossing)| is 0.220 with p90 0.805 (bounds: quick p90<=0.45; production median<=0.10, p90<=0.30). Three findings shape the path forward. (1) Transform-domain features materially help, as the memo predicted: source-only p90 1.28 vs source+transform 0.81, median 0.42 vs 0.22, first-or-one-correction success 1.00. (2) The failure is class coverage, not model form: photo, photo-scene, text-screenshot, noise-lowlight, line-art and grayscale classes sit at held-out p90 0.36-0.67 (near or past the quick bound), while saturated (p90 3.58) and sky-noise gradient (p90 4.07) dominate the tail - both effectively class-held-out because their families carry near-unique content. (3) The fallback discipline works: with honest intervals the simulated controller routes 99% of requests to the exact path, keeping bytes at 1.0016x but eliminating the wall win (expected reconstructions 3.98 vs the current median 4). Per the memo, production controller work stops here until the corpus grows (>=25 independent families per critical class; saturated and banding-stress content first); the PR5 mechanism exists behind the off-by-default `one-shot-controller` cargo feature and remains inert without a gate-passing model. + """ + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + method command + watches [ + "JPXL/crates/jpxl-encode-policy/src/quality_predictor_v2.rs", + "JPXL/tools/quality_predictor_v2.py", + "test-set/quality-corpus.json" + ] + derived_from [ @jpegxl-rs.work.pqc-one-shot-controller ] + verified_by [ @jpegxl-rs.evidence.pqc-one-shot-falsification-screen-2026-08-24 ] + author "GitHub Uploader" +} + +record jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/2 : observation { + title "The one-shot falsification screen fails on corpus coverage, not on the design: transform features halve the error, the tail is two under-represented classes" + state verified + scope [ path "JPXL/crates/jpxl-encode-policy/src/**", path "JPXL/tools/**" ] + statement """ + On the expanded 56-image / 38-family corpus with production-endpoint oracle labels (fresh Balanced pixel plans over the complete effective ladder; zero censored rows even at target 95, so the old predictor's 81-94% high-target saturation was an artifact of its HfMul=1 ladder cap), the memo's pre-PR5 falsification screen fails: best leave-one-family-out median |ln(predicted/oracle crossing)| is 0.220 with p90 0.805 (bounds: quick p90<=0.45; production median<=0.10, p90<=0.30). Three findings shape the path forward. (1) Transform-domain features materially help, as the memo predicted: source-only p90 1.28 vs source+transform 0.81, median 0.42 vs 0.22, first-or-one-correction success 1.00. (2) The failure is class coverage, not model form: photo, photo-scene, text-screenshot, noise-lowlight, line-art and grayscale classes sit at held-out p90 0.36-0.67 (near or past the quick bound), while saturated (p90 3.58) and sky-noise gradient (p90 4.07) dominate the tail - both effectively class-held-out because their families carry near-unique content. (3) The fallback discipline works: with honest intervals the simulated controller routes 99% of requests to the exact path, keeping bytes at 1.0016x but eliminating the wall win (expected reconstructions 3.98 vs the current median 4). Per the memo, production controller work stops here until the corpus grows (>=25 independent families per critical class; saturated and banding-stress content first); the PR5 mechanism exists behind the off-by-default `one-shot-controller` cargo feature and remains inert without a gate-passing model. + """ + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + method command + watches [ "JPXL/crates/jpxl-encode-policy/src/quality_predictor*.rs", "JPXL/tools/*.py" ] + derived_from [ @jpegxl-rs.work.pqc-one-shot-controller ] + supersedes [ + @jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/1 + ] + verified_by [ @jpegxl-rs.evidence.pqc-one-shot-falsification-screen-2026-08-24 ] + author "GitHub Uploader" +} + record jpegxl-rs.observation.pqc-pr4-development-split-2026-08-22/1 : observation { title "PR 4 on the development split: the score floor holds in 150/150 encodes, photographs beat cjxl -e7 at matched score, synthetic text and line art trail badly, and the quality path costs ~3.5x the rate path's wall time" state verified @@ -1837,6 +1915,56 @@ record jpegxl-rs.observation.pqc-pr4-rate-curve-axis-audit-2026-08-23/1 : observ ] } +record jpegxl-rs.observation.pqc-pr7-full-holdout-quality-rejected-2026-08-24/1 : observation { + title "Full PR7 holdout closes the OOM tail and rejects Quality promotion" + state verified + scope [ + path "JPXL/crates/jpxl-encode-policy/src/policy_bank.rs", + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl-encode-policy/src/reducer.rs", + path "JPXL/crates/jpxl-perceptual/src/**" + ] + statement """ + The current low-memory implementation completed the 27 formerly missing Quality cells and 26 companion Balanced rows, closing all 91 cells per effort on the 13-image locked holdout. The terminal reducer is safe and bounded (zero floor/decoder failures, at most six evaluations, final/before bytes geomean 0.991552), but public Quality promotion is rejected: matched-score bytes pass overall at 0.958292, while Contract B fails at Butteraugli mean ratio 1.047881 and worst pnorm3 ratio 1.868818. The added cells make the diagnosis sharper rather than reversing it: photos are nearly byte-neutral at 0.991367 and pass the BA/pnorm3 guards; non-photo byte wins coexist with saturated/tiny/text guard losses. Keep Quality feature-gated and keep its policy bank/reducer out of Fast and Balanced; a future promotion attempt needs a different quality policy or metric guidance, not relaxation of the gate. + """ + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + method command + watches [ + "JPXL/crates/jpxl-encode-policy/src/quality.rs", + "JPXL/crates/jpxl-encode-policy/src/policy_bank.rs", + "JPXL/crates/jpxl-encode-policy/src/reducer.rs", + "JPXL/crates/jpxl-perceptual/src/**" + ] + derived_from [ + @jpegxl-rs.evidence.pqc-pr7-quality-promotion-rejected-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr7-reducer-holdout-partial-2026-08-22/1 + ] + verified_by [ + @jpegxl-rs.evidence.pqc-pr7-holdout-complete-2026-08-24/1, + @jpegxl-rs.evidence.pqc-pr7-quality-promotion-full-2026-08-24/1 + ] +} + +record jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25/1 : observation { + title "Quiet-host wall anchors: 12 MP 5.42x (4t) / 4.25x (8t); the quality path's fixed costs alone sit near 2x, so the 2.0x target is unreachable by search improvements" + state verified + scope [ + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl-perceptual/src/**", + path "JPXL/crates/jpxl-plan-render/src/**" + ] + statement """ + Interleaved matched-score timing (wall_current.py recipe: one warm-up per kind, 10-run schedule, 5 GiB address-space cap, host state per run) on a quiet host with the current release binary: 4.3 MP anchor quality/rate 2.021/0.521 s = 3.88x at 4 threads and 1.824/0.566 s = 3.22x at 8; 12 MP anchor 5.464/1.007 s = 5.42x at 4 threads and 3.988/0.938 s = 4.25x at 8. The 4-thread rate median (1.007 s) reproduces the 2026-08-23 baseline exactly, so the comparable quiet-host series at 12 MP/4t is 8.00x (2026-08-23) -> 5.42x (after the one-shot promotion cut probes 5->3); the 2.46x recorded on 2026-08-25 divided by an inflated ~1.59 s rate denominator measured under load and is not comparable. Phase attribution (12 MP, 8 threads): plan 366 ms, render_metric 1985 ms (three pixel probes at 642-697 ms each), entropy 79 ms, emit 99 ms, two exact prices 89 ms each; ~1.4 s of the 3.99 s wall is outside the controller (source PPM load, metric reference precompute, feature extraction, output IO). Consequence: the quality path's fixed costs excluding all scored probes (~1.83 s at 8 threads) are already 1.95x the rate path, and a perfect single-probe controller would land near 2.7x, so the wall-anchors 2.0x acceptance cannot be met by search improvements alone; it requires reducing per-pixel render+metric cost ~2-3x (score-identity risk) or renegotiating the target. Separately, the band-parallel varblock scatter is exact but wall-neutral at the 12 MP anchor: render_metric 2043/2168 ms (pre-change binary) vs 2107/2057 ms (post), and the 4 MP varblock stage is 45 ms at 4 threads in both trees, so the earlier ~50-100 ms/probe scatter estimate was high. Raw runs, traces, and comparisons are in the kept scratch entry pqc-scatter-20260825 (wall-scatter.json, trace-*.jsonl). + """ + observed_at git:b0c128c785d7524c6002eb34ca7ebd990b24bf1e + method command + watches [ + "JPXL/crates/jpxl-plan-render/src/lib.rs", + "JPXL/crates/jpxl-perceptual/src/**", + "JPXL/crates/jpxl-encode-policy/src/quality.rs" + ] +} + record jpegxl-rs.observation.preopt-encode-baseline-2026-08-06/1 : observation { title "Pre-optimization encode wall-time ladder (release jpxl bench)" state superseded @@ -2244,6 +2372,19 @@ record jpegxl-rs.observation.q6-zero-run-term-explains-dct8-residual-only-2026-0 ] } +record jpegxl-rs.observation.quality-corpus-critical-class-families-2026-08-25/1 : observation { + title "Quality-corpus generator now provides 26 saturated and 25 gradient/banding families; locked holdout fixtures byte-unchanged" + state verified + scope [ path "JPXL/tools/make-quality-guard-fixtures.py", path "test-set/**" ] + statement """ + The synthetic quality-guard generator was extended with distinct deterministic content processes so the two critical classes named by the one-shot memo's section 10 now meet its >=25-families floor: saturated 4 -> 26 and gradient/banding-stress 8 -> 25 (each synthetic fixture is its own family). The corpus grew 56 -> 95 fixtures (calibration 43, development 32, holdout 20); the split audit reports no family in more than one split, regeneration is sha256-idempotent on a second run, codec_compare.py manifest-check validates all 95 images, and 16 generator unit tests pass (including a new >=25-families gate). All 13 fixtures pinned by the locked PR4-closure holdout manifest are byte-unchanged, so prior identity and rate-curve evidence remains comparable. SSIMULACRA2 oracle-label sweeps over the new families were deliberately not run; they are the remaining step before the model can be retrained on this coverage. + """ + observed_at git:6e13305b647024c4fe4a1eabcef36560cd586a26 + method command + watches [ "JPXL/tools/make-quality-guard-fixtures.py" ] + derived_from [ @jpegxl-rs.work.pqc-one-shot-controller ] +} + record jpegxl-rs.observation.quantizer-normalised-distortion-is-the-lever-2026-08-12/1 : observation { title "The standard's own quant matrices carry a third of the frequency weighting; measuring distortion in quantizer-normalised units cuts Y mispricing 2.65x to 1.86x" state verified diff --git a/.akr/records/jpegxl-rs/papercuts.akr b/.akr/records/jpegxl-rs/papercuts.akr index 2fda5ff5..0b82fc16 100644 --- a/.akr/records/jpegxl-rs/papercuts.akr +++ b/.akr/records/jpegxl-rs/papercuts.akr @@ -37,6 +37,18 @@ record jpegxl-rs.papercut.a-harmless-scratch-setup-command-was-rejected/1 : pape created_at 2026-08-18 } +record jpegxl-rs.papercut.a-work-revision-that-cites-already-committed/1 : papercut { + title "A work revision that cites already-committed evidence in verified_by…" + state verified + statement """ + A work revision that cites already-committed evidence in verified_by lands "not satisfied - predates the last change" unless the evidence rows are committed in the very same commit as the revision (the current-together grace); revising a record in a later .akr-only commit therefore un-satisfies checks whose measurements are still perfectly valid, and there is no way to re-land unchanged evidence. Retargeting a check citation onto fresh evidence needs either an observed_at at-or-after the revision commit (chicken-and-egg) or duplicated evidence records. Hit while pointing pqc-usable-efforts-cost's workspace-gates check at the 2026-08-25 gates evidence; repaired by dropping the revision commit. + """ + observed_at git:7f7c6d1433906f0f6751af94b2903cd3e3f80f72 + about "akr" + author "claude" + created_at 2026-08-25 +} + record jpegxl-rs.papercut.after-a-temporary-comparison-worktree-restored/1 : papercut { title "After a temporary comparison worktree restored a modified source with…" state verified @@ -128,6 +140,18 @@ record jpegxl-rs.papercut.akr-view-active-work-current-state-etc-returned/1 : pa created_at 2026-08-06 } +record jpegxl-rs.papercut.amending-an-akr-generated-commit-message-to/1 : papercut { + title "Amending an akr-generated commit message (to replace the generic…" + state verified + statement """ + Amending an akr-generated commit message (to replace the generic "chore:" subject with a descriptive one while keeping the AKR trailers byte-identical) is rejected by the commit-msg hook with AKR-C031 because the change transaction closes at akr git commit; the only way through is git commit --amend --no-verify. Either akr git commit could accept a subject/body override, or the hook could allow amends whose trailers match the last commit. + """ + observed_at git:a523d0157ac223678b9d02342d0a8a7fbc34d642 + about "akr" + author "claude" + created_at 2026-08-25 +} + record jpegxl-rs.papercut.cargo-fmt-all-check-fails-on-code-committed-at/1 : papercut { title "`cargo fmt --all --check` fails on code committed at HEAD (25213f6):…" state verified @@ -255,6 +279,18 @@ record jpegxl-rs.papercut.knowledge-complete-s-cited-evidence-gets-its/1 : paper created_at 2026-08-07 } +record jpegxl-rs.papercut.knowledge-get-with-detail-canonical-on-jpegxl/1 : papercut { + title "knowledge.get with detail canonical on…" + state verified + statement """ + knowledge.get with detail canonical on jpegxl-rs.work.pqc-usable-efforts-cost/7 truncated at the tool's token cap and the suggested continuation (detail summary) cannot return the acceptance block's source text either, so reproducing checks verbatim for a revision required reading .akr/records/jpegxl-rs/work.akr by hand. A continuation or an acceptance-only detail level would remove the need to touch the records directory. + """ + observed_at git:4bbb629fd4dc8bcf3cd33199b2d4aefdad39e351 + about "akr" + author "claude" + created_at 2026-08-25 +} + record jpegxl-rs.papercut.knowledge-propose-documents-observation/1 : papercut { title "knowledge.propose documents observation observed_at as required but…" state verified @@ -291,6 +327,18 @@ record jpegxl-rs.papercut.knowledge-propose-exposes-topic-for-every/1 : papercut created_at 2026-08-23 } +record jpegxl-rs.papercut.knowledge-propose-s-generic-slots-schema-does/1 : papercut { + title "knowledge.propose's generic `slots` schema does not expose that an…" + state verified + statement """ + knowledge.propose's generic `slots` schema does not expose that an observation's `method` must be one of manual/command/instrumented/observation. V-022 said to add `method`; supplying explanatory prose then produced dozens of parser diagnostics before the enum expectation appeared. + """ + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + about "akr" + author "codex" + created_at 2026-08-24 +} + record jpegxl-rs.papercut.knowledge-propose-says-observation-requires/1 : papercut { title "knowledge.propose says observation requires statement and observed_at,…" state verified @@ -384,6 +432,17 @@ record jpegxl-rs.papercut.the-frozen-pqc-holdout-reducer-scripts-and/1 : papercu created_at 2026-08-22 } +record jpegxl-rs.papercut.the-frozen-pr7-holdout-parser-assumed/1 : papercut { + title "The frozen PR7 holdout parser assumed `ssimulacra2` immediately…" + state verified + statement """ + The frozen PR7 holdout parser assumed `ssimulacra2` immediately followed `psnr_db`; the current compare line inserts `ssimulacra2_jpxl` and its version first, so an otherwise successful 12 MP resumed cell was needlessly rerun. Parse named fields independently or share the checked-in compare parser. + """ + observed_at git:41a2e3f20631e000954b9ea76612c70f1847e745 + author "codex" + created_at 2026-08-24 +} + record jpegxl-rs.papercut.the-installed-akr-0-3-1-rejected-four-existing/1 : papercut { title "The installed akr 0.3.1 rejected four existing observation records…" state verified diff --git a/.akr/records/jpegxl-rs/work.akr b/.akr/records/jpegxl-rs/work.akr index 2bb91876..45165403 100644 --- a/.akr/records/jpegxl-rs/work.akr +++ b/.akr/records/jpegxl-rs/work.akr @@ -1028,7 +1028,9 @@ record jpegxl-rs.work.arch-phase24-hfmul-overlay/3 : work { depends_on [ @jpegxl-rs.work.arch-phase23-cow-geometry ] part_of [ @jpegxl-rs.track.encoder-optimization ] supersedes [ @jpegxl-rs.work.arch-phase24-hfmul-overlay/2 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase25-multi-quantizer-workspace/1 : work { @@ -1094,7 +1096,9 @@ record jpegxl-rs.work.arch-phase25-multi-quantizer-workspace/1 : work { } depends_on [ @jpegxl-rs.work.arch-phase24-hfmul-overlay/3 ] part_of [ @jpegxl-rs.track.encoder-optimization ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase27-finalist-only-entropy/1 : work { @@ -1155,7 +1159,9 @@ record jpegxl-rs.work.arch-phase27-finalist-only-entropy/1 : work { depends_on [ @jpegxl-rs.evidence.phase26-rate-multiplicity-2026-08-16/1 ] implements [ @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase28-cfl-lf-scratch-reuse/1 : work { @@ -1538,7 +1544,9 @@ record jpegxl-rs.work.arch-phase31-lane4-vector-select/1 : work { verified_by [ @jpegxl-rs.evidence.phase31-profile-2026-08-16/2 ] } } - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase32-dirty-frontier-screen/1 : work { @@ -2778,7 +2786,9 @@ record jpegxl-rs.work.arch-phase5a-aq-policy-gate/1 : work { @jpegxl-rs.policy.research-out-of-core ] part_of [ @jpegxl-rs.track.encoder-optimization ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5b-restoration-screen/1 : work { @@ -2829,7 +2839,9 @@ record jpegxl-rs.work.arch-phase5b-restoration-screen/1 : work { @jpegxl-rs.policy.clean-room-boundary ] part_of [ @jpegxl-rs.track.encoder-optimization ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5c-lf-hf-balance-screen/1 : work { @@ -2925,7 +2937,9 @@ record jpegxl-rs.work.arch-phase5c-lf-hf-balance-screen/2 : work { ] part_of [ @jpegxl-rs.track.encoder-optimization ] supersedes [ @jpegxl-rs.work.arch-phase5c-lf-hf-balance-screen/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5d-cover-quality-screen/1 : work { @@ -2975,7 +2989,9 @@ record jpegxl-rs.work.arch-phase5d-cover-quality-screen/1 : work { @jpegxl-rs.policy.clean-room-boundary ] part_of [ @jpegxl-rs.track.encoder-optimization ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5e-epf-screen/1 : work { @@ -3026,7 +3042,9 @@ record jpegxl-rs.work.arch-phase5e-epf-screen/1 : work { @jpegxl-rs.policy.clean-room-boundary ] part_of [ @jpegxl-rs.track.encoder-optimization ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5f-aqoff-oracle-compat/1 : work { @@ -3144,7 +3162,9 @@ record jpegxl-rs.work.arch-phase5f-aqoff-oracle-compat/2 : work { ] part_of [ @jpegxl-rs.track.encoder-optimization ] supersedes [ @jpegxl-rs.work.arch-phase5f-aqoff-oracle-compat/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5g-aqoff-broader-corpus-gate/1 : work { @@ -3189,7 +3209,9 @@ record jpegxl-rs.work.arch-phase5g-aqoff-broader-corpus-gate/1 : work { @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5h-quant-lf-tail-gate/1 : work { @@ -3225,7 +3247,9 @@ record jpegxl-rs.work.arch-phase5h-quant-lf-tail-gate/1 : work { @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5i-active-epf-screen/1 : work { @@ -3270,7 +3294,9 @@ record jpegxl-rs.work.arch-phase5i-active-epf-screen/1 : work { @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5j-two-pass-error-aq-screen/1 : work { @@ -3319,7 +3345,9 @@ record jpegxl-rs.work.arch-phase5j-two-pass-error-aq-screen/1 : work { @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5k-active-epf-depth-screen/1 : work { @@ -3392,7 +3420,9 @@ record jpegxl-rs.work.arch-phase5k-active-epf-depth-screen/2 : work { ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] supersedes [ @jpegxl-rs.work.arch-phase5k-active-epf-depth-screen/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5l-chroma-qm-allocation-screen/1 : work { @@ -3441,7 +3471,9 @@ record jpegxl-rs.work.arch-phase5l-chroma-qm-allocation-screen/1 : work { @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5m-spatial-epf-sharpness-screen/1 : work { @@ -3486,7 +3518,9 @@ record jpegxl-rs.work.arch-phase5m-spatial-epf-sharpness-screen/1 : work { @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5n-fine-aq-lattice-screen/1 : work { @@ -3533,7 +3567,9 @@ record jpegxl-rs.work.arch-phase5n-fine-aq-lattice-screen/1 : work { @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5o-special8-transform-screen/1 : work { @@ -3588,7 +3624,9 @@ record jpegxl-rs.work.arch-phase5o-special8-transform-screen/1 : work { @jpegxl-rs.policy.optimization-correctness-gates/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase5p-special8-exact-distortion-screen/1 : work { @@ -3683,7 +3721,9 @@ record jpegxl-rs.work.arch-phase6-0-distortion-currency/1 : work { @jpegxl-rs.policy.optimization-correctness-gates ] part_of [ @jpegxl-rs.track.encoder-optimization ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase6-1-quantizer-normalised-residual/1 : work { @@ -4622,7 +4662,9 @@ record jpegxl-rs.work.arch-phase8-4-finalist-token-tape/2 : work { depends_on [ @jpegxl-rs.work.arch-phase8-3-anchor-sketch ] part_of [ @jpegxl-rs.track.encoder-optimization ] supersedes [ @jpegxl-rs.work.arch-phase8-4-finalist-token-tape/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase8-5-leaf-finishing/1 : work { @@ -4969,7 +5011,9 @@ record jpegxl-rs.work.arch-phase8-6-butteraugli-localization/2 : work { ] part_of [ @jpegxl-rs.track.encoder-optimization ] supersedes [ @jpegxl-rs.work.arch-phase8-6-butteraugli-localization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.arch-phase8-6-finalist-measured-hf-allocation/1 : work { @@ -6152,7 +6196,9 @@ record jpegxl-rs.work.gap-g2-selective-coefficient-refinement/1 : work { Adopts the staged entropy-cost, context-aware trailing-cost, bounded finalist beam, exact-walk mismatch, fresh-chroma, and promotion-gate shape as an experiment; the report remains non-authoritative. """ } - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.gap-g3-bounded-truthful-rate-controller/1 : work { @@ -6250,7 +6296,9 @@ record jpegxl-rs.work.gap-g3-bounded-truthful-rate-controller/1 : work { Implements the G3 milestone boundary after completed G2. """ } - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.gap-g4-selective-cover-refresh/1 : work { @@ -6734,6 +6782,62 @@ record jpegxl-rs.work.general-use-api-cli/2 : work { author "GitHub Uploader" } +record jpegxl-rs.work.jpeg-bitstream-recompression/1 : work { + title "Lossless JPEG bitstream recompression: carry JPEG1 coefficients into JPEG XL directly instead of decode-and-reencode" + state active + scope [ + path "JPXL/crates/jpxl-bitstream/**", + path "JPXL/crates/jpxl-cli/**", + path "JPXL/crates/jpxl-decode/**", + path "JPXL/crates/jpxl-encode/**", + path "JPXL/crates/jpxl/**", + path "JPXL/docs/jpeg-recompression-plan.md" + ] + intent """ + Carry JPEG1 quantized DCT coefficients into JPEG XL directly (18181-2 s9.11 jbrd + Annex A) instead of decode-and-reencode, through the phased plan in JPXL/docs/jpeg-recompression-plan.md: A jpeg codec roundtrip, B coefficient carriage, C bit-exact reconstruction, D density, E oracle interop. + """ + note """ + Phase A landed at commit c1cdbf5 (crate jpxl-jpeg): coefficient-level 10918-1 codec, bit-exact re-emission over the full 5,218-file Pol Art archive (5,217 identical; the one exception is a provably truncated file, typed-refused), typed refusal of arithmetic/hierarchical/lossless/12-bit, 34 crate tests. The load-bearing discovery for later phases: libjpeg's progressive AC EOB-run SPLITS are not derivable from coefficients and must be recorded and replayed (ScanSegment.eob_runs) - the jbrd carriage in Phases B/C must preserve this per-scan run-length record, exactly as 18181-2 s9.11's HuffmanCode/padding metadata anticipates. Phase B next: coefficient carriage into a YCbCr VarDCT frame; the encoder still lacks do_YCbCr/jpeg_upsampling/RAW-quant signalling (verified earlier), which is Phase B's first work item. + """ + acceptance { + check phase-a-jpeg-codec-roundtrip { + statement """ + jpxl-jpeg serialize(parse(x)) is byte-identical to x over >=500 deterministically sampled archive JPEGs covering baseline, progressive, 4:4:4/4:2:0/4:2:2, greyscale, restart intervals, Exif/ICC/XMP and trailing garbage; arithmetic-coded, hierarchical, 12-bit and lossless JPEG inputs are refused with a typed error. + """ + method command + command "cd JPXL && cargo run --release -p jpxl-jpeg --example roundtrip_sweep -- \"/mnt/Samsung980_1TB/Pol Desktop/Pol Art Folder 08-24-2017\"" + verified_by [ @jpegxl-rs.evidence.jpeg-phase-a-roundtrip-2026-08-25/1 ] + } + check phase-b-coefficient-carriage { + statement """ + Every JPEG quantized DCT coefficient round-trips exactly through jpxl encode to a YCbCr VarDCT frame and jpxl-decode's coefficient export, on the Phase A corpus, at multiple worker counts. + """ + method command + } + check phase-c-bit-exact-reconstruction { + statement """ + The original JPEG is reconstructed byte-exactly from the .jxl file alone (codestream + jbrd, 18181-2 Annex A) on the Phase A corpus, and the .jxl is smaller than the source JPEG on >=95% of it. + """ + method command + } + check phase-e-oracle-interop { + statement """ + djxl reconstructs our recompressed .jxl to the byte-exact original, and we reconstruct cjxl --lossless_jpeg=1 output byte-exactly, on a 50-image sample, treating both oracles as black boxes. + """ + method command + } + } + part_of [ @jpegxl-rs.track.vardct-encoder/1 ] + source { + kind internal + role origin + path "JPXL/docs/jpeg-recompression-plan.md" + use """ + The implementation plan this record tracks: architecture (new jpxl-jpeg crate, container work in jpxl-bitstream, no policy involvement), phase gates, and open questions. + """ + } +} + record jpegxl-rs.work.native-windows-quality-tooling/1 : work { title "Native Windows equivalents for the last bash-only quality/conformance tooling (cjxl-match, fetch-conformance)" state completed @@ -7033,6 +7137,133 @@ record jpegxl-rs.work.optional-semantic-guidance-consumer/1 : work { part_of [ @jpegxl-rs.track.encoder-optimization/1 ] } +record jpegxl-rs.work.pipeline-metadata-seam/1 : work { + title "EXIF box and colour-space signalling for the archive pipeline" + state proposed + intent """ + Give the archive pipeline the two metadata seams it needs from the + encoder: a Part 2 Exif container box exposed as Encoder::with_exif + (validated raw-TIFF payload, container implied), and declarative + colour-space signalling (sRGB, linear sRGB, Display P3, Rec.2020) + through the Annex E ColourEncoding bundle on the lossless Modular + path, with lossy VarDCT rejecting non-sRGB input by typed error + because XYB and the metric are defined on sRGB. Downstream, + raw-autotune supplies the TIFF blob and colour tag and openarc + routes them through archiving; the contract is a bare TIFF blob, + never an APP1 wrapper. + """ + implements [ @jpegxl-rs.requirement.general-use-integration-surface ] +} + +record jpegxl-rs.work.pqc-one-shot-controller/1 : work { + title "One-shot program: shadow-first common-case one-shot quality controller" + state superseded + scope [ + path "JPXL/crates/jpxl-encode-policy/**", + path "JPXL/crates/jpxl/**", + path "JPXL/tools/**", + path "test-set/**" + ] + intent """ + Adopt the 2026-08-24 advisor memo's common-case one-shot design in shadow-first stages to close the open Balanced wall requirement (<= 2.0x matched-rate on the 4.3 MP and 12 MP anchors). Staged: (PR 1) jpxl.quality-trace/2 with prediction/decision-path fields plus family_id/variant_id in the corpus manifest; (PR 2) a production-endpoint label generator over the full effective-scale ladder - fresh-structure Balanced crossings, saturation represented as censoring, local loss slopes; (PR 3) a source-only shadow model QualityPredictionV2 (log-only, no bitstream effect); (PR 4) a request-scoped transform-summary shadow model reading CandidateForwardCache; (PR 5, only after the gate) the feature-gated common-case path: one predicted fresh plan, canonical verification before entropy, one slope-based correction, then continuation of the existing navigator under the same total caps. + """ + target 2026-09-30 + note """ + 2026-08-24 status: PR 1-4 are implemented and measured; the PR 5 mechanism exists behind the off-by-default `one-shot-controller` cargo feature (model candidate as first fresh plan, model beta as the single-probe slope prior, fallback discipline routing to the exact controller) but stays inert per the memo's stopping rule: the falsification screen FAILED on the 38-family corpus (best LOFO median 0.220 / p90 0.805 vs quick p90<=0.45 and production 0.10/0.30; see evidence pqc-one-shot-falsification-screen-2026-08-24). The verdict is corpus coverage, not design: transform features halve the error and the tail is the saturated and sky-noise classes (effectively class-held-out). Reopening the gate requires corpus growth per the memo's section 10 (>= 25 independent families per critical class; saturated and banding-stress content first), then rerunning quality_oracle_labels.py sweep/labels and quality_predictor_v2.py train. Deferred: in-search transform-summary wall measurement (standalone is 21.9% of matched-rate on the 12 MP anchor, over the 5% bolt-on gate; the shared-cache in-search cost is the number that matters for PR 5 promotion). + """ + depends_on [ @jpegxl-rs.work.pqc-one-shot-pr0-hard-floor ] + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] + source { + kind external + role origin + document "jpxl-one-shot-quality-controller-memo-2026-08-24" + use """ + Adopted the memo's verdict (common-case one shot, canonical verification as the correctness gate), its staged PR 1-5 sequence, its falsification and promotion gates, and its explicit non-goals. + """ + } +} + +record jpegxl-rs.work.pqc-one-shot-controller/2 : work { + title "One-shot program: shadow-first common-case one-shot quality controller" + state superseded + scope [ + path "JPXL/crates/jpxl-encode-policy/**", + path "JPXL/crates/jpxl/**", + path "JPXL/tools/**", + path "test-set/**" + ] + intent """ + Adopt the 2026-08-24 advisor memo's common-case one-shot design to close the open Balanced wall requirement. PR 1-4 (trace/2, production-endpoint oracle labels, source-only and transform-summary shadow models) are implemented and measured; PR 5 (model-seeded first plan, one slope correction, exact-navigator continuation under unchanged caps with canonical verification before every emission) is implemented and, per the user's 2026-08-24 directive and the 2026-08-25 promotion A/B, is now the DEFAULT controller seed (`one-shot-controller` in the policy crate's default features; building without default features restores the legacy table seed). + """ + target 2026-09-30 + note """ + 2026-08-25 status: promoted to default on measured evidence - zero floor violations on 441 never-tuned holdout cells, byte geomean 0.997 (locked holdout) / 0.990 (painting ext-holdout), reconstructions -18% and wall -16% on paintings, 12 MP anchor 8.02x -> 6.31x vs matched-rate. Model qpv2-st-1: pooled quantile regression over 9 source + 10 DCT8-summary features, 130 training families (user's painting collection added via quality_corpus_extend.py); routing: candidate rung when confident, median seed when uncertain-in-distribution, legacy seed when OOD (tiny frames, out-of-envelope features). OPEN ITEMS: (1) the 12 MP wall anchor remains above the 2.0x matched-rate requirement - the residual is render/metric cost per the memo's diagnosis, a separate optimization; (2) worst-cell byte regressions concentrate in sub-kilobyte saturated fixtures (max +131 bytes) and target-95 cells (<=1.12x, higher achieved) - saturated and banding-stress family coverage is still the corpus gap (memo section 10: >= 25 families per critical class); (3) achieved-score dips above the floor exist in both arms (pre-existing bounded-search artifact); (4) 108 unswept ext-polart images remain for future 2-hour label runs (resumable). Reproduce: quality_oracle_labels.py sweep/labels, quality_predictor_v2.py train, one_shot_promotion_ab.py; artifacts kept in .agent/scratch/one-shot-qpv2-20260824/. + """ + depends_on [ @jpegxl-rs.work.pqc-one-shot-pr0-hard-floor ] + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] + supersedes [ @jpegxl-rs.work.pqc-one-shot-controller/1 ] + author "GitHub Uploader" + source { + kind external + role origin + document "jpxl-one-shot-quality-controller-memo-2026-08-24" + use """ + Adopted the memo's verdict (common-case one shot, canonical verification as the correctness gate), its staged PR 1-5 sequence, its falsification and promotion gates, and its explicit non-goals. + """ + } +} + +record jpegxl-rs.work.pqc-one-shot-controller/3 : work { + title "One-shot program: shadow-first common-case one-shot quality controller" + state active + scope [ + path "JPXL/crates/jpxl-encode-policy/**", + path "JPXL/crates/jpxl/**", + path "JPXL/tools/**", + path "test-set/**" + ] + intent """ + Adopt the 2026-08-24 advisor memo's common-case one-shot design to close the open Balanced wall requirement. PR 1-4 (trace/2, production-endpoint oracle labels, source-only and transform-summary shadow models) are implemented and measured; PR 5 (model-seeded first plan, one slope correction, exact-navigator continuation under unchanged caps with canonical verification before every emission) is implemented and, per the user's 2026-08-24 directive and the 2026-08-25 promotion A/B, is now the DEFAULT controller seed (`one-shot-controller` in the policy crate's default features; building without default features restores the legacy table seed). + """ + target 2026-09-30 + note """ + 2026-08-25 update: open item (2), the saturated/banding corpus coverage gap, is closed on the GENERATION side at commit 6e13305 — the generator now provides 26 saturated and 25 gradient/banding-stress families (memo section 10 asks >= 25 per critical class), split-audited, sha256-idempotent, with the 13 locked holdout fixtures byte-unchanged (see @jpegxl-rs.observation.quality-corpus-critical-class-families-2026-08-25). The remaining steps on that item are the SSIMULACRA2 oracle-label sweeps over the 39 new families and a qpv2 retrain/gate on the widened coverage. Other open items stand: (1) the 12 MP wall anchor remains above the 2.0x matched-rate requirement (the quiet-host re-baseline in @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25 shows the residual is fixed render/metric cost, not search); (3) achieved-score dips above the floor exist in both arms (pre-existing bounded-search artifact); (4) 108 unswept ext-polart images remain for future label runs (resumable). Reproduce: quality_oracle_labels.py sweep/labels, quality_predictor_v2.py train, one_shot_promotion_ab.py; artifacts kept in .agent/scratch/one-shot-qpv2-20260824/. + """ + depends_on [ @jpegxl-rs.work.pqc-one-shot-pr0-hard-floor ] + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] + supersedes [ @jpegxl-rs.work.pqc-one-shot-controller/2 ] + source { + kind external + role origin + document "jpxl-one-shot-quality-controller-memo-2026-08-24" + use """ + Adopted the memo's verdict (common-case one shot, canonical verification as the correctness gate), its staged PR 1-5 sequence, its falsification and promotion gates, and its explicit non-goals. + """ + } +} + +record jpegxl-rs.work.pqc-one-shot-pr0-hard-floor/1 : work { + title "PR 0 of the one-shot program: enforce the hard score floor at the facade and CLI" + state completed + scope [ + path "JPXL/crates/jpxl-cli/**", + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl/**" + ] + intent """ + Implement decision quality-miss-fallback-semantics: refuse under-target perceptual results by default (Error::TargetNotMet with a structured QualityMiss), add the explicit Lossless and BestEffort fallbacks to the facade and --quality-fallback to the CLI, keep the CLI from creating an output file on refusal, preserve the failed search's quality trace for calibration, correct the module and budget documentation to state the bounded-finalist selection and the pixel_probes + 1 rescue accounting honestly, and pin the previously untested miss path with facade, policy-layer and CLI tests. All met-path codestreams must remain byte-identical. + """ + note """ + First slice of the 2026-08-24 one-shot advisor memo's plan; deliberately excludes any search-policy or predictor change. The wall/prediction program is tracked separately under pqc-one-shot-controller. + """ + implements [ @jpegxl-rs.decision.quality-miss-fallback-semantics ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] + author "GitHub Uploader" +} + record jpegxl-rs.work.pqc-pr0-provenance-corpus-calibration/1 : work { title "PQC PR 0: register sources, quality-guard corpus with splits, initial-rung calibration data" state completed @@ -7309,6 +7540,58 @@ record jpegxl-rs.work.pqc-pr5-policy-bank/1 : work { part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] } +record jpegxl-rs.work.pqc-pr7-holdout-closure/1 : work { + title "Close the OOM-interrupted PR7 Quality/reducer holdout sweep" + state completed + scope [ + path ".agent/scratch/pr7-holdout-closure-20260824/**", + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl-encode-policy/src/reducer.rs", + path "JPXL/crates/jpxl-perceptual/src/**", + path "JPXL/crates/jpxl/src/lib.rs", + path "JPXL/tools/**" + ] + intent """ + Resume only the 27 Quality cells left by the 2026-08-22 locked-holdout run, using the current low-memory implementation and serialized process caps. Recompute the complete 91-cell Quality/reducer and matched-score Contract B report. Keep Quality feature-gated unless the standing promotion contract passes; implement only fixes justified by the completed evidence. + """ + note """ + The prior run has 64/91 Quality cells plus 65 Balanced cells. The common-axis production quality/rate curve is already complete on 13/13 and is not rerun here. + """ + acceptance { + check full-holdout { + statement """ + The resumable locked-holdout runner covers all 91 Quality cells and matching Balanced curves, with zero canonical-floor or decoder failures and reducer work within its configured bound. + """ + method command + verified_by [ @jpegxl-rs.evidence.pqc-pr7-holdout-complete-2026-08-24/1 ] + } + check promotion-contract { + statement """ + The complete common-score byte ratio and Contract B guard metrics are reported, and Quality remains feature-gated unless every standing promotion condition passes. + """ + method observation + verified_by [ @jpegxl-rs.evidence.pqc-pr7-promotion-disposition-2026-08-24/1 ] + } + check resulting-disposition { + statement """ + Any defect exposed by the completed cells is fixed and tested; if no defect is exposed, the evidence records that no source-policy change is warranted. + """ + method manual + verified_by [ @jpegxl-rs.evidence.pqc-pr7-policy-gating-test-2026-08-24/1 ] + } + check workspace-gates { + statement """ + The required workspace build, release-test, strict-clippy, and formatting gates pass after any resulting changes. + """ + method command + command "cd JPXL && cargo build --workspace && cargo test --workspace --release && cargo clippy --workspace --all-targets -- -D warnings && cargo fmt --all --check" + verified_by [ @jpegxl-rs.evidence.pqc-pr7-closure-workspace-gates-2026-08-24/1 ] + } + } + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] +} + record jpegxl-rs.work.pqc-pr7-terminal-reducer/1 : work { title "PQC PR 7: finalist terminal-coefficient reducer exchanging measured score reserve for exact bytes, and the Quality-effort promotion gate" state abandoned @@ -7828,7 +8111,7 @@ record jpegxl-rs.work.pqc-usable-efforts-cost/4 : work { record jpegxl-rs.work.pqc-usable-efforts-cost/5 : work { title "PQC usable efforts: reduce Fast/Balanced wall time and peak memory" - state active + state superseded scope [ path "JPXL/crates/jpxl-encode-policy/src/quality.rs", path "JPXL/crates/jpxl-perceptual/src/**", @@ -7884,9 +8167,253 @@ record jpegxl-rs.work.pqc-usable-efforts-cost/5 : work { ] } +record jpegxl-rs.work.pqc-usable-efforts-cost/6 : work { + title "PQC usable efforts: reduce Fast/Balanced wall time and peak memory" + state superseded + scope [ + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl-perceptual/src/**", + path "JPXL/crates/jpxl-plan-render/src/**", + path "JPXL/crates/jpxl/src/lib.rs" + ] + intent """ + Profile and reduce the production Fast and Balanced perceptual path's full-frame render/metric allocation and rescue-probe cost. Work only on usable efforts: do not spend measurement time on the feature-gated Quality reference effort. Pure scorer, renderer, and lifetime changes must preserve Fast/Balanced codestream bytes; any deliberate search-policy change requires the standing Contract B screen. + """ + note """ + Memory, production identity, and workspace gates pass; wall remains the sole open check but moved twice in 2026-08-24/25. First, the one-shot controller promotion (see @jpegxl-rs.work.pqc-one-shot-controller) cut probe COUNT: 12 MP anchor matched-rate ratio 8.02x -> 6.31x on the sweep host. Second, per-probe cost fell at commit 9c6e64a: the serial varblock reconstruction and transfer-curve linearization now band over the executor with byte-identical output (see @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25) — release quality wall on the anchor -21%, ratio 3.11x -> 2.46x on the measuring host, peak memory unchanged. Still short of the <=2x target. Ranked remaining directions: (1) exact — parallel or overlapped scatter of varblock samples (~50-100 ms/probe); (2) exact but risky — fusing the metric's sequential blurs in low_memory mode (score-identity risk, not attempted); (3) score-changing, largest headroom — navigation on a reduced pyramid or downscaled reconstruction with full-resolution final verification only; both render and metric scale with pixels, but it changes navigation inputs and therefore selected streams, so it requires the standing Contract B promotion screen (interleaved A/B, floor violations, byte geomean, achieved-score inversions) before any default flip. + """ + acceptance { + check memory-12mp { + statement """ + Balanced q85 on the locked 12 MP anchor peaks at no more than 2.0 GB RSS, with the measurement serialized and process-capped. + """ + method observation + verified_by [ @jpegxl-rs.evidence.pqc-low-memory-12mp-2026-08-23/1 ] + } + check production-identity { + statement """ + Pure scorer/renderer/lifetime changes leave Fast and Balanced codestreams byte-identical across the locked identity cells, threads 1/4, and AVX2 on/off; any search-policy change instead passes the standing Contract B gate. + """ + method command + command "python3 .agent/scratch/pqc-cost-20260823/identity_current.py" + verified_by [ @jpegxl-rs.evidence.pqc-low-memory-production-identity-2026-08-23/1 ] + } + check wall-anchors { + statement """ + At matched achieved SSIMULACRA2, Balanced end-to-end wall is no more than 2.0x the rate path on both the 4.3 MP and 12 MP anchors; phase timings identify any remaining gap. + """ + method observation + } + check workspace-gates { + statement """ + The workspace release tests, clippy, and formatting gates pass. + """ + method command + command "cd JPXL && cargo test --workspace --release && cargo clippy --workspace --all-targets -- -D warnings && cargo fmt --all --check" + verified_by [ @jpegxl-rs.evidence.pqc-low-memory-workspace-gates-2026-08-23/1 ] + } + } + depends_on [ @jpegxl-rs.work.pqc-pr4-quality-navigator/1 ] + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] + supersedes [ @jpegxl-rs.work.pqc-usable-efforts-cost/5 ] + supported_by [ + @jpegxl-rs.evidence.pqc-low-memory-wall-anchors-2026-08-23/1, + @jpegxl-rs.evidence.pqc-pr4-holdout-wall-reported-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr4b-memory-12mp-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr4b-probe-speed-2026-08-22/1, + @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25 + ] +} + +record jpegxl-rs.work.pqc-usable-efforts-cost/7 : work { + title "PQC usable efforts: reduce Fast/Balanced wall time and peak memory" + state superseded + scope [ + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl-perceptual/src/**", + path "JPXL/crates/jpxl-plan-render/src/**", + path "JPXL/crates/jpxl/src/lib.rs" + ] + intent """ + Profile and reduce the production Fast and Balanced perceptual path's full-frame render/metric allocation and rescue-probe cost. Work only on usable efforts: do not spend measurement time on the feature-gated Quality reference effort. Pure scorer, renderer, and lifetime changes must preserve Fast/Balanced codestream bytes; any deliberate search-policy change requires the standing Contract B screen. + """ + note """ + 2026-08-25 quiet-host re-baseline (see @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25): memory-12mp and production-identity re-verified on the current binary; wall-anchors remains the sole open check and its honest numbers are 12 MP 5.42x (4 threads) / 4.25x (8) and 4.3 MP 3.88x / 3.22x — the comparable series at 12 MP/4t is 8.00x (2026-08-23) -> 5.42x (one-shot promotion cut probes 5->3); the earlier 2.46x figure divided by a load-inflated rate denominator and is not comparable. Direction (1) from rev 6 is closed: the varblock sample scatter is now band-parallel and exact (commit b0c128c) but measured wall-neutral at the anchor; the serial residual was ~30-45 ms/probe, not 50-100. Phase attribution shows the binding constraint: the quality path's fixed costs excluding all scored probes (~1.83 s at 8 threads: source load, reference precompute, plan, entropy, emit) already sit at 1.95x the rate path, and a perfect single-probe controller would land near 2.7x, so the 2.0x acceptance CANNOT be met by search-policy improvements alone. Remaining live directions: (a) exact metric/render per-pixel cost cuts of ~2-3x (blur fusion and SIMD in the metric, faster reference precompute; score-identity risk, unproven); (b) score-changing reduced-resolution navigation with full-resolution final verification (Contract B screen required; by itself lands ~2.5-2.9x); (c) renegotiate the wall-anchors target (e.g. <=2.5x at 8 threads, or define it against a fixed non-probe budget). Direction (c) is a decision for the operator; the check stays open rather than weakened unilaterally. + """ + acceptance { + check memory-12mp { + statement """ + Balanced q85 on the locked 12 MP anchor peaks at no more than 2.0 GB RSS, with the measurement serialized and process-capped. + """ + method observation + verified_by [ @jpegxl-rs.evidence.pqc-memory-12mp-2026-08-25/1 ] + } + check production-identity { + statement """ + Pure scorer/renderer/lifetime changes leave Fast and Balanced codestreams byte-identical across the locked identity cells, threads 1/4, and AVX2 on/off; any search-policy change instead passes the standing Contract B gate. + """ + method command + command "python3 .agent/scratch/pqc-scatter-20260825/identity_scatter.py" + verified_by [ @jpegxl-rs.evidence.pqc-scatter-production-identity-2026-08-25/1 ] + } + check wall-anchors { + statement """ + At matched achieved SSIMULACRA2, Balanced end-to-end wall is no more than 2.0x the rate path on both the 4.3 MP and 12 MP anchors; phase timings identify any remaining gap. + """ + method observation + } + check workspace-gates { + statement """ + The workspace release tests, clippy, and formatting gates pass. + """ + method command + command "cd JPXL && cargo test --workspace --release && cargo clippy --workspace --all-targets -- -D warnings && cargo fmt --all --check" + verified_by [ @jpegxl-rs.evidence.pqc-low-memory-workspace-gates-2026-08-23/1 ] + } + } + depends_on [ @jpegxl-rs.work.pqc-pr4-quality-navigator/1 ] + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] + supersedes [ @jpegxl-rs.work.pqc-usable-efforts-cost/6 ] + supported_by [ + @jpegxl-rs.evidence.pqc-low-memory-wall-anchors-2026-08-23/1, + @jpegxl-rs.evidence.pqc-pr4-holdout-wall-reported-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr4b-memory-12mp-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr4b-probe-speed-2026-08-22/1, + @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25, + @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25 + ] +} + +record jpegxl-rs.work.pqc-usable-efforts-cost/8 : work { + title "PQC usable efforts: reduce Fast/Balanced wall time and peak memory" + state superseded + scope [ + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl-perceptual/src/**", + path "JPXL/crates/jpxl-plan-render/src/**", + path "JPXL/crates/jpxl/src/lib.rs" + ] + intent """ + Profile and reduce the production Fast and Balanced perceptual path's full-frame render/metric allocation and rescue-probe cost. Work only on usable efforts: do not spend measurement time on the feature-gated Quality reference effort. Pure scorer, renderer, and lifetime changes must preserve Fast/Balanced codestream bytes; any deliberate search-policy change requires the standing Contract B screen. + """ + note """ + 2026-08-25 direction (a) executed in part (commit c0253b3, exact metric-path cost cuts): the vertical blur pass writes column strips into the output in place instead of per-strip buffers plus a serial scatter; the horizontal blur pass runs four rows as independent recursion lanes (AVX2-dispatched); redundant scratch re-zeroing is dropped where every element is provably overwritten; from_srgb16 uses a 65536-entry transfer table. All exact: 6/6 A/B cells byte-identical vs the pre-change binary, 52/52 locked matrix identical, 12 MP bytes unchanged. Measured effect on the standing interleaved recipe (load_1m 1.8-4.5 vs 5.2-7.7 for the prior run; ratios are the load-robust comparison): 4.3 MP 3.87x -> 3.40x (4 threads) and 3.22x -> 2.76x (8); 12 MP 5.42x -> 5.07x and 4.25x -> 3.50x. 12 MP peak RSS 2,043,056 -> 2,008,188 KiB. Pre-change attribution at 12 MP/8t (~513 ms per probe: render 195, linearize 18, score 290 of which scale-0 candidate blurs ~160, pool ~28, convert ~17; reference precompute 285 ms; retention is Moments up to MOMENTS_PIXEL_CAP = 24 MP, so no redundant reference re-blur exists at the anchors). Remaining exact headroom is smaller and identified: probe render (~195 ms/probe), pool maps, XYB convert, and reference precompute; a quiet-host re-baseline is still needed for honest absolute walls. Rev 7's structural conclusion stands: fixed costs keep the 2.0x wall-anchors target out of reach of search-policy improvements alone, and the remaining live directions are (a) further exact per-pixel cuts (diminishing), (b) Contract-B-screened reduced-resolution navigation, (c) an operator renegotiation of the target; the check stays open rather than weakened unilaterally. + """ + acceptance { + check memory-12mp { + statement """ + Balanced q85 on the locked 12 MP anchor peaks at no more than 2.0 GB RSS, with the measurement serialized and process-capped. + """ + method observation + verified_by [ @jpegxl-rs.evidence.pqc-metric-cuts-memory-12mp-2026-08-25/1 ] + } + check production-identity { + statement """ + Pure scorer/renderer/lifetime changes leave Fast and Balanced codestreams byte-identical across the locked identity cells, threads 1/4, and AVX2 on/off; any search-policy change instead passes the standing Contract B gate. + """ + method command + command "python3 .agent/scratch/pqc-metric-cuts-20260825/identity_scatter.py" + verified_by [ @jpegxl-rs.evidence.pqc-metric-cuts-production-identity-2026-08-25/1 ] + } + check wall-anchors { + statement """ + At matched achieved SSIMULACRA2, Balanced end-to-end wall is no more than 2.0x the rate path on both the 4.3 MP and 12 MP anchors; phase timings identify any remaining gap. + """ + method observation + } + check workspace-gates { + statement """ + The workspace release tests, clippy, and formatting gates pass. + """ + method command + command "cd JPXL && cargo test --workspace --release && cargo clippy --workspace --all-targets -- -D warnings && cargo fmt --all --check" + verified_by [ @jpegxl-rs.evidence.pqc-metric-cuts-workspace-gates-2026-08-25/1 ] + } + } + depends_on [ @jpegxl-rs.work.pqc-pr4-quality-navigator/1 ] + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] + supersedes [ @jpegxl-rs.work.pqc-usable-efforts-cost/7 ] + supported_by [ + @jpegxl-rs.evidence.pqc-low-memory-wall-anchors-2026-08-23/1, + @jpegxl-rs.evidence.pqc-metric-cuts-wall-anchors-2026-08-25/1, + @jpegxl-rs.evidence.pqc-pr4-holdout-wall-reported-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr4b-memory-12mp-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr4b-probe-speed-2026-08-22/1, + @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25, + @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25 + ] +} + +record jpegxl-rs.work.pqc-usable-efforts-cost/9 : work { + title "PQC usable efforts: reduce Fast/Balanced wall time and peak memory" + state active + scope [ + path "JPXL/crates/jpxl-encode-policy/src/quality.rs", + path "JPXL/crates/jpxl-perceptual/src/**", + path "JPXL/crates/jpxl-plan-render/src/**", + path "JPXL/crates/jpxl/src/lib.rs" + ] + intent """ + Profile and reduce the production Fast and Balanced perceptual path's full-frame render/metric allocation and rescue-probe cost. Work only on usable efforts: do not spend measurement time on the feature-gated Quality reference effort. Pure scorer, renderer, and lifetime changes must preserve Fast/Balanced codestream bytes; any deliberate search-policy change requires the standing Contract B screen. + """ + note """ + 2026-08-25 direction (a), second round (commit c23bc00), now guided by a real cycle profile after the operator enabled perf: __powf_fma was 5.4% of cycles (a per-sample sRGB encode in the probe render whose result the quantizer immediately consumed, plus the 16-bit prep transfer), and the two-step linearize pass another ~3%. The probe render now produces depth-quantized linear planes directly: levels come from thresholds bisected over the f32 bit lattice against the actual linear_to_srgb->scale->round->clamp composition, guide-table accelerated and cached per bit depth; from_srgb16_with uses a 65536-entry transfer table. Exact: 6/6 A/B vs the pre-change binary, 52/52 locked matrix, a dedicated classifier-equivalence unit test. Quiet-host 12 MP 8-thread quality wall 2.39-2.53 s (2.62 s before the fusion, ~3.99 s before the day's first round). Interleaved ratios across the day: 4.3 MP 3.87x -> 3.40x -> 3.06x (4t) and 3.22x -> 2.76x -> 2.50x (8t); 12 MP 5.42x -> 5.07x -> 4.65x (4t); the round-2 12 MP 8t schedule cell was load-polluted (see the scratch README). Remaining hot leaves by cycle share: blur ~24%, EPF ~10%, pool ~7% (order-locked f64 summation, not vectorisable exactly), XYB convert ~7%, plan-phase quantize ~6%. The 4.3 MP 8-thread anchor now sits at 2.50x; the 2.0x target on both anchors still cannot be met by exact cuts alone (12 MP fixed costs dominate), so the live directions remain (a) further exact cuts with visibly diminishing headroom, (b) Contract-B-screened reduced-resolution navigation, (c) an operator renegotiation of the target; the check stays open rather than weakened unilaterally. + """ + acceptance { + check memory-12mp { + statement """ + Balanced q85 on the locked 12 MP anchor peaks at no more than 2.0 GB RSS, with the measurement serialized and process-capped. + """ + method observation + verified_by [ @jpegxl-rs.evidence.pqc-fused-render-memory-12mp-2026-08-25/1 ] + } + check production-identity { + statement """ + Pure scorer/renderer/lifetime changes leave Fast and Balanced codestreams byte-identical across the locked identity cells, threads 1/4, and AVX2 on/off; any search-policy change instead passes the standing Contract B gate. + """ + method command + command "python3 .agent/scratch/pqc-metric-cuts-20260825/identity_scatter.py" + verified_by [ + @jpegxl-rs.evidence.pqc-fused-render-production-identity-2026-08-25/1 + ] + } + check wall-anchors { + statement """ + At matched achieved SSIMULACRA2, Balanced end-to-end wall is no more than 2.0x the rate path on both the 4.3 MP and 12 MP anchors; phase timings identify any remaining gap. + """ + method observation + } + check workspace-gates { + statement """ + The workspace release tests, clippy, and formatting gates pass. + """ + method command + command "cd JPXL && cargo test --workspace --release && cargo clippy --workspace --all-targets -- -D warnings && cargo fmt --all --check" + verified_by [ @jpegxl-rs.evidence.pqc-fused-render-workspace-gates-2026-08-25/1 ] + } + } + depends_on [ @jpegxl-rs.work.pqc-pr4-quality-navigator/1 ] + implements [ @jpegxl-rs.decision.perceptual-quality-contract/1 ] + part_of [ @jpegxl-rs.track.perceptual-quality-controller/1 ] + supersedes [ @jpegxl-rs.work.pqc-usable-efforts-cost/8 ] + supported_by [ + @jpegxl-rs.evidence.pqc-fused-render-wall-anchors-2026-08-25/1, + @jpegxl-rs.evidence.pqc-low-memory-wall-anchors-2026-08-23/1, + @jpegxl-rs.evidence.pqc-metric-cuts-wall-anchors-2026-08-25/1, + @jpegxl-rs.evidence.pqc-pr4-holdout-wall-reported-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr4b-memory-12mp-2026-08-22/1, + @jpegxl-rs.evidence.pqc-pr4b-probe-speed-2026-08-22/1, + @jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25, + @jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25 + ] +} + record jpegxl-rs.work.publish-readme-benchmark-mit/1 : work { title "Publish-ready README, reproducible libjxl comparison, and MIT-only licensing" - state proposed + state completed scope [ path "AGENTS.md", path "JPEG_XL_CLEAN_IMPLEMENTATION_LESSONS.md", @@ -7909,24 +8436,28 @@ record jpegxl-rs.work.publish-readme-benchmark-mit/1 : work { A documented command produces a quality-and-speed comparison using JPXL and cjxl/djxl on the same inputs with recorded provenance. """ method command + verified_by [ @jpegxl-rs.evidence.bench-vs-libjxl-reproducible-2026-08-25/1 ] } check mit-only { statement """ Workspace package metadata and repository licensing surface declare MIT-only terms. """ method command + verified_by [ @jpegxl-rs.evidence.mit-only-licensing-audit-2026-08-25/1 ] } check readme-current-state { statement """ README accurately describes the supported scope, known exclusions, clean-room policy, build instructions, and the recorded benchmark. """ method manual + verified_by [ @jpegxl-rs.evidence.readme-current-state-2026-08-25/1 ] } check workspace-gates { statement """ The required workspace build, test, lint, and formatting gates pass. """ method command + verified_by [ @jpegxl-rs.evidence.workspace-gates-2026-08-25/1 ] } } } @@ -8068,7 +8599,9 @@ record jpegxl-rs.work.quality-q2-chroma-hf/1 : work { } depends_on [ @jpegxl-rs.work.quality-q1-quantizer/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.quality-q3-ladder-ceiling-and-allocation-screens/1 : work { @@ -8139,7 +8672,9 @@ record jpegxl-rs.work.quality-q3-ladder-ceiling-and-allocation-screens/1 : work } depends_on [ @jpegxl-rs.work.quality-q2-chroma-hf/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.quality-q4-cover-rate-model/1 : work { @@ -8203,7 +8738,9 @@ record jpegxl-rs.work.quality-q4-cover-rate-model/1 : work { } depends_on [ @jpegxl-rs.work.quality-q3-ladder-ceiling-and-allocation-screens/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.quality-q5-anchored-controller-above-ceiling/1 : work { @@ -8239,7 +8776,9 @@ record jpegxl-rs.work.quality-q5-anchored-controller-above-ceiling/1 : work { } depends_on [ @jpegxl-rs.work.quality-q4-cover-rate-model/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.quality-q6-cheaper-exhaustive-search/1 : work { @@ -8282,7 +8821,9 @@ record jpegxl-rs.work.quality-q6-cheaper-exhaustive-search/1 : work { } depends_on [ @jpegxl-rs.work.quality-q5-anchored-controller-above-ceiling/1 ] part_of [ @jpegxl-rs.track.encoder-optimization/1 ] - source { kind legacy } + source { + kind legacy + } } record jpegxl-rs.work.quality-q7-effective-scale-controller/1 : work { diff --git a/.gitignore b/.gitignore index d1973c29..9498729d 100644 --- a/.gitignore +++ b/.gitignore @@ -37,4 +37,10 @@ target/ .agent/scratch/ # Source-export bundle produced by copy-code-only.py (regenerable). -JPXL/JPXL.zip +JPXL/JPXL.zip + +# Python bytecode caches from the tools/ test suites. +__pycache__/ + +# Agent-pack snapshot archives (regenerable exports, never history). +/jpegXL-rs-agent-pack-*.7z diff --git a/.mcp.json b/.mcp.json index d7314938..4021a6f2 100644 --- a/.mcp.json +++ b/.mcp.json @@ -8,7 +8,7 @@ "codegraph": { "type": "stdio", "command": "codegraph", - "args": ["serve", "--mcp", "--path", "${workspaceFolder}"] + "args": ["serve", "--mcp"] } } } diff --git a/JPXL/Cargo.lock b/JPXL/Cargo.lock index b908f453..9a33910b 100644 --- a/JPXL/Cargo.lock +++ b/JPXL/Cargo.lock @@ -357,6 +357,10 @@ dependencies = [ "jpxl-core", ] +[[package]] +name = "jpxl-jpeg" +version = "0.3.0" + [[package]] name = "jpxl-perceptual" version = "0.3.0" diff --git a/JPXL/Cargo.toml b/JPXL/Cargo.toml index 2338cb30..75560972 100644 --- a/JPXL/Cargo.toml +++ b/JPXL/Cargo.toml @@ -5,6 +5,7 @@ members = [ "crates/jpxl-bitstream", "crates/jpxl-core", "crates/jpxl-entropy", + "crates/jpxl-jpeg", "crates/jpxl-decode", "crates/jpxl-encode", "crates/jpxl-encode-policy", @@ -75,6 +76,16 @@ debug = true lto = "thin" codegen-units = 1 +# The shipping profile: `release` plus full (fat) LTO. Build it through +# `tools/build-release-final.sh`, which wraps it in the measured PGO cycle +# (instrument → train on the canonical test-set images → rebuild with the +# profile); PGO alone was measured at -36.65% wall on 2400x1800 and -53.48% +# on 4000x3000 target-rate encodes. `release` (thin LTO) remains the profile +# for benchmarks and promoted timings so historical numbers stay comparable. +[profile.release-final] +inherits = "release" +lto = "fat" + # Optimised like `release` but built for iteration speed: no LTO, sixteen # codegen units, incremental, and with debug assertions and overflow checks # kept ON so the `debug_assert!` shape guards in the codec still fire under diff --git a/JPXL/crates/jpxl-cli/Cargo.toml b/JPXL/crates/jpxl-cli/Cargo.toml index b937fa85..fecaeec4 100644 --- a/JPXL/crates/jpxl-cli/Cargo.toml +++ b/JPXL/crates/jpxl-cli/Cargo.toml @@ -28,6 +28,9 @@ simd = ["jpxl-encode/simd", "jpxl-encode-policy/simd", "jpxl-core/simd"] # Compatibility feature for explicitly forwarding Fast rate-preset support. # Policy enables it normally; Quality remains the runtime default. anchor-sketch = ["jpxl-encode-policy/anchor-sketch"] +# One-shot program PR 5: forwards the policy crate's feature-gated +# common-case one-shot start to the binary, for promotion A/B runs. +one-shot-controller = ["jpxl-encode-policy/one-shot-controller"] # Exposes the exhaustive-reference `quality` lossy effort on `--effort` / # `--lossy-preset`. Forwards to `jpxl/quality-effort`; off by default. quality-effort = ["jpxl/quality-effort"] diff --git a/JPXL/crates/jpxl-cli/src/main.rs b/JPXL/crates/jpxl-cli/src/main.rs index 1548c904..74e29425 100644 --- a/JPXL/crates/jpxl-cli/src/main.rs +++ b/JPXL/crates/jpxl-cli/src/main.rs @@ -47,8 +47,17 @@ Usage: jpxl analyze-atlas Export the diagnostic AnalysisAtlasV2; this research command does not affect encoding - jpxl features [--json] Print the quality controller's frame source - features as one JSON line (calibration tool) + jpxl features [--json] [--transform-summary] + Print the quality controller's frame source + features as one JSON line (calibration tool); + --transform-summary adds the DCT8-derived + transform features + jpxl quality-ladder --scales s1,s2,... [--effort fast|balanced] [--price] + [--threads N] + Build the production quality pixel plan + fresh at each effective scale, score each + canonically, optionally exact-price it, and + print JSONL (oracle-label calibration tool) jpxl bench [opts] Time one encode path (see `jpxl bench --help`) jpxl --help Show this message jpxl --version Show the version @@ -80,6 +89,13 @@ Lossy options (8- or 16-bit RGB; any one selects the VarDCT path): effort's default (fast 70, balanced 85). This is the normal way to ask for lossy output. --lossy --quality with the effort's default score. + --quality-fallback What to emit when the bounded search cannot + verify the requested score: lossless (a + mathematically lossless stream) or + best-effort (the finest verified under-target + stream, reported by its true score). Without + this flag such an encode fails, exits 1, and + writes nothing. --effort Lossy effort fast|balanced (search-latency budget, also picks the --quality default), or a digit 1..9 for lossless Modular effort. @@ -254,6 +270,7 @@ fn run(args: &[String]) -> u8 { "compare" => cmd_compare(rest), "analyze-atlas" => cmd_analyze_atlas(rest), "features" => cmd_features(rest), + "quality-ladder" => cmd_quality_ladder(rest), "bench" => cmd_bench(rest), other => { fail(&format!("unknown command `{other}`")); @@ -474,6 +491,9 @@ fn cmd_encode(args: &[String]) -> u8 { // `Some(score)` is an explicit score, inner `None` means "use the effort's // default score". let mut quality: Option> = None; + // What `--quality` emits when the bounded search cannot verify the score; + // `Refuse` (fail, write nothing) unless `--quality-fallback` says otherwise. + let mut quality_fallback = jpxl::QualityFallback::Refuse; // Fixed-quantizer expert mode. let mut global_scale: Option = None; // Lossy effort (search-latency budget); also picks the `--quality` default. @@ -524,6 +544,20 @@ fn cmd_encode(args: &[String]) -> u8 { } } "--lossy" => quality = Some(None), + "--quality-fallback" => { + let Some(mode) = rest.next() else { + fail("`--quality-fallback` needs `lossless` or `best-effort`"); + return EXIT_ERROR; + }; + quality_fallback = match mode.as_str() { + "lossless" => jpxl::QualityFallback::Lossless, + "best-effort" => jpxl::QualityFallback::BestEffort, + _ => { + fail("`--quality-fallback` needs `lossless` or `best-effort`"); + return EXIT_ERROR; + } + }; + } "--global-scale" => { let Some(value) = rest.next().and_then(|v| v.parse::().ok()) else { fail("`--global-scale` needs a positive representable integer"); @@ -924,7 +958,7 @@ fn cmd_encode(args: &[String]) -> u8 { let mut mode_label: Option = None; let encoded = if let Some(explicit) = quality { let score = explicit.unwrap_or_else(|| lossy_effort.default_score()); - match encode_quality(&image, score, &options, lossy_effort) { + match encode_quality(&image, score, &options, lossy_effort, quality_fallback) { Ok((bytes, line, mode)) => { perceptual_line = Some(line); mode_label = Some(mode); @@ -2133,13 +2167,30 @@ fn cmd_analyze_atlas(args: &[String]) -> u8 { /// `jpxl features [--json]`: print the quality controller's frame /// source features as one JSON line, for the initial-rung calibration tooling. fn cmd_features(args: &[String]) -> u8 { - let input = match args { - [input] => input, - [input, flag] | [flag, input] if flag == "--json" => input, - _ => { - fail("`features` takes an input raster and an optional `--json` flag"); - return EXIT_ERROR; + let mut input: Option<&String> = None; + let mut transform_summary = false; + for arg in args { + match arg.as_str() { + "--json" => {} + "--transform-summary" => transform_summary = true, + other if other.starts_with("--") => { + fail( + "`features` takes an input raster and optional `--json` / `--transform-summary` flags", + ); + return EXIT_ERROR; + } + _ if input.is_none() => input = Some(arg), + _ => { + fail("`features` takes exactly one input raster"); + return EXIT_ERROR; + } } + } + let Some(input) = input else { + fail( + "`features` takes an input raster and optional `--json` / `--transform-summary` flags", + ); + return EXIT_ERROR; }; let bytes = match read_path(input) { Ok(bytes) => bytes, @@ -2169,7 +2220,224 @@ fn cmd_features(args: &[String]) -> u8 { image.height(), frame.is_grayscale(), ); - println!("{}", features.to_json()); + if transform_summary { + let mut request = jpxl_encode_policy::EncodeRequest::for_quality( + jpxl_encode_policy::RateSearchPreset::Balanced, + ); + request.bits_per_sample = image.bits_per_sample(); + match jpxl_encode_policy::transform_feature_summary(&frame, &request) { + Ok(summary) => println!( + "{{\"source_features\":{},\"transform_features\":{}}}", + features.to_json(), + summary.to_json() + ), + Err(error) => { + fail(&format!("{input}: {error}")); + return EXIT_ERROR; + } + } + } else { + println!("{}", features.to_json()); + } + EXIT_OK +} + +/// `jpxl quality-ladder --scales s1,s2,... [--effort fast|balanced] [--price] +/// [--threads N] `: the one-shot program's oracle-label sweep. +/// +/// Builds the production quality pixel plan fresh at every requested +/// effective scale, scores each reconstruction canonically, optionally +/// exact-prices it, and prints one `jpxl.quality-ladder/1` JSONL record per +/// point after a header record carrying the source features. No navigation +/// and no output file: this is measurement for the offline crossing trainer +/// (`tools/quality_oracle_labels.py`), not an encoder mode. +fn cmd_quality_ladder(args: &[String]) -> u8 { + let mut scales: Vec = Vec::new(); + let mut effort = jpxl::Effort::Balanced; + let mut price = false; + let mut threads: Option = None; + let mut positional: Vec<&String> = Vec::new(); + let mut rest = args.iter(); + while let Some(arg) = rest.next() { + match arg.as_str() { + "--scales" => { + let Some(list) = rest.next() else { + fail("`--scales` needs a comma-separated list of effective scales"); + return EXIT_ERROR; + }; + for part in list.split(',') { + match part.trim().parse::() { + Ok(scale) if scale >= 1 => scales.push(scale), + _ => { + fail("`--scales` entries must be positive integers"); + return EXIT_ERROR; + } + } + } + } + "--effort" => { + let Some(mode) = rest.next() else { + fail("`--effort` needs fast or balanced"); + return EXIT_ERROR; + }; + effort = match mode.as_str() { + "fast" => jpxl::Effort::Fast, + "balanced" => jpxl::Effort::Balanced, + _ => { + fail("`--effort` needs fast or balanced"); + return EXIT_ERROR; + } + }; + } + "--price" => price = true, + "--threads" => { + let Some(value) = rest.next().and_then(|v| v.parse::().ok()) else { + fail("`--threads` needs a positive worker count"); + return EXIT_ERROR; + }; + if value == 0 { + fail("`--threads` needs a positive worker count"); + return EXIT_ERROR; + } + threads = Some(value); + } + other if other.starts_with("--") => { + fail(&format!("unknown quality-ladder option `{other}`")); + return EXIT_ERROR; + } + _ => positional.push(arg), + } + } + let [input] = positional.as_slice() else { + fail("`quality-ladder` takes exactly one input raster"); + return EXIT_ERROR; + }; + if scales.is_empty() { + fail("`quality-ladder` needs `--scales s1,s2,...`"); + return EXIT_ERROR; + } + if scales.len() > 4096 { + fail("`quality-ladder` caps a sweep at 4096 points"); + return EXIT_ERROR; + } + + let bytes = match read_path(input) { + Ok(bytes) => bytes, + Err(error) => { + fail(&format!("{input}: {error}")); + return EXIT_ERROR; + } + }; + let image = match image_io::decode_input(&bytes, None) { + Ok(image) => image, + Err(error) => { + fail(&format!("{input}: {error}")); + return EXIT_ERROR; + } + }; + let (width, height, bits_per_sample, rgb) = match image_to_rgb16(&image) { + Ok(parts) => parts, + Err(error) => { + fail(&format!("{input}: {error}")); + return EXIT_ERROR; + } + }; + if width < jpxl_perceptual::MIN_DIMENSION || height < jpxl_perceptual::MIN_DIMENSION { + fail(&format!( + "{input}: below the perceptual metric's {0}x{0} floor", + jpxl_perceptual::MIN_DIMENSION + )); + return EXIT_ERROR; + } + + // Map requested effective scales onto ladder rungs, ascending, deduped. + let mut rungs: Vec = scales + .iter() + .map(|&scale| jpxl_encode_policy::rung_for_effective_scale(scale)) + .collect(); + rungs.sort_unstable(); + rungs.dedup(); + + let mut request = jpxl_encode_policy::EncodeRequest::for_quality(effort.into()); + if let Some(threads) = threads { + request.resources = jpxl_encode::EncodeResources::groups(threads); + } + request.bits_per_sample = bits_per_sample; + let executor = request.resources.executor(); + let frame = match jpxl_encode_policy::PreparedFrame::from_srgb16_with( + width, + height, + &rgb, + bits_per_sample, + Some(&executor), + ) { + Ok(frame) => frame, + Err(error) => { + fail(&format!("{input}: {error}")); + return EXIT_ERROR; + } + }; + let mut evaluator = match jpxl_perceptual::PlanRenderEvaluator::from_srgb16( + width, + height, + &rgb, + bits_per_sample, + &executor, + ) { + Ok(evaluator) => evaluator, + Err(_) => { + fail(&format!( + "{input}: the frame cannot be scored by the perceptual metric" + )); + return EXIT_ERROR; + } + }; + let atlas = jpxl_encode_policy::AnalysisAtlas::analyze(&frame); + let features = jpxl_encode_policy::source_features(&atlas, width, height, frame.is_grayscale()); + use jpxl_encode_policy::PerceptualEvaluator as _; + println!( + "{{\"schema\":\"jpxl.quality-ladder/1\",\"input\":\"{}\",\"width\":{width},\ + \"height\":{height},\"bit_depth\":{bits_per_sample},\"effort\":\"{}\",\ + \"metric_version\":\"{}\",\"price\":{price},\"points\":{},\"source_features\":{}}}", + input.replace('\\', "/"), + effort_name(effort), + evaluator.metric_version(), + rungs.len(), + features.to_json(), + ); + let points = match jpxl_encode_policy::sweep_frame_perceptual( + &frame, + &atlas, + &request, + &rungs, + price, + &mut evaluator, + &executor, + ) { + Ok(points) => points, + Err(error) => { + fail(&format!("{input}: {error}")); + return EXIT_ERROR; + } + }; + for p in points { + println!( + "{{\"rung\":{},\"global_scale\":{},\"hf_mul\":{},\"quant_lf\":{},\ + \"effective_scale\":{},\"score\":{},\"bytes\":{},\"plan_ms\":{},\ + \"render_metric_ms\":{},\"price_ms\":{}}}", + p.rung.get(), + p.quantizer.global_scale.get(), + p.quantizer.hf_mul.get(), + p.quantizer.quant_lf.get(), + p.effective_scale, + p.score, + p.exact_bytes + .map_or_else(|| "null".to_owned(), |b| format!("{b}")), + p.plan_ms, + p.render_metric_ms, + p.price_ms, + ); + } EXIT_OK } @@ -2395,6 +2663,7 @@ fn perceptual_status_str(status: jpxl::PerceptualStatus) -> &'static str { jpxl::PerceptualStatus::UnderTargetWorkCap => "under_target_work_cap", jpxl::PerceptualStatus::RescuedFreshStructure => "rescued_fresh_structure", jpxl::PerceptualStatus::RoutedToLossless => "routed_to_lossless", + jpxl::PerceptualStatus::FallbackLossless => "fallback_lossless", jpxl::PerceptualStatus::UnsupportedTooSmall => "unsupported_too_small", } } @@ -2451,12 +2720,15 @@ fn image_to_rgb16(image: &jpxl_encode::Image) -> Result<(u32, u32, u32, Vec /// /// Returns the codestream, the perceptual report line, and the summary mode /// string. A score of 100 routes to the lossless encoder; anything lower runs -/// the quality controller. +/// the quality controller. When the controller cannot verify the score and +/// `fallback` is [`jpxl::QualityFallback::Refuse`], this returns `Err` and +/// the caller writes nothing. fn encode_quality( image: &jpxl_encode::Image, score: f64, options: &jpxl_encode::EncodeOptions, effort: jpxl::Effort, + fallback: jpxl::QualityFallback, ) -> Result<(Vec, String, String), String> { let (width, height, bits_per_sample, rgb) = image_to_rgb16(image)?; let encoder = jpxl::Encoder::new() @@ -2464,37 +2736,50 @@ fn encode_quality( .with_container(options.container) .with_jxlp_fragment_size(options.jxlp_fragment_size) .with_effort(effort) + .with_quality_fallback(fallback) .with_ssimulacra2_score(score) .map_err(|error| error.to_string())?; match encoder.encode_rgb16_reported(width, height, bits_per_sample, &rgb) { Ok((bytes, jpxl::EncodeReport::Perceptual(outcome))) => { let line = format_perceptual_line(&outcome, effort_name(effort)); - // `JPXL_QUALITY_TRACE=` appends the controller's - // `jpxl.quality-trace/1` record, the harness's and the - // predictor calibration's input. - if let (Some(path), Some(trace)) = ( - std::env::var_os("JPXL_QUALITY_TRACE"), - outcome.trace_json.as_deref(), - ) { - use std::io::Write as _; - let appended = std::fs::OpenOptions::new() - .create(true) - .append(true) - .open(&path) - .and_then(|mut file| writeln!(file, "{trace}")); - if let Err(error) = appended { - eprintln!("warning: could not write JPXL_QUALITY_TRACE: {error}"); - } - } - let mode = format!("lossy VarDCT (perceptual), ssimulacra2>={score:.4}"); + append_quality_trace(outcome.trace_json.as_deref()); + let mode = if outcome.status == jpxl::PerceptualStatus::FallbackLossless { + format!("lossless Modular (fallback: ssimulacra2>={score:.4} unmet)") + } else { + format!("lossy VarDCT (perceptual), ssimulacra2>={score:.4}") + }; Ok((bytes, line, mode)) } Ok(_) => Err("perceptual encode produced an unexpected report".to_owned()), Err(jpxl::Error::Unsupported(what)) => Err(format!("ssimulacra2>={score:.4}: {what}")), + // A refused under-target encode still ran a full search; its trace is + // calibration input, so the harness sees it even though nothing is + // written. + Err(jpxl::Error::TargetNotMet(miss)) => { + append_quality_trace(miss.trace_json.as_deref()); + Err(jpxl::Error::TargetNotMet(miss).to_string()) + } Err(error) => Err(error.to_string()), } } +/// Appends one `jpxl.quality-trace/2` record to the `JPXL_QUALITY_TRACE` +/// path, when both exist. The harness's and the predictor calibration's +/// input; a write failure warns and never fails the encode. +fn append_quality_trace(trace: Option<&str>) { + if let (Some(path), Some(trace)) = (std::env::var_os("JPXL_QUALITY_TRACE"), trace) { + use std::io::Write as _; + let appended = std::fs::OpenOptions::new() + .create(true) + .append(true) + .open(&path) + .and_then(|mut file| writeln!(file, "{trace}")); + if let Err(error) = appended { + eprintln!("warning: could not write JPXL_QUALITY_TRACE: {error}"); + } + } +} + /// Wraps a lossy codestream the way [`jpxl_encode::encode`] wraps a lossless /// one, so `--container` and `--jxlp` reach the VarDCT paths too. /// diff --git a/JPXL/crates/jpxl-cli/tests/cli_quality.rs b/JPXL/crates/jpxl-cli/tests/cli_quality.rs index 073872af..dbfd1b15 100644 --- a/JPXL/crates/jpxl-cli/tests/cli_quality.rs +++ b/JPXL/crates/jpxl-cli/tests/cli_quality.rs @@ -121,6 +121,106 @@ fn conflicting_targets_rejected() { assert!(stderr.contains("one lossy target"), "{stderr}"); } +/// The 64x64 gradient tops out below 99.5 on this metric, so `--quality +/// 99.5 --effort fast` is a deterministic miss. Without a fallback the CLI +/// must fail, exit 1, and leave no output file behind. +#[test] +fn an_unmet_quality_refuses_by_default_and_writes_nothing() { + let (input, output) = fixture("quality_refused", 64, 64); + let out = encode( + &input, + &output, + &["--effort", "fast", "--quality", "99.5", "--threads", "1"], + ); + assert_eq!(out.status.code(), Some(1)); + let stderr = String::from_utf8_lossy(&out.stderr); + assert!( + stderr.contains("quality target not met"), + "the refusal names itself: {stderr}" + ); + assert!( + !output.exists(), + "a refused encode must not create the output file" + ); +} + +#[test] +fn quality_fallback_lossless_emits_and_says_so() { + let (input, output) = fixture("quality_fb_lossless", 64, 64); + let out = encode( + &input, + &output, + &[ + "--effort", + "fast", + "--quality", + "99.5", + "--quality-fallback", + "lossless", + "--threads", + "1", + ], + ); + assert_eq!( + out.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&out.stderr) + ); + let stdout = String::from_utf8_lossy(&out.stdout); + assert!( + stdout.contains("status=fallback_lossless") && stdout.contains("achieved=100.0000"), + "the fallback is reported: {stdout}" + ); + assert!(output.exists() && output.metadata().map(|m| m.len()).unwrap_or(0) > 0); + let info = run(&["info", output.to_str().expect("utf8 path")]); + assert_eq!(info.status.code(), Some(0), "output is a JPEG XL stream"); +} + +#[test] +fn quality_fallback_best_effort_emits_the_under_target_stream() { + let (input, output) = fixture("quality_fb_best_effort", 64, 64); + let out = encode( + &input, + &output, + &[ + "--effort", + "fast", + "--quality", + "99.5", + "--quality-fallback", + "best-effort", + "--threads", + "1", + ], + ); + assert_eq!( + out.status.code(), + Some(0), + "{}", + String::from_utf8_lossy(&out.stderr) + ); + let stdout = String::from_utf8_lossy(&out.stdout); + assert!( + stdout.contains("status=saturated_top"), + "the true terminal status is reported: {stdout}" + ); + assert!(output.exists() && output.metadata().map(|m| m.len()).unwrap_or(0) > 0); +} + +#[test] +fn a_bad_quality_fallback_value_is_rejected() { + let (input, output) = fixture("quality_fb_bad", 64, 64); + let out = encode( + &input, + &output, + &["--quality", "--quality-fallback", "nope"], + ); + assert_eq!(out.status.code(), Some(1)); + let stderr = String::from_utf8_lossy(&out.stderr); + assert!(stderr.contains("`--quality-fallback` needs"), "{stderr}"); +} + #[test] fn global_scale_encodes() { let (input, output) = fixture("global_scale", 64, 64); diff --git a/JPXL/crates/jpxl-encode-policy/Cargo.toml b/JPXL/crates/jpxl-encode-policy/Cargo.toml index 79b674e9..cf97091e 100644 --- a/JPXL/crates/jpxl-encode-policy/Cargo.toml +++ b/JPXL/crates/jpxl-encode-policy/Cargo.toml @@ -19,7 +19,7 @@ jpxl-entropy.workspace = true wide = { workspace = true, optional = true } [features] -default = ["parallel", "simd", "anchor-sketch", "g5-bounded-entropy"] +default = ["parallel", "simd", "anchor-sketch", "g5-bounded-entropy", "one-shot-controller"] parallel = ["jpxl-encode/parallel"] # Leaf SIMD for HF quantize choose (and transitive DCT/Gaborish). simd = ["dep:wide", "jpxl-encode/simd", "jpxl-core/simd"] @@ -31,6 +31,16 @@ anchor-sketch = [] # alternative ranked from the near-target anchor and finalist censuses. The # feature remains explicit as a byte-identical-off control after promotion. g5-bounded-entropy = [] +# One-shot program PR 5: the common-case one-shot start — the generated +# crossing model's risk-adjusted candidate (or, when uncertain, its median) +# becomes the first fresh plan and its slope prior steers the first +# correction, with the exact navigator continuing under the same total caps +# and canonical verification before every emission. Default since the +# 2026-08-24 promotion A/B: zero floor violations on 441 never-tuned +# holdout cells, byte geomean 0.997/0.990 vs the previous default, and +# fewer reconstructions. Build without default features for the legacy +# table-seeded controller. +one-shot-controller = [] # S8 Phase D (AKR source `outside-advice-2026-08-06` §8; # `jpegxl-rs.work.arch-s8-full-redesign-scoped`): # the Phase C-validated (zero safety violations, exhaustively checked) diff --git a/JPXL/crates/jpxl-encode-policy/src/candidate.rs b/JPXL/crates/jpxl-encode-policy/src/candidate.rs index ac84a14b..e8528972 100644 --- a/JPXL/crates/jpxl-encode-policy/src/candidate.rs +++ b/JPXL/crates/jpxl-encode-policy/src/candidate.rs @@ -139,6 +139,23 @@ impl<'a> CandidateSearchContext<'a> { ) } + /// Reduces the frame's aligned DCT8x8 candidates into the one-shot + /// program's [`TransformFeatureSummary`](crate::quality_features::TransformFeatureSummary), + /// filling this context's shared forward cache so the later pixel plans + /// read the same warm entries (PR 4). Quantizer-independent: safe to call + /// before any quantizer choice exists. + pub(crate) fn prepare_quality_transform_summary( + &mut self, + request: &EncodeRequest, + ) -> Result { + crate::quality_transform_summary( + self.transform_frame, + request, + &mut self.fwd_cache, + Some(self.executor), + ) + } + /// Trains entropy for already planned pixels and returns the writer-ready /// plan. Pixels are untouched: the candidate's score is unchanged. pub(crate) fn attach_entropy( diff --git a/JPXL/crates/jpxl-encode-policy/src/lib.rs b/JPXL/crates/jpxl-encode-policy/src/lib.rs index 9b164d0b..fcc82c4b 100644 --- a/JPXL/crates/jpxl-encode-policy/src/lib.rs +++ b/JPXL/crates/jpxl-encode-policy/src/lib.rs @@ -74,7 +74,9 @@ pub mod field; pub mod policy_bank; pub mod quality; pub mod quality_features; +pub mod quality_prediction; pub mod quality_predictor; +pub mod quality_predictor_v2; pub mod quantize; pub mod rate; pub mod reducer; @@ -125,14 +127,16 @@ pub use field::{AqMode, AqTuning}; use field::{DesiredQuantField, mul_lattice_for}; pub use policy_bank::{PerceptualPolicy, rank_alternatives}; pub use quality::{ - PerceptualEvaluator, PerceptualObservation, PolicyTrial, ProbeKind, QualityBudget, - QualityOutcome, QualityProbe, QualityStats, QualityStatus, StructureSource, - search_frame_perceptual, search_frame_perceptual_with_budget, status_name, + LadderPoint, PerceptualEvaluator, PerceptualObservation, PolicyTrial, ProbeKind, QualityBudget, + QualityOutcome, QualityPredictionTrace, QualityProbe, QualityStats, QualityStatus, QualityWork, + StructureSource, search_frame_perceptual, search_frame_perceptual_with_budget, status_name, + sweep_frame_perceptual, }; -pub use quality_features::{SourceFeatures, source_features}; +pub use quality_features::{SourceFeatures, TransformFeatureSummary, source_features}; +pub use quality_prediction::{QualityPredictionV2, predict_v2, shadow_prediction_trace}; pub use rate::{ LadderSearch, QuantizerChoice, RateOutcome, RatePhase, RateProbeStats, RateStatus, RateStep, - Rung, search_frame, + Rung, effective_scale, rung_for_effective_scale, search_frame, }; pub use request::{ AdaptiveSharpness, ChromaHfPolicy, CoverFrequencyWeight, CoverMode, CoverRateModel, @@ -615,9 +619,19 @@ fn build_pixel_plan( } else { request.quantizer_choice }; - let fast_fixed_cover = - request.rate_preset == RateSearchPreset::Fast && entropy_search.uses_fast_entropy(); - let quantizer_transforms = if fast_fixed_cover { + // Fast navigation is deliberately allowed a cheaper structural policy. + // Fixed 8x8 blocks avoid the hierarchical cover's transform-bank scoring + // throughout Fast navigation and finalist planning. The Quality request + // keeps its configured mode. This is a preset-only trade: no Quality/Full + // plan can enter this arm because those requests do not carry + // `RateSearchPreset::Fast` here. + let cover_mode = + if request.rate_preset == RateSearchPreset::Fast && entropy_search.uses_fast_entropy() { + CoverMode::FixedDct8x8 + } else { + request.budget.cover_mode + }; + let quantizer_transforms = if cover_mode == CoverMode::FixedDct8x8 { &FAST_TRANSFORMS[..] } else { &SQUARE_TRANSFORMS[..] @@ -637,20 +651,12 @@ fn build_pixel_plan( .with_dead_zone_scale(request.dead_zone_scale) .with_zero_token_bits(request.zero_token_bits) .with_rate_model(request.cover_rate_model); - cache.prepare(&geometry)?; - - // Fast navigation is deliberately allowed a cheaper structural policy. - // Fixed 8x8 blocks avoid the hierarchical cover's transform-bank scoring - // throughout Fast navigation and finalist planning. The Quality request - // keeps its configured mode. This is a preset-only trade: no Quality/Full - // plan can enter this arm because those requests do not carry - // `RateSearchPreset::Fast` here. - let cover_mode = - if request.rate_preset == RateSearchPreset::Fast && entropy_search.uses_fast_entropy() { - CoverMode::FixedDct8x8 - } else { - request.budget.cover_mode - }; + // Reserve only the square families this search will score, on the request + // thread, before cover fans out. Fast's fixed DCT8x8 cover never touches + // DCT16/DCT32, and pre-creating those banks reserved two empty full-grid + // coefficient arenas per LF group (about two thirds of the forward-cache + // commit on the production Fast path). + cache.prepare_families(&geometry, quantizer_transforms)?; // The cover is selected before chroma-from-luma is estimated: the estimate // regresses over the coefficients of the *selected* transforms, so the @@ -2409,19 +2415,30 @@ impl CandidateGroupBank { fn new( rect: jpxl_encode::vardct::Rect, blocks: jpxl_encode::vardct::BlockGrid, + families: &[TransformType], ) -> Result { - // Reserve all large coefficient arenas on the request thread before - // planning fans out. That keeps their allocator ownership stable - // across warm-up and timed rate probes instead of stranding an arena - // in whichever worker happened to encounter a transform family first. + // Reserve the families this search will score on the request thread + // before planning fans out. That keeps their allocator ownership + // stable across warm-up and timed rate probes instead of stranding an + // arena in whichever worker first encountered a transform. Families + // that this cover never scores stay `None`; `get_or_insert` can still + // create one later if a rescue rebuilds structure with a wider set. + let mut banks = [None, None, None]; + for &transform in families { + let Some(index) = Self::family(transform) else { + continue; + }; + let Some(slot) = banks.get_mut(index) else { + continue; + }; + if slot.is_none() { + *slot = Some(DenseForwardBank::new(blocks, transform)?); + } + } Ok(Self { rect, blocks, - banks: [ - Some(DenseForwardBank::new(blocks, TransformType::Dct8x8)?), - Some(DenseForwardBank::new(blocks, TransformType::Dct16x16)?), - Some(DenseForwardBank::new(blocks, TransformType::Dct32x32)?), - ], + banks, cover_complete: false, complete_hits: std::sync::atomic::AtomicU64::new(0), }) @@ -2581,6 +2598,14 @@ impl CandidateForwardCache { } fn prepare(&mut self, geometry: &VardctGeometry) -> Result<()> { + self.prepare_families(geometry, &SQUARE_TRANSFORMS) + } + + fn prepare_families( + &mut self, + geometry: &VardctGeometry, + families: &[TransformType], + ) -> Result<()> { if !self.groups.is_empty() { if self.groups.len() == usize::try_from(geometry.num_lf_groups()).unwrap_or(usize::MAX) { @@ -2602,7 +2627,7 @@ impl CandidateForwardCache { what: "an LF group outside the frame's block grid", })?; groups.push(std::sync::RwLock::new(CandidateGroupBank::new( - rect, blocks, + rect, blocks, families, )?)); } self.groups = groups; @@ -2700,6 +2725,249 @@ fn ensure_cover_candidates_cached( Ok(()) } +/// Per-LF-group partial accumulators of the transform summary. +struct TransformPartial { + histogram: Vec, + blocks: u64, + ac_cells: u64, + total_y: f64, + total_low: f64, + total_high: f64, + total_row: f64, + total_col: f64, + total_xb: f64, + near_1e3: u64, + near_1e2: u64, + dc_sum: f64, + dc_sumsq: f64, +} + +/// Reduces the frame's aligned DCT8x8 candidates into a +/// [`TransformFeatureSummary`](quality_features::TransformFeatureSummary), +/// filling the shared forward cache as it goes (one-shot program PR 4). +/// +/// Every coefficient computed here is one the hierarchical cover search +/// would compute anyway — the fill goes through the same +/// [`CandidateGroupBank::get_or_insert`] the cover reads — so the later +/// pixel plan reuses the warm entries rather than re-transforming. LF +/// groups fill in parallel on the request executor exactly like cover +/// construction; each group's partials accumulate in block raster order +/// and combine in LF-group index order, so the result is deterministic +/// across worker counts and SIMD modes. +pub(crate) fn quality_transform_summary( + transform_frame: &PreparedFrame, + request: &EncodeRequest, + cache: &mut CandidateForwardCache, + executor: Option<&jpxl_encode::EncodeExecutor>, +) -> Result { + const EPS: f64 = 1e-30; + // Fixed-bin histogram of per-block ln(Y AC energy): [-46, 18) at 0.125. + const HIST_LO: f64 = -46.0; + const HIST_WIDTH: f64 = 0.125; + const HIST_BINS: usize = 512; + + let decision = FrameDecision { + width: transform_frame.width(), + height: transform_frame.height(), + group_size_shift: VARDCT_GROUP_SIZE_SHIFT, + num_passes: 1, + bits_per_sample: request.bits_per_sample, + }; + let geometry = decision.geometry()?; + cache.prepare(&geometry)?; + let shared: &CandidateForwardCache = cache; + let n_groups = usize::try_from(geometry.num_lf_groups()).unwrap_or(usize::MAX); + + let summarize_group = |index: usize| -> Result { + let mut scratch = ForwardScratch::new(); + let mut partial = TransformPartial { + histogram: vec![0u64; HIST_BINS], + blocks: 0, + ac_cells: 0, + total_y: 0.0, + total_low: 0.0, + total_high: 0.0, + total_row: 0.0, + total_col: 0.0, + total_xb: 0.0, + near_1e3: 0, + near_1e2: 0, + dc_sum: 0.0, + dc_sumsq: 0.0, + }; + let lock = shared.group(index)?; + let mut bank = lock.write().map_err(|_| PolicyError::Unsupported { + what: "a poisoned forward-cache bank", + })?; + let (rect, grid) = (bank.rect, bank.blocks); + for by in 0..grid.height { + for bx in 0..grid.width { + let px = rect.x0.saturating_add(bx.saturating_mul(8)); + let py = rect.y0.saturating_add(by.saturating_mul(8)); + let fwd = bank.get_or_insert( + transform_frame, + TransformType::Dct8x8, + px, + py, + &mut scratch, + )?; + let [cx, cy, cb] = fwd.coeffs; + let mut block_y = 0.0f64; + for (cell, &value) in cy.iter().enumerate() { + if cell == 0 { + let dc = f64::from(value); + partial.dc_sum += dc; + partial.dc_sumsq += dc * dc; + continue; + } + let (row, col) = (cell / 8, cell % 8); + let energy = f64::from(value) * f64::from(value); + block_y += energy; + if row + col <= 2 { + partial.total_low += energy; + } + if row.max(col) >= 4 { + partial.total_high += energy; + } + if row > col { + partial.total_row += energy; + } else if col > row { + partial.total_col += energy; + } + let magnitude = f64::from(value).abs(); + if magnitude < 1e-3 { + partial.near_1e3 += 1; + } + if magnitude < 1e-2 { + partial.near_1e2 += 1; + } + partial.ac_cells += 1; + } + for lane in [cx, cb] { + for &value in lane.iter().skip(1) { + partial.total_xb += f64::from(value) * f64::from(value); + } + } + partial.total_y += block_y; + partial.blocks += 1; + let ln_energy = (block_y + EPS).ln(); + #[allow( + clippy::cast_possible_truncation, + clippy::cast_sign_loss, + reason = "clamped into 0..HIST_BINS before the narrowing" + )] + let bin = (((ln_energy - HIST_LO) / HIST_WIDTH).clamp(0.0, (HIST_BINS - 1) as f64)) + as usize; + if let Some(count) = partial.histogram.get_mut(bin) { + *count += 1; + } + } + } + Ok(partial) + }; + + let partials: Vec = if let Some(executor) = executor { + executor.map_ordered(n_groups, summarize_group)? + } else { + (0..n_groups).map(summarize_group).collect::>()? + }; + + // Fixed-order combination of the per-group partials. + let mut histogram = vec![0u64; HIST_BINS]; + let mut blocks = 0u64; + let mut ac_cells = 0u64; + let (mut total_y, mut total_low, mut total_high) = (0.0f64, 0.0f64, 0.0f64); + let (mut total_row, mut total_col) = (0.0f64, 0.0f64); + let mut total_xb = 0.0f64; + let (mut near_1e3, mut near_1e2) = (0u64, 0u64); + let (mut dc_sum, mut dc_sumsq) = (0.0f64, 0.0f64); + for partial in &partials { + for (total, &count) in histogram.iter_mut().zip(partial.histogram.iter()) { + *total += count; + } + blocks += partial.blocks; + ac_cells += partial.ac_cells; + total_y += partial.total_y; + total_low += partial.total_low; + total_high += partial.total_high; + total_row += partial.total_row; + total_col += partial.total_col; + total_xb += partial.total_xb; + near_1e3 += partial.near_1e3; + near_1e2 += partial.near_1e2; + dc_sum += partial.dc_sum; + dc_sumsq += partial.dc_sumsq; + } + + let quantile = |q: f64| -> f64 { + #[allow( + clippy::cast_precision_loss, + clippy::cast_possible_truncation, + clippy::cast_sign_loss, + reason = "block counts stay far inside f64's exact-integer range" + )] + { + let want = (q * blocks as f64).min(blocks.saturating_sub(1) as f64) as u64; + let mut seen = 0u64; + for (bin, &count) in histogram.iter().enumerate() { + seen += count; + if seen > want { + return HIST_LO + (bin as f64 + 0.5) * HIST_WIDTH; + } + } + HIST_LO + (HIST_BINS as f64 - 0.5) * HIST_WIDTH + } + }; + + #[allow( + clippy::cast_precision_loss, + reason = "block and cell counts stay far inside f64's exact-integer range" + )] + Ok(quality_features::TransformFeatureSummary { + blocks, + ln_ac_y_mean: (total_y / (blocks as f64).max(1.0) + EPS).ln(), + ln_ac_y_q50: quantile(0.5), + ln_ac_y_q90: quantile(0.9), + ln_ac_y_q99: quantile(0.99), + high_low_ratio: total_high / (total_low + EPS), + directional_asymmetry: (total_row - total_col).abs() / (total_row + total_col + EPS), + chroma_ac_ratio: total_xb / (total_y + EPS), + near_zero_frac_1e3: near_1e3 as f64 / (ac_cells as f64).max(1.0), + near_zero_frac_1e2: near_1e2 as f64 / (ac_cells as f64).max(1.0), + dc_variance_y: { + let n = (blocks as f64).max(1.0); + let mean = dc_sum / n; + (dc_sumsq / n - mean * mean).max(0.0) + }, + }) +} + +/// Standalone [`TransformFeatureSummary`](quality_features::TransformFeatureSummary) +/// of a frame, for calibration tooling (`jpxl features --transform-summary`). +/// +/// Builds a throwaway forward cache; the in-search path +/// (`QualityBudget::transform_shadow`) shares the cover's cache instead. +/// Applies the request's Gaborish preconditioning so the coefficients are +/// the ones the encode itself would transform. +/// +/// # Errors +/// +/// Whatever the preconditioner or forward transform refuses. +pub fn transform_feature_summary( + frame: &PreparedFrame, + request: &EncodeRequest, +) -> Result { + let transform_owned = if request.restoration.gaborish { + Some(prepare_gaborish_frame(frame)?) + } else { + None + }; + let transform_frame = transform_owned.as_ref().unwrap_or(frame); + let mut cache = CandidateForwardCache::new(); + let executor = request.resources.executor(); + quality_transform_summary(transform_frame, request, &mut cache, Some(&executor)) +} + /// Borrows the (already-cached) forward coefficients for every selected /// varblock, in `varblocks` order. See [`ensure_forwards_cached`]. fn gather_forward_refs<'cache>( @@ -2874,6 +3142,34 @@ fn set_lf( } } +/// Merges a chunk's group-grid LF planes into the group reduction. +/// +/// Chunks of one LF group cover disjoint varblocks, and unwritten cells stay +/// zero, so adding plane-wise is the same as scattering each patch. The add +/// is the vectorisable merge; overlapping non-zero writes would wrap in +/// debug and are a cover bug. +fn add_lf_planes( + dest: &mut [Vec; NUM_CHANNELS], + src: &[Vec; NUM_CHANNELS], +) -> Result<()> { + for channel in 0..NUM_CHANNELS { + let (Some(d), Some(s)) = (dest.get_mut(channel), src.get(channel)) else { + return Err(PolicyError::Unsupported { + what: "a missing LF plane while merging a quantization chunk", + }); + }; + if d.len() != s.len() { + return Err(PolicyError::Unsupported { + what: "a quantization chunk whose LF plane size changed", + }); + } + for (slot, &value) in d.iter_mut().zip(s.iter()) { + *slot += value; + } + } + Ok(()) +} + /// Phase 7.1's estimate of what one interior-zero coefficient token costs. /// /// Token *counts* are exact — the walk emits one per order position up to the @@ -3737,11 +4033,17 @@ fn coefficient_arena_capacity(varblocks: &[VarblockDecision]) -> usize { /// mutating coefficients that an earlier plan can still observe. struct QuantizationWorkspace { arenas: Vec>, + /// Per-chunk LF planes in group-grid layout, returned after each probe's + /// reduction so the next sequential quantizer does not re-reserve them. + chunk_lf: Vec<[Vec; NUM_CHANNELS]>, } impl QuantizationWorkspace { fn new() -> Self { - Self { arenas: Vec::new() } + Self { + arenas: Vec::new(), + chunk_lf: Vec::new(), + } } fn take_arena(&mut self, slot: usize, capacity: usize) -> Arc<[i32]> { @@ -3764,6 +4066,33 @@ impl QuantizationWorkspace { *destination = arena; } } + + fn take_chunk_lf(&mut self, slot: usize, capacity: usize) -> [Vec; NUM_CHANNELS] { + if self.chunk_lf.len() <= slot { + self.chunk_lf.resize_with(slot.saturating_add(1), || { + core::array::from_fn(|_| Vec::new()) + }); + } + let mut planes = core::array::from_fn(|_| Vec::new()); + if let Some(stored) = self.chunk_lf.get_mut(slot) { + std::mem::swap(&mut planes, stored); + } + for plane in &mut planes { + if plane.len() != capacity { + plane.clear(); + plane.resize(capacity, 0); + } else { + plane.fill(0); + } + } + planes + } + + fn put_chunk_lf(&mut self, slot: usize, planes: [Vec; NUM_CHANNELS]) { + if let Some(stored) = self.chunk_lf.get_mut(slot) { + *stored = planes; + } + } } /// Large quantization buffers reserved by the request thread before fanout. @@ -3857,20 +4186,22 @@ fn quantization_chunks(groups: &[PlannedGroup], workers: usize) -> Vec; NUM_CHANNELS], + lf_width: u32, arena: std::sync::Arc<[i32]>, starts: Vec<[usize; NUM_CHANNELS]>, coefficients: Vec, } impl QuantChunkWorkspace { - fn new(varblocks: &[VarblockDecision], arena: Arc<[i32]>) -> Self { - let mut lf_cap = 0usize; - for vb in varblocks { - let n = vb.transform.block_dims().0; - lf_cap = lf_cap.saturating_add(n.saturating_mul(n)); - } + fn new( + varblocks: &[VarblockDecision], + arena: Arc<[i32]>, + lf_values: [Vec; NUM_CHANNELS], + lf_width: u32, + ) -> Self { Self { - lf_values: core::array::from_fn(|_| Vec::with_capacity(lf_cap)), + lf_values, + lf_width, arena, starts: Vec::with_capacity(varblocks.len()), coefficients: Vec::with_capacity(varblocks.len()), @@ -3907,12 +4238,26 @@ fn quantize_groups_parallel( .iter() .enumerate() .map(|(index, chunk)| { - let varblocks = groups + let (blocks, varblocks) = groups .get(chunk.group) - .and_then(|(_, _, _, varblocks)| varblocks.get(chunk.start..chunk.end)) - .unwrap_or(&[]); + .map(|(_, blocks, _, varblocks)| (*blocks, varblocks.get(chunk.start..chunk.end))) + .unwrap_or(( + jpxl_encode::vardct::BlockGrid { + width: 0, + height: 0, + }, + None, + )); + let varblocks = varblocks.unwrap_or(&[]); let arena = quant_workspace.take_arena(index, coefficient_arena_capacity(varblocks)); - std::sync::Mutex::new(Some(QuantChunkWorkspace::new(varblocks, arena))) + let lf_cells = usize::try_from(blocks.area()).unwrap_or(0); + let lf_values = quant_workspace.take_chunk_lf(index, lf_cells); + std::sync::Mutex::new(Some(QuantChunkWorkspace::new( + varblocks, + arena, + lf_values, + blocks.width, + ))) }) .collect(); let quantize_one = |index: usize| { @@ -3980,7 +4325,7 @@ fn quantize_groups_parallel( .collect(); for (index, (chunk, quantized)) in chunks.iter().copied().zip(quantized_chunks).enumerate() { quant_workspace.put_arena(index, quantized.arena); - let (_, blocks, _, group_varblocks) = + let (_, _, _, group_varblocks) = groups.get(chunk.group).ok_or(PolicyError::Unsupported { what: "a missing planned LF group during chunk reduction", })?; @@ -4000,46 +4345,9 @@ fn quantize_groups_parallel( .ok_or(PolicyError::Unsupported { what: "a missing LF-group quantization reduction", })?; - let mut lf_cursor = 0usize; - for vb in varblocks { - let n = vb.transform.block_dims().0; - let cells = n.saturating_mul(n); - let lf_end = lf_cursor - .checked_add(cells) - .ok_or(PolicyError::Unsupported { - what: "a quantized LF patch range overflow", - })?; - for channel in 0..NUM_CHANNELS { - let values = quantized - .lf_values - .get(channel) - .and_then(|plane| plane.get(lf_cursor..lf_end)) - .ok_or(PolicyError::Unsupported { - what: "a quantized LF patch shorter than its varblock", - })?; - for (index, value) in values.iter().copied().enumerate() { - set_lf( - &mut builder.lf_planes, - channel, - vb.origin.bx() + u32::try_from(index % n).unwrap_or(0), - vb.origin.by() + u32::try_from(index / n).unwrap_or(0), - blocks.width, - value, - ); - } - } - lf_cursor = lf_end; - } - if quantized - .lf_values - .iter() - .any(|plane| plane.len() != lf_cursor) - { - return Err(PolicyError::Unsupported { - what: "a quantized chunk with trailing LF patch values", - }); - } + add_lf_planes(&mut builder.lf_planes, &quantized.lf_values)?; builder.coefficients.extend(quantized.coefficients); + quant_workspace.put_chunk_lf(index, quantized.lf_values); } groups @@ -4096,7 +4404,9 @@ fn quantize_chunk( })?; let (bx, by) = (vb.origin.bx(), vb.origin.by()); let factors = varblock_cfl(correlation, cfl, bx, by); - let lf_values = &mut workspace.lf_values; + let lf_planes = &mut workspace.lf_values; + let lf_width = workspace.lf_width; + let n = transform.block_dims().0; quantize_square_varblock( &fwd.coeffs, transform, @@ -4105,10 +4415,15 @@ fn quantize_chunk( factors, &mut tscratch, &mut qscratch, - |channel, _index, value| { - if let Some(plane) = lf_values.get_mut(channel) { - plane.push(value); - } + |channel, idx, value| { + set_lf( + lf_planes, + channel, + bx + u32::try_from(idx % n).unwrap_or(0), + by + u32::try_from(idx / n).unwrap_or(0), + lf_width, + value, + ); }, slot, hf_quants.truncate_trailing, diff --git a/JPXL/crates/jpxl-encode-policy/src/quality.rs b/JPXL/crates/jpxl-encode-policy/src/quality.rs index ef2ca048..3a9549f7 100644 --- a/JPXL/crates/jpxl-encode-policy/src/quality.rs +++ b/JPXL/crates/jpxl-encode-policy/src/quality.rs @@ -19,9 +19,16 @@ //! //! Every probe is a full-frame score of reconstructed pixels; nothing here //! infers a score from a rate. Budgets are hard: a production effort stops at -//! its probe and price caps and reports what it could verify. The target is a -//! floor — a stream is never emitted below it unless the finest quantizer -//! cannot reach it, and then it says so ([`QualityStatus::SaturatedTop`]). +//! its navigation probe and price caps (plus at most one rescue probe beyond +//! the navigation cap when nothing has met the target yet) and reports what +//! it could verify. "Smallest" is bounded the same way: the selection is the +//! smallest among the exact-priced finalists this budget retained, not a +//! global minimum over every conceivable stream meeting the score. The +//! target is a floor — an outcome whose stream is below it says so +//! explicitly: [`QualityStatus::SaturatedTop`] when the ladder's finest rung +//! missed, [`QualityStatus::UnderTargetWorkCap`] when the probes ran out +//! first. The public facade refuses both by default rather than returning +//! them as ordinary successes. use std::time::Instant; @@ -134,7 +141,12 @@ pub const MIN_AIM_MARGIN: f64 = 0.25; /// [`TRIAL_EXACT_PRICES`] cap. #[derive(Debug, Clone, Copy, PartialEq)] pub struct QualityBudget { - /// Full-frame render-and-score evaluations in the baseline solve. + /// Full-frame render-and-score evaluations the baseline solve's + /// *navigation* may spend. When they run out with nothing meeting the + /// target, one rescue probe (`rescue_probe`) may still run beyond this + /// cap, so the observable per-solve maximum is `pixel_probes + 1`; the + /// trace and [`QualityStats::pixel_probes`] count it like any other + /// probe. pub pixel_probes: u32, /// Entropy trainings followed by an exact emission in the baseline solve. pub exact_prices: u32, @@ -155,6 +167,14 @@ pub struct QualityBudget { /// its exact size is smaller and its canonical score still meets the /// target. pub reducer: Option, + /// Whether the search reduces the frame's DCT8x8 candidates into a + /// [`TransformFeatureSummary`](crate::quality_features::TransformFeatureSummary) + /// before solving. On in Fast and Balanced with the (default) + /// `one-shot-controller` feature: the summary feeds the crossing + /// predictor's seed, and its parallel prefill warms the same forward + /// cache the cover search reads, so in-search it measured net-neutral + /// wall (promotion A/B, 2026-08-24). + pub transform_shadow: bool, } /// Hard pixel-probe cap of one policy-bank trial: probe the baseline crossing @@ -199,10 +219,13 @@ pub const MIN_TRIAL_SAVING_FRACTION: f64 = 0.005; /// [`search_frame_perceptual_with_budget`] to opt in. pub const BALANCED_DEFAULT_POLICY_TRIALS: u32 = 0; -/// Whether Balanced runs the terminal reducer by default. Off until PR 7's -/// gate (matched-score bytes down on the locked holdout with zero floor -/// violations and bounded wall) is measured and recorded; the feature-gated -/// Quality effort always runs it. +/// Whether Balanced runs the terminal reducer by default. +/// +/// Off after PR 7's full locked-holdout closure: the reducer stayed bounded +/// and floor-safe, but the complete Quality candidate failed the standing +/// Contract B promotion screen, while the earlier Balanced development screen +/// exceeded its wall bound. The feature-gated Quality reference still runs it. +/// See AKR evidence `pqc-pr7-holdout-complete-2026-08-24`. pub const BALANCED_DEFAULT_REDUCER: Option = None; impl QualityBudget { @@ -222,6 +245,7 @@ impl QualityBudget { policy_trials: 0, reserve: 0.06, reducer: None, + transform_shadow: cfg!(feature = "one-shot-controller"), }, RateSearchPreset::Balanced => Self { pixel_probes: 5, @@ -230,6 +254,7 @@ impl QualityBudget { policy_trials: BALANCED_DEFAULT_POLICY_TRIALS, reserve: 0.03, reducer: BALANCED_DEFAULT_REDUCER, + transform_shadow: cfg!(feature = "one-shot-controller"), }, RateSearchPreset::Quality => Self { pixel_probes: 10, @@ -238,6 +263,7 @@ impl QualityBudget { policy_trials: 11, reserve: 0.02, reducer: Some(ReducerLimits::QUALITY), + transform_shadow: false, }, } } @@ -362,6 +388,70 @@ pub struct QualityStats { pub reducer_edits: u32, /// Exact bytes the terminal reducer saved (0 when its stream was not kept). pub reducer_bytes_saved: u64, + /// Whole-search work totals (baseline, policy trials and reducer alike), + /// split by the units the one-shot program prices separately. A probe is + /// one pixel plan, one full-frame reconstruction and one canonical metric + /// evaluation; an exact price is one entropy training and one emission; a + /// finalist rebuilt from a dropped plan is one extra pixel plan. + pub work: QualityWork, +} + +/// Whole-search work totals for the `jpxl.quality-trace/2` record. +/// +/// Unlike the effort-budget counters above these are *totals across the whole +/// search* — baseline solve, policy-bank trials and the terminal reducer — +/// so the trace shows what a request actually cost, not just what counted +/// against the baseline caps. (A reducer pass that finds nothing to remove +/// reports no evaluations, so its canonical scores are not included in that +/// one case.) +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct QualityWork { + /// Candidate pixel plans built (scored probes and unscored rebuilds). + pub pixel_plans: u32, + /// Full-frame reconstructions rendered for scoring. + pub reconstructions: u32, + /// Canonical metric evaluations of a reconstruction. + pub metric_evaluations: u32, + /// Entropy trainings. + pub entropy_trainings: u32, + /// Exact codestream emissions. + pub emissions: u32, +} + +/// A shadow predictor's counterfactual decision for one search, recorded in +/// the `jpxl.quality-trace/2` record without influencing the search (the +/// one-shot program's PR 3/4 shadow stage; the exact controller stays +/// authoritative). `None` until a shadow model is wired in. +#[derive(Debug, Clone, PartialEq)] +pub struct QualityPredictionTrace { + /// The generated model's version string. + pub model_version: &'static str, + /// The feature-vector schema the model consumed. + pub feature_schema: &'static str, + /// Predicted median crossing rung for the target. + pub median_rung: u32, + /// Risk-adjusted candidate rung a one-shot controller would plan first. + pub candidate_rung: u32, + /// Calibrated lower crossing-interval rung. + pub interval_low: u32, + /// Calibrated upper crossing-interval rung. + pub interval_high: u32, + /// Predicted local loss exponent `-d ln(loss) / d ln(scale)`. + pub local_loss_exponent: f64, + /// Predicted risk (0..=1) that the target saturates the ladder. + pub saturation_risk: f64, + /// Out-of-distribution flags that fired, by name. + pub ood_flags: Vec<&'static str>, + /// Why the shadow would have routed to the exact controller, if it would. + pub fallback_reason: Option<&'static str>, + /// The canonical score the search observed at (or nearest to) the + /// candidate rung, for counterfactual first-plan accounting. + pub first_observed_score: Option, + /// The slope-corrected rung the shadow would have re-planned at. + pub correction_rung: Option, + /// The counterfactual route: `one_shot`, `corrected`, or + /// `fallback_exact`. + pub decision_path: &'static str, } /// What a completed score-targeted search chose. @@ -393,6 +483,10 @@ pub struct QualityOutcome { pub policy_trials: Vec, /// The source features the prediction was made from. pub features: SourceFeatures, + /// The shadow predictor's counterfactual record, when one is wired in. + pub prediction: Option, + /// The PR 4 transform-feature summary, when the budget asked for one. + pub transform_features: Option, /// The metric the scores come from. pub metric_version: &'static str, } @@ -405,7 +499,7 @@ impl QualityOutcome { self.achieved_score - self.requested_score } - /// The `jpxl.quality-trace/1` record of this search, one JSON line. + /// The `jpxl.quality-trace/2` record of this search, one JSON line. #[must_use] pub fn trace_json(&self, effort: &str) -> String { let probes: Vec = self @@ -458,11 +552,43 @@ impl QualityOutcome { || "null".to_owned(), |(lo, hi)| format!("[{},{}]", lo.get(), hi.get()), ); + let prediction = self.prediction.as_ref().map_or_else( + || "null".to_owned(), + |p| { + let flags: Vec = p.ood_flags.iter().map(|f| format!("\"{f}\"")).collect(); + format!( + "{{\"model_version\":\"{}\",\"feature_schema\":\"{}\",\"median_rung\":{},\ + \"candidate_rung\":{},\"interval_low\":{},\"interval_high\":{},\ + \"local_loss_exponent\":{},\"saturation_risk\":{},\"ood_flags\":[{}],\ + \"fallback_reason\":{},\"first_observed_score\":{},\"correction_rung\":{},\ + \"decision_path\":\"{}\"}}", + p.model_version, + p.feature_schema, + p.median_rung, + p.candidate_rung, + p.interval_low, + p.interval_high, + p.local_loss_exponent, + p.saturation_risk, + flags.join(","), + p.fallback_reason + .map_or_else(|| "null".to_owned(), |r| format!("\"{r}\"")), + p.first_observed_score + .map_or_else(|| "null".to_owned(), |s| format!("{s}")), + p.correction_rung + .map_or_else(|| "null".to_owned(), |r| format!("{r}")), + p.decision_path, + ) + }, + ); format!( - "{{\"schema\":\"jpxl.quality-trace/1\",\"metric_version\":\"{}\",\"score_guard\":{},\ + "{{\"schema\":\"jpxl.quality-trace/2\",\"metric_version\":\"{}\",\"score_guard\":{},\ \"effort\":\"{}\",\"source_features\":{},\"predicted_rung\":{},\"bracket\":{},\ \"pixel_probes\":{},\"exact_prices\":{},\"structural_builds\":{},\"policy_trials\":{},\ \"policy_winner_margin_bytes\":{},\"reducer\":{{\"evaluations\":{},\"edits\":{},\"bytes_saved\":{}}},\ + \"work\":{{\"pixel_plans\":{},\"reconstructions\":{},\"metric_evaluations\":{},\ + \"entropy_trainings\":{},\"emissions\":{}}},\"prediction\":{},\ + \"transform_features\":{},\ \"requested_score\":{},\"achieved_score\":{},\ \"guard_margin\":{},\"final_exact_bytes\":{},\"status\":\"{}\",\"saturated\":{},\ \"wall_by_phase\":{{\"plan\":{},\"render_metric\":{},\"entropy\":{},\"emit\":{}}},\ @@ -483,6 +609,15 @@ impl QualityOutcome { self.stats.reducer_evaluations, self.stats.reducer_edits, self.stats.reducer_bytes_saved, + self.stats.work.pixel_plans, + self.stats.work.reconstructions, + self.stats.work.metric_evaluations, + self.stats.work.entropy_trainings, + self.stats.work.emissions, + prediction, + self.transform_features + .as_ref() + .map_or_else(|| "null".to_owned(), |t| t.to_json()), self.requested_score, self.achieved_score, self.achieved_score - self.requested_score - self.guard, @@ -694,6 +829,11 @@ struct Navigator<'c, 'a, 'r, 't, 'e> { policy_id: u32, anchor: Option, anchor_rung: Option, + /// A model-predicted local loss exponent used in place of + /// [`PRIOR_LOSS_EXPONENT`] while only one probe exists (the one-shot + /// program's slope prior). Measured slopes always win once two probes + /// bracket a direction. + prior_beta: Option, probes: Vec, trace: &'t mut Vec, local: QualityStats, @@ -759,6 +899,9 @@ impl Navigator<'_, '_, '_, '_, '_> { let millis = u64::try_from(score_start.elapsed().as_millis()).unwrap_or(u64::MAX); self.local.render_metric_ms = self.local.render_metric_ms.saturating_add(millis); self.local.pixel_probes = self.local.pixel_probes.saturating_add(1); + self.local.work.pixel_plans = self.local.work.pixel_plans.saturating_add(1); + self.local.work.reconstructions = self.local.work.reconstructions.saturating_add(1); + self.local.work.metric_evaluations = self.local.work.metric_evaluations.saturating_add(1); let feasible = score >= self.threshold(); self.trace.push(QualityProbe { policy_id: self.policy_id, @@ -859,7 +1002,7 @@ impl Navigator<'_, '_, '_, '_, '_> { let dy = loss(from.1).ln() - loss(o.1).ln(); (dx.abs() > f64::EPSILON && dy / dx < 0.0).then(|| (-dy / dx).clamp(0.2, 3.0)) }) - .unwrap_or(PRIOR_LOSS_EXPONENT); + .unwrap_or_else(|| self.prior_beta.unwrap_or(PRIOR_LOSS_EXPONENT)); let margin = EXPANSION_MARGIN.powi(i32::try_from(attempt.saturating_add(1)).unwrap_or(1)); let mut ratio = (loss(from.1) / loss(self.threshold())).powf(1.0 / alpha); ratio = if finer { @@ -987,6 +1130,8 @@ fn price_pixels( let emit_ms = u64::try_from(emit_start.elapsed().as_millis()).unwrap_or(u64::MAX); nav.local.emit_ms = nav.local.emit_ms.saturating_add(emit_ms); nav.local.exact_prices = nav.local.exact_prices.saturating_add(1); + nav.local.work.entropy_trainings = nav.local.work.entropy_trainings.saturating_add(1); + nav.local.work.emissions = nav.local.work.emissions.saturating_add(1); let feasible = score >= nav.threshold(); nav.trace.push(QualityProbe { policy_id: nav.policy_id, @@ -1061,6 +1206,7 @@ fn solve_baseline( structure_tier: EntropySearch, finalist_entropy: EntropySearch, predicted: Rung, + prior_beta: Option, trace: &mut Vec, ) -> Result<(PolicySolve, QualityStats, Option)> { let mut nav = Navigator { @@ -1076,6 +1222,7 @@ fn solve_baseline( policy_id: 0, anchor: None, anchor_rung: None, + prior_beta, probes: Vec::new(), trace, local: QualityStats { @@ -1152,14 +1299,17 @@ fn solve_baseline( .and_then(|p| p.pixels.take()); let (pixels, geometry) = match retained { Some(planned) => planned, - None => nav.ctx.pixel_plan_for( - nav.request, - quantizer, - nav.enable_cfl, - nav.structure_tier, - AnchorReuse::None, - None, - )?, + None => { + nav.local.work.pixel_plans = nav.local.work.pixel_plans.saturating_add(1); + nav.ctx.pixel_plan_for( + nav.request, + quantizer, + nav.enable_cfl, + nav.structure_tier, + AnchorReuse::None, + None, + )? + } }; let priced = price_pixels( &mut nav, @@ -1184,6 +1334,7 @@ fn solve_baseline( (Some(anchor), StructureSource::Reused) => AnchorReuse::CoverAndCfl(anchor), _ => AnchorReuse::None, }; + nav.local.work.pixel_plans = nav.local.work.pixel_plans.saturating_add(1); nav.ctx.pixel_plan_for( nav.request, quantizer, @@ -1243,20 +1394,15 @@ fn solve_baseline( )) } -/// One bank alternative's bounded solve: a priced finalist and its work. -struct TrialSolve { - finalist: PricedFinalist, - local: QualityStats, -} - /// Solves one policy-bank alternative to the same target under a small budget /// ([`TRIAL_PIXEL_PROBES`] pixel probes, [`TRIAL_EXACT_PRICES`] exact price), /// starting from the baseline's crossing rung `seed`. /// /// `shared_anchor` is `Some` only for a quantizer-side alternative that reuses /// the baseline's cover and CfL; a CfL or restoration alternative passes `None` -/// and builds a fresh plan. Returns `None` when the alternative found no -/// feasible stream inside its budget. +/// and builds a fresh plan. The finalist is `None` when the alternative found +/// no feasible stream inside its budget; its stats come back either way so +/// the search's whole-work totals stay honest. /// /// `ctx` is the baseline's warm context: a quantizer-side trial reads its /// forward coefficients from that cache and spends zero fresh structural @@ -1278,7 +1424,7 @@ fn solve_trial( shared_anchor: Option<&StructuralAnchor>, policy_id: u32, trace: &mut Vec, -) -> Result> { +) -> Result<(Option, QualityStats)> { let seed_fresh = shared_anchor.is_none(); let anchor_rung = shared_anchor.as_ref().map(|_| seed); let mut nav = Navigator { @@ -1294,6 +1440,7 @@ fn solve_trial( policy_trials: 0, reserve, reducer: None, + transform_shadow: false, }, enable_cfl, structure_tier, @@ -1301,6 +1448,7 @@ fn solve_trial( policy_id, anchor: shared_anchor.cloned(), anchor_rung, + prior_beta: None, probes: Vec::new(), trace, local: QualityStats::default(), @@ -1326,11 +1474,11 @@ fn solve_trial( .min_by_key(|(_, p)| p.rung) .map(|(i, _)| i); let Some(index) = chosen_index else { - return Ok(None); + return Ok((None, nav.local)); }; let (quantizer, score, structure) = { let Some(p) = nav.probes.get(index) else { - return Ok(None); + return Ok((None, nav.local)); }; (p.quantizer, p.score, p.structure) }; @@ -1342,6 +1490,7 @@ fn solve_trial( (Some(anchor), StructureSource::Reused) => AnchorReuse::CoverAndCfl(anchor), _ => AnchorReuse::None, }; + nav.local.work.pixel_plans = nav.local.work.pixel_plans.saturating_add(1); nav.ctx.pixel_plan_for( nav.request, quantizer, @@ -1354,11 +1503,11 @@ fn solve_trial( }; let finalist = price_pixels(&mut nav, quantizer, score, structure, &pixels, &geometry)?; let local = nav.local; - Ok(Some(TrialSolve { finalist, local })) + Ok((Some(finalist), local)) } -/// Folds a trial's timings (not its probe/price counts, which stay baseline- -/// scoped) into the combined stats. +/// Folds a trial's timings and whole-search work totals (not its probe/price +/// counts, which stay baseline-scoped) into the combined stats. fn fold_timings(combined: &mut QualityStats, trial: &QualityStats) { combined.plan_ms = combined.plan_ms.saturating_add(trial.plan_ms); combined.render_metric_ms = combined @@ -1366,6 +1515,23 @@ fn fold_timings(combined: &mut QualityStats, trial: &QualityStats) { .saturating_add(trial.render_metric_ms); combined.entropy_ms = combined.entropy_ms.saturating_add(trial.entropy_ms); combined.emit_ms = combined.emit_ms.saturating_add(trial.emit_ms); + combined.work.pixel_plans = combined + .work + .pixel_plans + .saturating_add(trial.work.pixel_plans); + combined.work.reconstructions = combined + .work + .reconstructions + .saturating_add(trial.work.reconstructions); + combined.work.metric_evaluations = combined + .work + .metric_evaluations + .saturating_add(trial.work.metric_evaluations); + combined.work.entropy_trainings = combined + .work + .entropy_trainings + .saturating_add(trial.work.entropy_trainings); + combined.work.emissions = combined.work.emissions.saturating_add(trial.work.emissions); } /// Runs the terminal reducer on the winning finalist and replaces it when the @@ -1410,6 +1576,14 @@ fn reduce_winner( return Ok(()); }; stats.reducer_evaluations = reduced.stats.evaluations; + stats.work.reconstructions = stats + .work + .reconstructions + .saturating_add(reduced.stats.evaluations); + stats.work.metric_evaluations = stats + .work + .metric_evaluations + .saturating_add(reduced.stats.evaluations); let ctx = CandidateSearchContext::new(frame, transform_frame, atlas, request, executor); let entropy_start = Instant::now(); @@ -1421,6 +1595,8 @@ fn reduce_winner( let emit_ms = u64::try_from(emit_start.elapsed().as_millis()).unwrap_or(u64::MAX); stats.emit_ms = stats.emit_ms.saturating_add(emit_ms); stats.exact_prices = stats.exact_prices.saturating_add(1); + stats.work.entropy_trainings = stats.work.entropy_trainings.saturating_add(1); + stats.work.emissions = stats.work.emissions.saturating_add(1); let kept = emission.sizing.total < winner.sizing.total; trace.push(QualityProbe { policy_id: REDUCER_POLICY_ID, @@ -1447,6 +1623,121 @@ fn reduce_winner( /// The `policy_id` the trace gives the reducer's exact price. pub const REDUCER_POLICY_ID: u32 = u32::MAX; +/// One fresh-structure point of a quality-oracle ladder sweep. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct LadderPoint { + /// The rung that was built and scored. + pub rung: Rung, + /// The quantizer at that rung (with the effort's `quant_lf` coupling). + pub quantizer: QuantizerChoice, + /// `global_scale * HfMul` of that quantizer. + pub effective_scale: u64, + /// The canonical score of the fresh-structure reconstruction. + pub score: f64, + /// The exact codestream size, when pricing was requested. + pub exact_bytes: Option, + /// Milliseconds in pixel planning. + pub plan_ms: u64, + /// Milliseconds in rendering and scoring. + pub render_metric_ms: u64, + /// Milliseconds in entropy training and emission (0 when not priced). + pub price_ms: u64, +} + +/// Sweeps `rungs` with the production quality pixel policy of `request`'s +/// effort — a fresh cover/CfL build per rung over one shared forward-DCT +/// cache — scoring every rung canonically and exact-pricing it when `price` +/// is set. +/// +/// This is the one-shot program's oracle-label measurement (PR 2): a +/// crossing predictor must train on the same fresh-structure operating +/// points the production controller emits, not on fixed-quantizer +/// approximations of them. No navigation happens here; every requested rung +/// is built, scored and dropped before the next, so at most one frame-sized +/// pixel plan is alive at a time. +/// +/// # Errors +/// +/// As [`search_frame_perceptual`]. +pub fn sweep_frame_perceptual( + frame: &PreparedFrame, + atlas: &AnalysisAtlas, + request: &EncodeRequest, + rungs: &[Rung], + price: bool, + evaluator: &mut dyn PerceptualEvaluator, + executor: &jpxl_encode::EncodeExecutor, +) -> Result> { + let preset = request.rate_preset; + let (enable_cfl, structure_tier, finalist_entropy) = planning_tiers(preset); + let transform_owned = if request.restoration.gaborish { + Some(crate::prepare_gaborish_frame(frame)?) + } else { + None + }; + let transform_frame = transform_owned.as_ref().unwrap_or(frame); + let baseline_policy = crate::policy_bank::PerceptualPolicy::baseline(preset); + let base_request = baseline_policy.apply(request); + let mut ctx = + CandidateSearchContext::new(frame, transform_frame, atlas, &base_request, executor); + + let mut points = Vec::with_capacity(rungs.len()); + for &rung in rungs { + let quantizer = QuantizerChoice::at(rung, base_request.quant_lf)?; + let plan_start = Instant::now(); + let (pixels, geometry) = ctx.pixel_plan_for( + &base_request, + quantizer, + enable_cfl, + structure_tier, + AnchorReuse::None, + None, + )?; + let plan_ms = u64::try_from(plan_start.elapsed().as_millis()).unwrap_or(u64::MAX); + let score_start = Instant::now(); + let (observation, retained) = evaluator.evaluate_owned(pixels)?; + let render_metric_ms = u64::try_from(score_start.elapsed().as_millis()).unwrap_or(u64::MAX); + let mut price_ms = 0u64; + let exact_bytes = if price { + // A memory-bounded evaluator may have dropped the plan; the fresh + // rebuild is byte-identical by construction. + let pixels = match retained { + Some(pixels) => pixels, + None => { + ctx.pixel_plan_for( + &base_request, + quantizer, + enable_cfl, + structure_tier, + AnchorReuse::None, + None, + )? + .0 + } + }; + let price_start = Instant::now(); + let plan = + ctx.attach_entropy_for(&base_request, &pixels, &geometry, finalist_entropy)?; + let emission = emit_codestream_with_executor(&plan, ctx.executor())?; + price_ms = u64::try_from(price_start.elapsed().as_millis()).unwrap_or(u64::MAX); + Some(emission.sizing.total) + } else { + None + }; + points.push(LadderPoint { + rung, + quantizer, + effective_scale: effective_scale(rung), + score: observation.score, + exact_bytes, + plan_ms, + render_metric_ms, + price_ms, + }); + } + Ok(points) +} + /// Runs the score-targeted search over a prepared frame at the preset's budget. /// /// `request` carries the effort (`rate_preset`) and the starting policy; its @@ -1515,6 +1806,38 @@ pub fn search_frame_perceptual_with_budget( let base_request = baseline_policy.apply(request); let mut ctx = CandidateSearchContext::new(frame, transform_frame, atlas, &base_request, executor); + // PR 4 shadow measurement: reduce the DCT8x8 candidates before the solve + // so the summary is quantizer-independent and the fill warms the cover's + // own cache. Off in every production preset. + let transform_features = if budget.transform_shadow { + Some(ctx.prepare_quality_transform_summary(&base_request)?) + } else { + None + }; + // PR 5 (feature-gated): the generated crossing model's risk-adjusted + // candidate becomes the first fresh plan and its slope steers the first + // correction; a fallback signal (OOD, wide interval, saturation risk, + // tiny frame, or no generated model) keeps the standard predictor start. + // Everything downstream — canonical verification before entropy, the + // bounded navigator, the total caps — is unchanged. The summary the + // prediction reads warmed the cover's own cache, so this costs no + // duplicated transform work. + #[cfg(feature = "one-shot-controller")] + let (predicted, prior_beta) = match transform_features + .as_ref() + .and_then(|tf| crate::quality_prediction::predict_v2(&features, tf, target_score)) + { + // Confident: the risk-adjusted candidate is the first fresh plan. + Some(p) if p.fallback_reason.is_none() => (p.candidate_rung, Some(p.local_loss_exponent)), + // Uncertain but in distribution (wide interval, saturation risk): + // the navigator still runs its full bounded search, so the model's + // median is simply a better seed than the legacy table. + Some(p) if p.ood_flags.is_empty() => (p.median_rung, Some(p.local_loss_exponent)), + // Out of distribution: keep the legacy predictor's start. + _ => (predicted, None), + }; + #[cfg(not(feature = "one-shot-controller"))] + let prior_beta: Option = None; let (baseline, mut stats, baseline_anchor) = solve_baseline( &mut ctx, &base_request, @@ -1526,6 +1849,7 @@ pub fn search_frame_perceptual_with_budget( structure_tier, finalist_entropy, predicted, + prior_beta, &mut trace, )?; @@ -1580,7 +1904,7 @@ pub fn search_frame_perceptual_with_budget( None }; let trial_request = policy.apply(request); - let trial = solve_trial( + let (trial_finalist, trial_stats) = solve_trial( &mut ctx, &trial_request, evaluator, @@ -1596,10 +1920,10 @@ pub fn search_frame_perceptual_with_budget( &mut trace, )?; stats.policy_trials = stats.policy_trials.saturating_add(1); - match trial { - Some(t) => { - fold_timings(&mut stats, &t.local); - let bytes = t.finalist.sizing.total; + fold_timings(&mut stats, &trial_stats); + match trial_finalist { + Some(finalist) => { + let bytes = finalist.sizing.total; let saving = incumbent_bytes.saturating_sub(bytes); // Quality demands a minimum saving; a single Balanced // pass keeps any strict improvement. @@ -1612,16 +1936,16 @@ pub fn search_frame_perceptual_with_budget( } else { saving > 0 }; - let keep = t.finalist.feasible && bytes < incumbent_bytes && enough; + let keep = finalist.feasible && bytes < incumbent_bytes && enough; policy_trials.push(PolicyTrial { id: policy_id, - rung: t.finalist.quantizer.rung.get(), - score: t.finalist.score, + rung: finalist.quantizer.rung.get(), + score: finalist.score, bytes, kept: false, }); if keep { - winner = t.finalist; + winner = finalist; winner_policy = policy; winner_is_trial = true; winner_trial_id = Some(policy_id); @@ -1695,6 +2019,12 @@ pub fn search_frame_perceptual_with_budget( QualityStatus::MetWorkCap }; let metric_version = evaluator.metric_version(); + // PR 3/4 shadow: the counterfactual one-shot record, computed from the + // finished search's own probes. Never touches the emitted bytes; absent + // when no transform summary was computed. + let prediction = transform_features.as_ref().and_then(|tf| { + crate::quality_prediction::shadow_prediction_trace(&features, tf, target_score, &trace) + }); Ok(QualityOutcome { codestream: winner.bytes, plan: winner.plan, @@ -1709,6 +2039,8 @@ pub fn search_frame_perceptual_with_budget( stats, policy_trials, features, + prediction, + transform_features, metric_version, }) } @@ -1762,8 +2094,7 @@ mod tests { } } - fn frame() -> PreparedFrame { - let (w, h) = (96u32, 80u32); + fn sized_frame(w: u32, h: u32) -> PreparedFrame { let rgb: Vec = (0..w * h) .flat_map(|i| { let x = i % w; @@ -1778,6 +2109,10 @@ mod tests { PreparedFrame::from_srgb8(w, h, &rgb).expect("frame") } + fn frame() -> PreparedFrame { + sized_frame(96, 80) + } + fn run(preset: RateSearchPreset, target: f64) -> (QualityOutcome, u32) { run_with_budget(preset, target, QualityBudget::for_preset(preset)) } @@ -1819,6 +2154,116 @@ mod tests { (outcome, evaluator.calls) } + /// PR 5 (feature-gated): the one-shot start plans the generated model's + /// risk-adjusted candidate as the first fresh plan when the model routes + /// there, and keeps the standard predictor start when any fallback + /// signal fires. The caps and floor semantics are unchanged either way. + #[cfg(feature = "one-shot-controller")] + #[test] + fn the_one_shot_start_follows_the_model_routing() { + let (w, h) = (160u32, 144u32); + let frame = sized_frame(w, h); + let atlas = AnalysisAtlas::analyze(&frame); + let mut request = EncodeRequest::for_quality(RateSearchPreset::Balanced); + request.restoration.gaborish = false; + let executor = request.resources.executor(); + let mut evaluator = CurveEvaluator { + calls: 0, + discard: false, + }; + let target = PerceptualTarget::new(PerceptualMetric::Ssimulacra2, 85.0).expect("target"); + let budget = QualityBudget::for_preset(RateSearchPreset::Balanced); + let outcome = search_frame_perceptual_with_budget( + &frame, + &atlas, + &request, + target, + &mut evaluator, + &executor, + budget, + ) + .expect("search"); + + let features = source_features(&atlas, w, h, frame.is_grayscale()); + let mut check_cache = crate::CandidateForwardCache::new(); + let summary = + crate::quality_transform_summary(&frame, &request, &mut check_cache, Some(&executor)) + .expect("transform summary"); + match crate::quality_prediction::predict_v2(&features, &summary, 85.0) { + Some(p) if p.fallback_reason.is_none() => { + assert_eq!( + outcome.stats.predicted, + Some(p.candidate_rung), + "the model's candidate is the first fresh plan" + ); + assert_eq!( + outcome.trace.first().map(|probe| probe.quantizer.rung), + Some(p.candidate_rung) + ); + } + Some(p) if p.ood_flags.is_empty() => { + assert_eq!( + outcome.stats.predicted, + Some(p.median_rung), + "an uncertain in-distribution route seeds the navigator with the median" + ); + } + _ => { + let standard = rung_for_scale(predicted_effective_scale(&features, 85.0)); + assert_eq!( + outcome.stats.predicted, + Some(standard), + "an out-of-distribution frame keeps the legacy predictor start" + ); + } + } + assert!(outcome.achieved_score >= 85.0); + // The one rescue probe beyond the navigation cap stays the maximum. + assert!(outcome.stats.pixel_probes <= budget.pixel_probes + 1); + assert!(outcome.prediction.is_some(), "the shadow trace still fills"); + } + + /// A solve that runs out of probes before anything meets a reachable + /// target reports [`QualityStatus::UnderTargetWorkCap`] — not a success + /// status — and its counters show the one rescue probe that ran beyond + /// the navigation cap. + #[test] + fn running_out_of_probes_reports_under_target_work_cap() { + let budget = QualityBudget { + pixel_probes: 1, + exact_prices: 1, + structural_builds: 1, + policy_trials: 0, + reserve: 0.03, + reducer: None, + transform_shadow: false, + }; + // The curve tops out near 99.96, so 99.9 is reachable on the ladder; + // one navigation probe plus one bounded rescue jump cannot get there. + let (outcome, _) = run_with_budget(RateSearchPreset::Balanced, 99.9, budget); + assert_eq!(outcome.status, QualityStatus::UnderTargetWorkCap); + assert!(!outcome.saturated, "{:?}", outcome.stats); + assert!(outcome.achieved_score < 99.9); + // The navigation cap of one, plus the single documented rescue probe. + assert_eq!(outcome.stats.pixel_probes, 2); + assert_eq!(outcome.stats.exact_prices, 1); + } + + /// Proves the failed Quality promotion screen cannot leak its policy-bank + /// or reducer work into either production effort's default budget. + #[test] + fn production_budgets_keep_quality_only_work_disabled() { + for preset in [RateSearchPreset::Fast, RateSearchPreset::Balanced] { + let budget = QualityBudget::for_preset(preset); + assert_eq!(budget.policy_trials, 0); + assert_eq!(budget.reducer, None); + } + + let quality = QualityBudget::for_preset(RateSearchPreset::Quality); + assert!(quality.policy_trials > 0); + assert_eq!(quality.reducer, Some(ReducerLimits::QUALITY)); + } + #[test] fn dropping_probe_plans_rebuilds_the_byte_identical_finalist() { let budget = QualityBudget::for_preset(RateSearchPreset::Balanced); @@ -2022,7 +2467,7 @@ mod tests { fn the_trace_is_machine_readable() { let (outcome, _) = run(RateSearchPreset::Fast, 70.0); let json = outcome.trace_json("fast"); - assert!(json.starts_with("{\"schema\":\"jpxl.quality-trace/1\"")); + assert!(json.starts_with("{\"schema\":\"jpxl.quality-trace/2\"")); assert!(json.contains("\"probes\":[{\"kind\":\"pixel\"")); assert!(json.contains("\"policy_id\":0")); assert!(json.contains("\"policy_trials\":0")); @@ -2044,10 +2489,12 @@ mod tests { }; let with = QualityBudget { reducer: Some(limits), + transform_shadow: false, ..QualityBudget::for_preset(RateSearchPreset::Balanced) }; let without = QualityBudget { reducer: None, + transform_shadow: false, ..QualityBudget::for_preset(RateSearchPreset::Balanced) }; let (reduced, _) = run_with_budget(RateSearchPreset::Balanced, 70.0, with); @@ -2132,6 +2579,7 @@ mod tests { structure_tier, finalist_entropy, predicted, + None, &mut trace, ) .expect("baseline solve"); @@ -2149,7 +2597,7 @@ mod tests { "a chroma-only alternative must reuse structure" ); let qs_request = qs_policy.apply(&request); - let qs = solve_trial( + let (qs_finalist, qs_stats) = solve_trial( &mut ctx, &qs_request, &mut evaluator, @@ -2164,10 +2612,10 @@ mod tests { 1, &mut trace, ) - .expect("quantizer-side trial") - .expect("a feasible quantizer-side finalist"); + .expect("quantizer-side trial"); + assert!(qs_finalist.is_some(), "a feasible quantizer-side finalist"); assert_eq!( - qs.local.structural_builds, 0, + qs_stats.structural_builds, 0, "a quantizer-side trial rebuilt structure" ); @@ -2180,7 +2628,7 @@ mod tests { "a CfL flip must not reuse structure" ); let st_request = st_policy.apply(&request); - let st = solve_trial( + let (st_finalist, st_stats) = solve_trial( &mut ctx, &st_request, &mut evaluator, @@ -2195,10 +2643,10 @@ mod tests { 2, &mut trace, ) - .expect("structural trial") - .expect("a feasible structural finalist"); + .expect("structural trial"); + assert!(st_finalist.is_some(), "a feasible structural finalist"); assert_eq!( - st.local.structural_builds, 1, + st_stats.structural_builds, 1, "a structural trial did not build exactly one cover" ); } diff --git a/JPXL/crates/jpxl-encode-policy/src/quality_features.rs b/JPXL/crates/jpxl-encode-policy/src/quality_features.rs index d124a375..01f6c659 100644 --- a/JPXL/crates/jpxl-encode-policy/src/quality_features.rs +++ b/JPXL/crates/jpxl-encode-policy/src/quality_features.rs @@ -79,6 +79,66 @@ impl SourceFeatures { } } +/// Compact quantizer-independent transform features of one frame (the +/// one-shot program's PR 4 summary). +/// +/// Reduced from the aligned DCT8x8 coefficients of the shared candidate +/// forward cache — the same bank the cover search reads — in fixed LF-group +/// then block-raster order, so the values are deterministic across worker +/// counts and the prefilled coefficients are reused, never recomputed, by +/// the later pixel plan. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct TransformFeatureSummary { + /// 8x8 blocks reduced. + pub blocks: u64, + /// `ln(mean per-block Y AC energy + eps)`. + pub ln_ac_y_mean: f64, + /// Median of per-block `ln(Y AC energy + eps)` (fixed-bin histogram). + pub ln_ac_y_q50: f64, + /// 90th percentile of per-block `ln(Y AC energy + eps)`. + pub ln_ac_y_q90: f64, + /// 99th percentile of per-block `ln(Y AC energy + eps)`. + pub ln_ac_y_q99: f64, + /// Frame high-band over low-band Y AC energy (radial split at + /// `row+col <= 2` / `max(row,col) >= 4`). + pub high_low_ratio: f64, + /// `|row - col|` energy asymmetry of the Y AC cells, in `0..=1`. + pub directional_asymmetry: f64, + /// Chroma (X+B) over Y AC energy. + pub chroma_ac_ratio: f64, + /// Fraction of Y AC cells with `|coefficient| < 1e-3`. + pub near_zero_frac_1e3: f64, + /// Fraction of Y AC cells with `|coefficient| < 1e-2`. + pub near_zero_frac_1e2: f64, + /// Variance of the per-block Y DC coefficient. + pub dc_variance_y: f64, +} + +impl TransformFeatureSummary { + /// A one-line JSON object, for the CLI, the trace and the calibration + /// tooling. + #[must_use] + pub fn to_json(&self) -> String { + format!( + "{{\"blocks\":{},\"ln_ac_y_mean\":{},\"ln_ac_y_q50\":{},\"ln_ac_y_q90\":{},\ + \"ln_ac_y_q99\":{},\"high_low_ratio\":{},\"directional_asymmetry\":{},\ + \"chroma_ac_ratio\":{},\"near_zero_frac_1e3\":{},\"near_zero_frac_1e2\":{},\ + \"dc_variance_y\":{}}}", + self.blocks, + self.ln_ac_y_mean, + self.ln_ac_y_q50, + self.ln_ac_y_q90, + self.ln_ac_y_q99, + self.high_low_ratio, + self.directional_asymmetry, + self.chroma_ac_ratio, + self.near_zero_frac_1e3, + self.near_zero_frac_1e2, + self.dc_variance_y, + ) + } +} + /// The quantile of `values` at fraction `q` in `0..=1`: the element at /// `floor((n - 1) q)` of the ascending total order. `0.0` for an empty slice. #[must_use] diff --git a/JPXL/crates/jpxl-encode-policy/src/quality_prediction.rs b/JPXL/crates/jpxl-encode-policy/src/quality_prediction.rs new file mode 100644 index 00000000..6b0d02ca --- /dev/null +++ b/JPXL/crates/jpxl-encode-policy/src/quality_prediction.rs @@ -0,0 +1,487 @@ +//! Shadow crossing predictor for the one-shot quality program (PR 3). +//! +//! Evaluates the generated [`crate::quality_predictor_v2`] model on a +//! frame's [`SourceFeatures`] and, after the exact controller has finished, +//! reconstructs the counterfactual one-shot decision for the +//! `jpxl.quality-trace/2` record. **Nothing here influences the search or +//! the emitted bytes** — the exact controller stays authoritative; this +//! module only says what a one-shot controller *would have done*, so the +//! offline evaluation can measure it. +//! +//! Determinism: scalar `f64` arithmetic in fixed order over generated +//! constants; no worker-count or SIMD dependence. + +use crate::quality::{ProbeKind, QualityProbe}; +use crate::quality_features::{SourceFeatures, TransformFeatureSummary}; +use crate::quality_predictor_v2::{ + QPV2_FEATURE_CENTERS, QPV2_FEATURE_DIM, QPV2_FEATURE_SCALES, QPV2_FEATURE_SCHEMA, QPV2_KNOTS, + QPV2_MODEL_VERSION, Qpv2Knot, +}; +use crate::rate::{Rung, effective_scale, rung_for_effective_scale}; + +/// Matches the trainer's `EPS`: logs stay finite on zero-valued features. +const FEATURE_EPS: f64 = 1e-9; + +/// Loss floor, as in the navigator's crossing interpolation. +const LOSS_EPSILON: f64 = 1e-3; + +/// Calibrated-interval width (in `ln scale`) beyond which the shadow routes +/// to the exact controller (the memo's `ln(1.5)` threshold). +pub const FALLBACK_LOG_WIDTH: f64 = 0.405_465; + +/// Saturation risk beyond which the shadow routes to the exact controller. +pub const FALLBACK_SATURATION_RISK: f64 = 0.5; + +/// Margin the training-time standardized feature range is widened by before +/// a feature counts as out of distribution. +const OOD_RANGE_MARGIN: f64 = 0.25; + +/// The model's declared domain floor: below this shortest side the metric +/// sits against its own floor and the shadow routes straight to the exact +/// controller. Mirrors the trainer's `MIN_DOMAIN_SIDE`. +pub const MIN_DOMAIN_SIDE: u32 = 128; + +/// Runtime clamp of the predicted local loss exponent, mirroring the +/// navigator's own slope clamp. +const BETA_RANGE: (f64, f64) = (0.2, 3.0); + +/// What the shadow model predicts for one `(frame, target)` request. +#[derive(Debug, Clone, PartialEq)] +pub struct QualityPredictionV2 { + /// Median crossing rung. + pub median_rung: Rung, + /// Risk-adjusted candidate rung (rounded toward the finer legal rung). + pub candidate_rung: Rung, + /// Lower calibrated crossing-interval rung. + pub interval_low: Rung, + /// Upper calibrated crossing-interval rung. + pub interval_high: Rung, + /// Predicted local loss exponent, clamped into [`BETA_RANGE`]. + pub local_loss_exponent: f64, + /// Predicted probability that the target saturates the ladder. + pub saturation_risk: f64, + /// Out-of-distribution flags that fired, by name. + pub ood_flags: Vec<&'static str>, + /// Why the shadow would route to the exact controller, if it would. + pub fallback_reason: Option<&'static str>, +} + +fn loss(score: f64) -> f64 { + (100.0 - score).max(LOSS_EPSILON) +} + +fn ln_eps(value: f32) -> f64 { + (f64::from(value).max(0.0) + FEATURE_EPS).ln() +} + +/// The `qpv2-st/1` feature vector of a frame: the nine source features +/// followed by the ten transform-summary fields in sorted key order. Must +/// mirror the trainer's `feature_vector(..., "source+transform")` exactly; +/// the schema string and the generated `QPV2_FEATURE_DIM` pin it — a +/// regenerated model with a different schema fails to compile here. +#[must_use] +pub fn qpv2_features( + features: &SourceFeatures, + transform: &TransformFeatureSummary, +) -> [f64; QPV2_FEATURE_DIM] { + let width = f64::from(features.width.max(1)); + let height = f64::from(features.height.max(1)); + [ + ln_eps(features.luma_variance_q10), + ln_eps(features.luma_variance_q50), + ln_eps(features.luma_variance_q90), + ln_eps(features.chroma_variance_q50), + f64::from(features.flat_fraction), + ln_eps(features.edge_proxy), + (width * height).log2(), + (width / height).ln(), + if features.grayscale { 1.0 } else { 0.0 }, + transform.chroma_ac_ratio, + transform.dc_variance_y, + transform.directional_asymmetry, + transform.high_low_ratio, + transform.ln_ac_y_mean, + transform.ln_ac_y_q50, + transform.ln_ac_y_q90, + transform.ln_ac_y_q99, + transform.near_zero_frac_1e2, + transform.near_zero_frac_1e3, + ] +} + +fn standardized( + features: &SourceFeatures, + transform: &TransformFeatureSummary, +) -> [f64; QPV2_FEATURE_DIM] { + let raw = qpv2_features(features, transform); + let mut out = [0.0; QPV2_FEATURE_DIM]; + for (slot, ((&value, ¢er), &scale)) in out.iter_mut().zip( + raw.iter() + .zip(QPV2_FEATURE_CENTERS.iter()) + .zip(QPV2_FEATURE_SCALES.iter()), + ) { + *slot = (value - center) / scale; + } + out +} + +/// `[intercept, w_0, ..]` dot a standardized feature vector. +fn affine(weights: &[f64; QPV2_FEATURE_DIM + 1], z: &[f64; QPV2_FEATURE_DIM]) -> f64 { + let mut acc = weights.first().copied().unwrap_or(0.0); + for (&weight, &value) in weights.iter().skip(1).zip(z.iter()) { + acc += weight * value; + } + acc +} + +/// Per-knot raw outputs before monotone projection. +struct KnotEval { + ln_loss: f64, + lower: f64, + median: f64, + candidate: f64, + upper: f64, + ln_beta: f64, + saturation_logit: f64, +} + +fn evaluate_knot(knot: &Qpv2Knot, z: &[f64; QPV2_FEATURE_DIM]) -> KnotEval { + KnotEval { + ln_loss: loss(knot.target).ln(), + lower: affine(&knot.lower, z), + median: affine(&knot.median, z), + candidate: affine(&knot.candidate, z), + upper: affine(&knot.upper, z), + ln_beta: affine(&knot.beta, z), + saturation_logit: affine(&knot.saturation, z), + } +} + +/// Linear interpolation of `field` against `ln loss(target)`, clamped to the +/// end knots. Returns 0 only on an empty table, which the caller precludes. +fn interpolate(evals: &[KnotEval], target_ln_loss: f64, field: impl Fn(&KnotEval) -> f64) -> f64 { + // Knots ascend in target, so their ln-loss descends. + let (Some(first), Some(last)) = (evals.first(), evals.last()) else { + return 0.0; + }; + if target_ln_loss >= first.ln_loss { + return field(first); + } + if target_ln_loss <= last.ln_loss { + return field(last); + } + for pair in evals.windows(2) { + let [a, b] = pair else { continue }; + if target_ln_loss <= a.ln_loss && target_ln_loss >= b.ln_loss { + let span = a.ln_loss - b.ln_loss; + if span <= f64::EPSILON { + return field(a); + } + let t = (a.ln_loss - target_ln_loss) / span; + return field(a) + t * (field(b) - field(a)); + } + } + field(last) +} + +/// The finer legal rung at or above an effective scale. +fn rung_ceil(scale: f64) -> Rung { + if !scale.is_finite() { + return Rung::TOP; + } + let clamped = scale.clamp(1.0, 9.0e18); + #[allow( + clippy::cast_possible_truncation, + clippy::cast_sign_loss, + reason = "clamped positive and far inside u64 range" + )] + let floor = rung_for_effective_scale(clamped as u64); + #[allow( + clippy::cast_precision_loss, + reason = "effective scales stay far inside f64's exact-integer range" + )] + if (effective_scale(floor) as f64) < scale { + Rung::new(floor.get().saturating_add(1)) + } else { + floor + } +} + +fn rung_floor(scale: f64) -> Rung { + if !scale.is_finite() { + return Rung::TOP; + } + let clamped = scale.clamp(1.0, 9.0e18); + #[allow( + clippy::cast_possible_truncation, + clippy::cast_sign_loss, + reason = "clamped positive and far inside u64 range" + )] + rung_for_effective_scale(clamped as u64) +} + +/// Evaluates the generated model for one request. `None` when no model has +/// been generated (empty knot table). +#[must_use] +pub fn predict_v2( + features: &SourceFeatures, + transform: &TransformFeatureSummary, + target: f64, +) -> Option { + if QPV2_KNOTS.is_empty() { + return None; + } + let z = standardized(features, transform); + + let mut ood_flags: Vec<&'static str> = Vec::new(); + if features.width.min(features.height) < MIN_DOMAIN_SIDE { + ood_flags.push("tiny_frame"); + } + for (i, (&value, &(lo, hi))) in z + .iter() + .zip(crate::quality_predictor_v2::QPV2_FEATURE_Z_RANGE.iter()) + .enumerate() + { + let span = (hi - lo).max(1e-6); + if value < lo - OOD_RANGE_MARGIN * span || value > hi + OOD_RANGE_MARGIN * span { + ood_flags.push(qpv2_feature_name(i)); + } + } + + // Monotone projection: a higher target must never predict a coarser + // scale, per output. Knots ascend in target, so run a max-accumulate. + let mut projected: Vec = QPV2_KNOTS.iter().map(|k| evaluate_knot(k, &z)).collect(); + let mut running = ( + f64::NEG_INFINITY, + f64::NEG_INFINITY, + f64::NEG_INFINITY, + f64::NEG_INFINITY, + ); + for eval in &mut projected { + running.0 = running.0.max(eval.lower); + running.1 = running.1.max(eval.median); + running.2 = running.2.max(eval.candidate); + running.3 = running.3.max(eval.upper); + eval.lower = running.0; + eval.median = running.1; + eval.candidate = running.2; + eval.upper = running.3; + } + + let x = loss(target).ln(); + let ln_lower = interpolate(&projected, x, |e| e.lower); + let ln_median = interpolate(&projected, x, |e| e.median); + let ln_candidate = interpolate(&projected, x, |e| e.candidate); + let ln_upper = interpolate(&projected, x, |e| e.upper); + let ln_beta = interpolate(&projected, x, |e| e.ln_beta); + let saturation_logit = interpolate(&projected, x, |e| e.saturation_logit); + let saturation_risk = 1.0 / (1.0 + (-saturation_logit.clamp(-30.0, 30.0)).exp()); + + let interval_width = (ln_upper - ln_lower).max(0.0); + let fallback_reason = if !ood_flags.is_empty() { + Some("ood_feature") + } else if interval_width > FALLBACK_LOG_WIDTH { + Some("wide_interval") + } else if saturation_risk > FALLBACK_SATURATION_RISK { + Some("saturation_risk") + } else { + None + }; + + Some(QualityPredictionV2 { + median_rung: rung_floor(ln_median.exp()), + candidate_rung: rung_ceil(ln_candidate.exp()), + interval_low: rung_floor(ln_lower.exp()), + interval_high: rung_ceil(ln_upper.exp()), + local_loss_exponent: ln_beta.exp().clamp(BETA_RANGE.0, BETA_RANGE.1), + saturation_risk, + ood_flags, + fallback_reason, + }) +} + +/// The schema name of feature `index`, for OOD flags. +#[must_use] +pub const fn qpv2_feature_name(index: usize) -> &'static str { + match index { + 0 => "ln_luma_q10", + 1 => "ln_luma_q50", + 2 => "ln_luma_q90", + 3 => "ln_chroma_q50", + 4 => "flat_fraction", + 5 => "ln_edge_proxy", + 6 => "log2_pixels", + 7 => "ln_aspect", + 8 => "grayscale", + 9 => "chroma_ac_ratio", + 10 => "dc_variance_y", + 11 => "directional_asymmetry", + 12 => "high_low_ratio", + 13 => "ln_ac_y_mean", + 14 => "ln_ac_y_q50", + 15 => "ln_ac_y_q90", + 16 => "ln_ac_y_q99", + 17 => "near_zero_frac_1e2", + 18 => "near_zero_frac_1e3", + _ => "unknown", + } +} + +/// The counterfactual one-shot record of a finished exact search. +/// +/// The exact controller's probes bound what the one-shot path would have +/// seen: above the coarsest canonically feasible probe the score curve is +/// assumed monotone (the same assumption the navigator's bracketing makes), +/// so a candidate at or finer than it would have met the target with its +/// first plan. `first_observed_score` is the score of the probe nearest the +/// candidate in log scale — a proxy, since the search did not necessarily +/// probe the candidate rung itself; the offline oracle evaluation measures +/// the exact counterfactual. +#[must_use] +pub fn shadow_prediction_trace( + features: &SourceFeatures, + transform: &TransformFeatureSummary, + target: f64, + probes: &[QualityProbe], +) -> Option { + let prediction = predict_v2(features, transform, target)?; + let pixel_probes: Vec<&QualityProbe> = probes + .iter() + .filter(|p| p.kind == ProbeKind::Pixel) + .collect(); + let coarsest_feasible = pixel_probes + .iter() + .filter(|p| p.feasible == Some(true)) + .map(|p| p.effective_scale) + .min(); + let candidate_scale = effective_scale(prediction.candidate_rung); + #[allow( + clippy::cast_precision_loss, + reason = "effective scales stay far inside f64's exact-integer range" + )] + let nearest = pixel_probes.iter().min_by(|a, b| { + let da = ((a.effective_scale as f64).ln() - (candidate_scale as f64).ln()).abs(); + let db = ((b.effective_scale as f64).ln() - (candidate_scale as f64).ln()).abs(); + da.total_cmp(&db) + }); + let first_observed_score = nearest.and_then(|p| p.score); + + // One slope correction from the proxy observation, aimed at the target. + #[allow( + clippy::cast_precision_loss, + reason = "effective scales stay far inside f64's exact-integer range" + )] + let correction_scale = first_observed_score.map(|observed| { + let beta = prediction.local_loss_exponent; + let shift = (loss(observed).ln() - loss(target).ln()) / beta; + ((candidate_scale as f64).ln() + shift).exp() + }); + let correction_rung = correction_scale.map(rung_ceil); + + let would_one_shot = coarsest_feasible.is_some_and(|feasible| candidate_scale >= feasible); + let would_correct = correction_rung + .map(effective_scale) + .zip(coarsest_feasible) + .is_some_and(|(corrected, feasible)| corrected >= feasible); + let decision_path = if prediction.fallback_reason.is_some() { + "fallback_exact" + } else if would_one_shot { + "one_shot" + } else if would_correct { + "corrected" + } else { + "fallback_exact" + }; + + Some(crate::quality::QualityPredictionTrace { + model_version: QPV2_MODEL_VERSION, + feature_schema: QPV2_FEATURE_SCHEMA, + median_rung: prediction.median_rung.get(), + candidate_rung: prediction.candidate_rung.get(), + interval_low: prediction.interval_low.get(), + interval_high: prediction.interval_high.get(), + local_loss_exponent: prediction.local_loss_exponent, + saturation_risk: prediction.saturation_risk, + ood_flags: prediction.ood_flags, + fallback_reason: prediction.fallback_reason, + first_observed_score, + correction_rung: correction_rung.map(Rung::get), + decision_path, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn features() -> SourceFeatures { + SourceFeatures { + width: 640, + height: 480, + grayscale: false, + luma_variance_q10: 1e-6, + luma_variance_q50: 1e-4, + luma_variance_q90: 1e-2, + chroma_variance_q50: 1e-5, + flat_fraction: 0.1, + edge_proxy: 1e-2, + } + } + + fn transform() -> TransformFeatureSummary { + TransformFeatureSummary { + blocks: 4800, + ln_ac_y_mean: -5.0, + ln_ac_y_q50: -6.0, + ln_ac_y_q90: -3.0, + ln_ac_y_q99: -2.0, + high_low_ratio: 0.2, + directional_asymmetry: 0.3, + chroma_ac_ratio: 0.15, + near_zero_frac_1e3: 0.6, + near_zero_frac_1e2: 0.8, + dc_variance_y: 0.02, + } + } + + #[test] + fn feature_vector_matches_the_schema_order() { + let raw = qpv2_features(&features(), &transform()); + assert!((raw[0] - (f64::from(1e-6f32) + FEATURE_EPS).ln()).abs() < 1e-12); + assert!((raw[6] - (640.0f64 * 480.0).log2()).abs() < 1e-12); + assert!((raw[7] - (640.0f64 / 480.0).ln()).abs() < 1e-12); + assert!(raw[8].abs() < f64::EPSILON); + // Transform fields follow in sorted key order. + assert!((raw[9] - 0.15).abs() < 1e-12, "chroma_ac_ratio first"); + assert!((raw[13] - (-5.0)).abs() < 1e-12, "ln_ac_y_mean fifth"); + assert!((raw[18] - 0.6).abs() < 1e-12, "near_zero_frac_1e3 last"); + } + + #[test] + fn predictions_are_monotone_in_the_target() { + // With any generated model, a higher target must never predict a + // coarser candidate scale. + let f = features(); + let t = transform(); + let mut previous = 0u64; + for target in [30.0, 50.0, 70.0, 80.0, 85.0, 90.0, 95.0] { + let Some(p) = predict_v2(&f, &t, target) else { + return; // no generated model in this build + }; + let scale = effective_scale(p.candidate_rung); + assert!( + scale >= previous, + "candidate scale fell from {previous} to {scale} at target {target}" + ); + previous = scale; + } + } + + #[test] + fn rung_ceil_rounds_toward_the_finer_rung() { + let r = rung_ceil(1000.5); + assert!(effective_scale(r) >= 1001); + let exact = rung_ceil(1000.0); + assert_eq!(effective_scale(exact), 1000); + } +} diff --git a/JPXL/crates/jpxl-encode-policy/src/quality_predictor_v2.rs b/JPXL/crates/jpxl-encode-policy/src/quality_predictor_v2.rs new file mode 100644 index 00000000..d78138ee --- /dev/null +++ b/JPXL/crates/jpxl-encode-policy/src/quality_predictor_v2.rs @@ -0,0 +1,941 @@ +//! Generated crossing-predictor model for the one-shot quality program. +//! +//! GENERATED by `tools/quality_predictor_v2.py 1.0.0` — do not edit +//! by hand; retrain and regenerate. The exact controller stays +//! authoritative: without the `one-shot-controller` feature this model +//! only feeds the shadow trace, never a bitstream. +//! +//! model_version: qpv2-st-1 +//! feature_schema: qpv2-st/1 +//! labels: 392effdab51fab52619614a59c314364b7d773eb9aa0ce7bf5f6bb110b49c950 +//! trained_at: 2026-08-24T15:10:14.303200+00:00 +//! git: 41a2e3f20631e000954b9ea76612c70f1847e745 (dirty: True) + +/// The generated model's version string. +pub const QPV2_MODEL_VERSION: &str = "qpv2-st-1"; +/// The feature-vector schema the coefficients expect. +pub const QPV2_FEATURE_SCHEMA: &str = "qpv2-st/1"; +/// The feature-vector dimension the coefficient arrays expect. The +/// runtime feature builder is sized against this at compile time. +pub const QPV2_FEATURE_DIM: usize = 19; + +/// Robust per-feature centers (medians over the training corpus). +pub const QPV2_FEATURE_CENTERS: [f64; 19] = [ + -9.804432031649899, + -7.527123141389171, + -5.094806725732717, + -7.578809711477392, + 0.022668848, + -5.285449316183245, + 20.69740200750271, + 0.28768207245178085, + 0.0, + 0.9391887812286073, + 0.03084669961656289, + 0.12828016954152927, + 0.33538596970509227, + -6.049970993368567, + -7.5625, + -5.0625, + -3.8125, + 0.9519056543634857, + 0.6940668783068783, +]; +/// Robust per-feature scales (1.4826 x MAD, floored). +pub const QPV2_FEATURE_SCALES: [f64; 19] = [ + 1.7434782317247506, + 1.6071485038785283, + 1.1280981048610355, + 1.5083122188406448, + 0.033274915118802, + 1.133360277934255, + 1.649302812754329, + 0.27030994010271736, + 1.0, + 0.05797373270690192, + 0.015056661603060293, + 0.1281692462974138, + 0.2569004217225202, + 1.0147714110886856, + 1.4826, + 1.11195, + 0.7413, + 0.04076729688731605, + 0.16466970129379055, +]; +/// Standardized-feature range seen in training, for OOD detection. +pub const QPV2_FEATURE_Z_RANGE: [(f64, f64); 19] = [ + (-6.262672860844934, 4.164861502241656), + (-8.210904383578129, 3.132408616811535), + (-13.853812043358483, 2.4359539240747887), + (-8.714678540211274, 3.3719360810073935), + (-0.6812593786960845, 29.371409318720058), + (-13.621278971326976, 2.2304082316550726), + (-1.7126795063878064, 1.7092853057315096), + (-3.6285356455593627, 2.269207202761088), + (0.0, 1.0), + (-16.20024685967468, 41.45050553876453), + (-2.042234699960017, 2.980187744798955), + (-1.0008654435234647, 6.801318222903994), + (-1.3055096112973485, 78.9263064391509), + (-62.11012756935515, 3.4515346765027033), + (-25.88358289491434, 3.4567651423175505), + (-36.75974639147444, 2.4731327847475155), + (-56.82584648590315, 2.3607176581680833), + (-15.220903994458967, 1.1797285890563396), + (-3.625112992913436, 1.8578592132580616), +]; + +/// One target knot's fitted linear models over standardized features. +/// Every coefficient array is `[intercept, w_0, .., w_n]`; crossing +/// outputs are `ln(effective_scale)`, `beta` is `ln(beta)`, and +/// `saturation` is a logit. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct Qpv2Knot { + /// The SSIMULACRA2 target this knot was fitted at. + pub target: f64, + /// Rows the crossing fits saw (uncensored). + pub rows: u32, + /// tau = 0.10 crossing quantile. + pub lower: [f64; 20], + /// tau = 0.50 crossing quantile. + pub median: [f64; 20], + /// tau = 0.90 crossing quantile (the one-shot candidate). + pub candidate: [f64; 20], + /// tau = 0.95 crossing quantile. + pub upper: [f64; 20], + /// ln(beta) least squares, or all-zero when too few rows. + pub beta: [f64; 20], + /// Saturation-risk logit. + pub saturation: [f64; 20], +} + +/// The fitted knots, ascending in target. +pub const QPV2_KNOTS: &[Qpv2Knot] = &[ + Qpv2Knot { + target: 30.0, + rows: 120, + lower: [ + 8.114267187113095, + -0.08228926450496746, + -0.035979849974954146, + -0.13988211404720818, + -0.3076403802731068, + -0.10340925312497726, + 0.24846174892886125, + 0.10118970836416148, + 0.01714749498939879, + -0.04059456546624838, + 0.023948490246457652, + 0.032747942818945344, + 0.009869219636491513, + -0.01746895538671726, + -0.10688071800844812, + 0.08243659064140416, + -0.05761006875297063, + 0.1397602914320146, + -0.12825290922215582, + -0.1790292259105136, + ], + median: [ + 8.326558131015705, + -0.027905864283035566, + -0.10399385231316674, + -0.46603046061674286, + 0.0779756166300174, + -0.013273056686474605, + 0.38046624283275293, + 0.0775637003066588, + 0.0010359257772471573, + -0.13721640679997588, + -0.01188775720990278, + -0.009256909203737758, + -0.013510311358363936, + 0.00833541366196533, + -0.017121227078915797, + 0.010611252598317671, + 0.04738672782396992, + 0.05164329678211175, + 0.00785653659463203, + -0.186589504365656, + ], + candidate: [ + 8.50667862429574, + -0.06933995160515433, + 0.13523416346587877, + -0.25453416624877684, + -0.09339691707056483, + -0.028861639096780363, + 0.08909729054020864, + 0.07452275903068231, + 0.0015514437296684763, + -0.20180215242465555, + 0.0028171503107653774, + 0.008544032212880864, + -0.00631294438531654, + 0.015266467397116429, + -0.03819215247543441, + -0.014991574919852809, + 0.07821162327823956, + 0.059255403532592794, + 0.0046808983555065185, + -0.16926610160313862, + ], + upper: [ + 8.563199847168615, + -0.05320482494928357, + 0.10762571042744223, + -0.15157121758954228, + -0.04457075214341936, + -0.02212236311624152, + -0.00469827180321417, + 0.053429486870360574, + -0.003065210953815426, + -0.1170463120141785, + 0.0024890621194557138, + 0.012482006834372618, + 0.014671341805076553, + 0.020702418247172033, + -0.05640535378764008, + -0.01835549989417961, + 0.07350561505793092, + 0.08129301706816137, + 0.0036466903063132336, + -0.15396948678804814, + ], + beta: [ + -0.41539412831554556, + -0.038269239125313985, + 0.4827660200910804, + -0.813234037882281, + -0.20364398146863608, + 0.042994046614659207, + 0.6638402356517423, + 0.0155425400805581, + -0.017248888193020114, + -0.0413544204655544, + -0.0007077226572284477, + 0.027678918068457018, + 3.604402143431094e-05, + 0.017872920354533223, + 0.004041951767086818, + -0.05328027163908902, + 0.05645081430217192, + -0.05734065113534405, + -0.050635552274912655, + 0.030131288157661782, + ], + saturation: [ + -6.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, + 0.0, 0.0, 0.0, + ], + }, + Qpv2Knot { + target: 50.0, + rows: 120, + lower: [ + 8.56447381353096, + -0.08228926450496746, + -0.02762538786775922, + -0.13988211404720818, + -0.3076403802731068, + -0.0959612385248504, + 0.24846174892886125, + 0.10118970836416148, + 0.01714749498939879, + -0.04059456546624838, + 0.023948490246457652, + 0.032747942818945344, + 0.009869219636491513, + -0.01746895538671726, + -0.10558317785670433, + 0.08243659064140416, + -0.05761006875297063, + 0.1397602914320146, + -0.12825290922215582, + -0.1790292259105136, + ], + median: [ + 8.770075795795721, + -0.027905864283035566, + -0.1094743229654219, + -0.46603046061674286, + 0.0779756166300174, + -0.011023632799554595, + 0.38046624283275293, + 0.0775637003066588, + 0.0010359257772471573, + -0.13721640679997588, + -0.01188775720990278, + -0.009256909203737758, + -0.013510311358363936, + 0.00833541366196533, + -0.01551082761154858, + 0.010611252598317671, + 0.04738672782396992, + 0.05164329678211175, + 0.00785653659463203, + -0.186589504365656, + ], + candidate: [ + 8.955111231206477, + -0.06933995160515433, + 0.13498530038145756, + -0.25453416624877684, + -0.09339691707056483, + -0.02462073423726832, + 0.08909729054020864, + 0.07452275903068231, + 0.0015514437296684763, + -0.20180215242465555, + 0.0028171503107653774, + 0.008544032212880864, + -0.00631294438531654, + 0.015266467397116429, + -0.034626256325926884, + -0.014991574919852809, + 0.07821162327823956, + 0.059255403532592794, + 0.0046808983555065185, + -0.16926610160313862, + ], + upper: [ + 9.01012687416021, + -0.05320482494928357, + 0.10195301530398537, + -0.15157121758954228, + -0.04457075214341936, + -0.01997235954572073, + -0.00469827180321417, + 0.053429486870360574, + -0.003065210953815426, + -0.1170463120141785, + 0.0024890621194557138, + 0.012482006834372618, + 0.014671341805076553, + 0.020702418247172033, + -0.053134849491821026, + -0.01835549989417961, + 0.07350561505793092, + 0.08129301706816137, + 0.0036466903063132336, + -0.15396948678804814, + ], + beta: [ + -0.39999861818939886, + -0.038269239125313985, + 0.4664120354776818, + -0.813234037882281, + -0.20364398146863608, + 0.033889492171596564, + 0.6638402356517423, + 0.0155425400805581, + -0.017248888193020114, + -0.0413544204655544, + -0.0007077226572284477, + 0.027678918068457018, + 3.604402143431094e-05, + 0.017872920354533223, + 0.007145290132814428, + -0.05328027163908902, + 0.05645081430217192, + -0.05734065113534405, + -0.050635552274912655, + 0.030131288157661782, + ], + saturation: [ + -6.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, + 0.0, 0.0, 0.0, + ], + }, + Qpv2Knot { + target: 70.0, + rows: 120, + lower: [ + 9.247968782009997, + -0.08228926450496746, + -0.014941805531242256, + -0.13988211404720818, + -0.3076403802731068, + -0.08465380721303281, + 0.24846174892886125, + 0.10118970836416148, + 0.01714749498939879, + -0.04059456546624838, + 0.023948490246457652, + 0.032747942818945344, + 0.009869219636491513, + -0.01746895538671726, + -0.10361327756409233, + 0.08243659064140416, + -0.05761006875297063, + 0.1397602914320146, + -0.12825290922215582, + -0.1790292259105136, + ], + median: [ + 9.443415712448457, + -0.027905864283035566, + -0.11779466721605637, + -0.46603046061674286, + 0.0779756166300174, + -0.007608601081964171, + 0.38046624283275293, + 0.0775637003066588, + 0.0010359257772471573, + -0.13721640679997588, + -0.01188775720990278, + -0.009256909203737758, + -0.013510311358363936, + 0.00833541366196533, + -0.013065950373857306, + 0.010611252598317671, + 0.04738672782396992, + 0.05164329678211175, + 0.00785653659463203, + -0.186589504365656, + ], + candidate: [ + 9.635912917676867, + -0.06933995160515433, + 0.13460748127338384, + -0.25453416624877684, + -0.09339691707056483, + -0.018182274731822518, + 0.08909729054020864, + 0.07452275903068231, + 0.0015514437296684763, + -0.20180215242465555, + 0.0028171503107653774, + 0.008544032212880864, + -0.00631294438531654, + 0.015266467397116429, + -0.029212581952229446, + -0.014991574919852809, + 0.07821162327823956, + 0.059255403532592794, + 0.0046808983555065185, + -0.16926610160313862, + ], + upper: [ + 9.688642818398053, + -0.05320482494928357, + 0.09334083958936223, + -0.15157121758954228, + -0.04457075214341936, + -0.01670826582423255, + -0.00469827180321417, + 0.053429486870360574, + -0.003065210953815426, + -0.1170463120141785, + 0.0024890621194557138, + 0.012482006834372618, + 0.014671341805076553, + 0.020702418247172033, + -0.04816963330127428, + -0.01835549989417961, + 0.07350561505793092, + 0.08129301706816137, + 0.0036466903063132336, + -0.15396948678804814, + ], + beta: [ + -0.37662545331149566, + -0.038269239125313985, + 0.4415837332228263, + -0.813234037882281, + -0.20364398146863608, + 0.020067134198791168, + 0.6638402356517423, + 0.0155425400805581, + -0.017248888193020114, + -0.0413544204655544, + -0.0007077226572284477, + 0.027678918068457018, + 3.604402143431094e-05, + 0.017872920354533223, + 0.011856718250476218, + -0.05328027163908902, + 0.05645081430217192, + -0.05734065113534405, + -0.050635552274912655, + 0.030131288157661782, + ], + saturation: [ + -6.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, + 0.0, 0.0, 0.0, + ], + }, + Qpv2Knot { + target: 80.0, + rows: 120, + lower: [ + 9.790489259433608, + -0.08228926450496746, + -0.004874279855140785, + -0.13988211404720818, + -0.3076403802731068, + -0.07567859405620016, + 0.24846174892886125, + 0.10118970836416148, + 0.01714749498939879, + -0.04059456546624838, + 0.023948490246457652, + 0.032747942818945344, + 0.009869219636491513, + -0.01746895538671726, + -0.10204967973746466, + 0.08243659064140416, + -0.05761006875297063, + 0.1397602914320146, + -0.12825290922215582, + -0.1790292259105136, + ], + median: [ + 9.977875671771509, + -0.027905864283035566, + -0.12439889598635413, + -0.46603046061674286, + 0.0779756166300174, + -0.004897937912099513, + 0.38046624283275293, + 0.0775637003066588, + 0.0010359257772471573, + -0.13721640679997588, + -0.01188775720990278, + -0.009256909203737758, + -0.013510311358363936, + 0.00833541366196533, + -0.011125342136190267, + 0.010611252598317671, + 0.04738672782396992, + 0.05164329678211175, + 0.00785653659463203, + -0.186589504365656, + ], + candidate: [ + 10.176295616903781, + -0.06933995160515433, + 0.13430758937608425, + -0.25453416624877684, + -0.09339691707056483, + -0.013071781917896005, + 0.08909729054020864, + 0.07452275903068231, + 0.0015514437296684763, + -0.20180215242465555, + 0.0028171503107653774, + 0.008544032212880864, + -0.00631294438531654, + 0.015266467397116429, + -0.024915506858362992, + -0.014991574919852809, + 0.07821162327823956, + 0.059255403532592794, + 0.0046808983555065185, + -0.16926610160313862, + ], + upper: [ + 10.227211221946934, + -0.05320482494928357, + 0.08650497115458725, + -0.15157121758954228, + -0.04457075214341936, + -0.014117408881894293, + -0.00469827180321417, + 0.053429486870360574, + -0.003065210953815426, + -0.1170463120141785, + 0.0024890621194557138, + 0.012482006834372618, + 0.014671341805076553, + 0.020702418247172033, + -0.04422851949292957, + -0.01835549989417961, + 0.07350561505793092, + 0.08129301706816137, + 0.0036466903063132336, + -0.15396948678804814, + ], + beta: [ + -0.3580731286372142, + -0.038269239125313985, + 0.4218764010344182, + -0.813234037882281, + -0.20364398146863608, + 0.009095711449017933, + 0.6638402356517423, + 0.0155425400805581, + -0.017248888193020114, + -0.0413544204655544, + -0.0007077226572284477, + 0.027678918068457018, + 3.604402143431094e-05, + 0.017872920354533223, + 0.015596389132661061, + -0.05328027163908902, + 0.05645081430217192, + -0.05734065113534405, + -0.050635552274912655, + 0.030131288157661782, + ], + saturation: [ + -6.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, + 0.0, 0.0, 0.0, + ], + }, + Qpv2Knot { + target: 85.0, + rows: 120, + lower: [ + 10.175413663955043, + -0.08228926450496746, + 0.0022687432880238896, + -0.13988211404720818, + -0.3076403802731068, + -0.06931057897914127, + 0.24846174892886125, + 0.10118970836416148, + 0.01714749498939879, + -0.04059456546624838, + 0.023948490246457652, + 0.032747942818945344, + 0.009869219636491513, + -0.01746895538671726, + -0.10094028942433975, + 0.08243659064140416, + -0.05761006875297063, + 0.1397602914320146, + -0.12825290922215582, + -0.1790292259105136, + ], + median: [ + 10.357081047686453, + -0.027905864283035566, + -0.12908467086954853, + -0.46603046061674286, + 0.0779756166300174, + -0.0029746917860300126, + 0.38046624283275293, + 0.0775637003066588, + 0.0010359257772471573, + -0.13721640679997588, + -0.01188775720990278, + -0.009256909203737758, + -0.013510311358363936, + 0.00833541366196533, + -0.009748458679475859, + 0.010611252598317671, + 0.04738672782396992, + 0.05164329678211175, + 0.00785653659463203, + -0.186589504365656, + ], + candidate: [ + 10.559703243656253, + -0.06933995160515433, + 0.13409481268876539, + -0.25453416624877684, + -0.09339691707056483, + -0.00944582956204468, + 0.08909729054020864, + 0.07452275903068231, + 0.0015514437296684763, + -0.20180215242465555, + 0.0028171503107653774, + 0.008544032212880864, + -0.00631294438531654, + 0.015266467397116429, + -0.021866683559479638, + -0.014991574919852809, + 0.07821162327823956, + 0.059255403532592794, + 0.0046808983555065185, + -0.16926610160313862, + ], + upper: [ + 10.60933158542999, + -0.05320482494928357, + 0.08165484531392139, + -0.15157121758954228, + -0.04457075214341936, + -0.012279166627028998, + -0.00469827180321417, + 0.053429486870360574, + -0.003065210953815426, + -0.1170463120141785, + 0.0024890621194557138, + 0.012482006834372618, + 0.014671341805076553, + 0.020702418247172033, + -0.04143225474540787, + -0.01835549989417961, + 0.07350561505793092, + 0.08129301706816137, + 0.0036466903063132336, + -0.15396948678804814, + ], + beta: [ + -0.3449100447999933, + -0.038269239125313985, + 0.40789382632432863, + -0.813234037882281, + -0.20364398146863608, + 0.0013113631258636054, + 0.6638402356517423, + 0.0155425400805581, + -0.017248888193020114, + -0.0413544204655544, + -0.0007077226572284477, + 0.027678918068457018, + 3.604402143431094e-05, + 0.017872920354533223, + 0.01824972784950946, + -0.05328027163908902, + 0.05645081430217192, + -0.05734065113534405, + -0.050635552274912655, + 0.030131288157661782, + ], + saturation: [ + -6.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, + 0.0, 0.0, 0.0, + ], + }, + Qpv2Knot { + target: 90.0, + rows: 120, + lower: [ + 10.717934141378654, + -0.08228926450496746, + 0.012336268964125348, + -0.13988211404720818, + -0.3076403802731068, + -0.06033536582230862, + 0.24846174892886125, + 0.10118970836416148, + 0.01714749498939879, + -0.04059456546624838, + 0.023948490246457652, + 0.032747942818945344, + 0.009869219636491513, + -0.01746895538671726, + -0.09937669159771208, + 0.08243659064140416, + -0.05761006875297063, + 0.1397602914320146, + -0.12825290922215582, + -0.1790292259105136, + ], + median: [ + 10.891541007009502, + -0.027905864283035566, + -0.1356888996398463, + -0.46603046061674286, + 0.0779756166300174, + -0.00026402861616535607, + 0.38046624283275293, + 0.0775637003066588, + 0.0010359257772471573, + -0.13721640679997588, + -0.01188775720990278, + -0.009256909203737758, + -0.013510311358363936, + 0.00833541366196533, + -0.007807850441808824, + 0.010611252598317671, + 0.04738672782396992, + 0.05164329678211175, + 0.00785653659463203, + -0.186589504365656, + ], + candidate: [ + 11.100085942883167, + -0.06933995160515433, + 0.13379492079146577, + -0.25453416624877684, + -0.09339691707056483, + -0.004335336748118172, + 0.08909729054020864, + 0.07452275903068231, + 0.0015514437296684763, + -0.20180215242465555, + 0.0028171503107653774, + 0.008544032212880864, + -0.00631294438531654, + 0.015266467397116429, + -0.017569608465613187, + -0.014991574919852809, + 0.07821162327823956, + 0.059255403532592794, + 0.0046808983555065185, + -0.16926610160313862, + ], + upper: [ + 11.14789998897887, + -0.05320482494928357, + 0.07481897687914642, + -0.15157121758954228, + -0.04457075214341936, + -0.009688309684690744, + -0.00469827180321417, + 0.053429486870360574, + -0.003065210953815426, + -0.1170463120141785, + 0.0024890621194557138, + 0.012482006834372618, + 0.014671341805076553, + 0.020702418247172033, + -0.03749114093706316, + -0.01835549989417961, + 0.07350561505793092, + 0.08129301706816137, + 0.0036466903063132336, + -0.15396948678804814, + ], + beta: [ + -0.3263577201257119, + -0.038269239125313985, + 0.38818649413592055, + -0.813234037882281, + -0.20364398146863608, + -0.009660059623909615, + 0.6638402356517423, + 0.0155425400805581, + -0.017248888193020114, + -0.0413544204655544, + -0.0007077226572284477, + 0.027678918068457018, + 3.604402143431094e-05, + 0.017872920354533223, + 0.0219893987316943, + -0.05328027163908902, + 0.05645081430217192, + -0.05734065113534405, + -0.050635552274912655, + 0.030131288157661782, + ], + saturation: [ + -6.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, + 0.0, 0.0, 0.0, + ], + }, + Qpv2Knot { + target: 95.0, + rows: 120, + lower: [ + 11.6453790233237, + -0.08228926450496746, + 0.0295468177833915, + -0.13988211404720818, + -0.3076403802731068, + -0.04499213758841707, + 0.24846174892886125, + 0.10118970836416148, + 0.01714749498939879, + -0.04059456546624838, + 0.023948490246457652, + 0.032747942818945344, + 0.009869219636491513, + -0.01746895538671726, + -0.09670370345795948, + 0.08243659064140416, + -0.05761006875297063, + 0.1397602914320146, + -0.12825290922215582, + -0.1790292259105136, + ], + median: [ + 11.805206342247498, + -0.027905864283035566, + -0.14697890329333843, + -0.46603046061674286, + 0.0779756166300174, + 0.004369880679768804, + 0.38046624283275293, + 0.0775637003066588, + 0.0010359257772471573, + -0.13721640679997588, + -0.01188775720990278, + -0.009256909203737758, + -0.013510311358363936, + 0.00833541366196533, + -0.004490358747427376, + 0.010611252598317671, + 0.04738672782396992, + 0.05164329678211175, + 0.00785653659463203, + -0.186589504365656, + ], + candidate: [ + 12.023876268862551, + -0.06933995160515433, + 0.13328225220684728, + -0.25453416624877684, + -0.09339691707056483, + 0.004401108421659668, + 0.08909729054020864, + 0.07452275903068231, + 0.0015514437296684763, + -0.20180215242465555, + 0.0028171503107653774, + 0.008544032212880864, + -0.00631294438531654, + 0.015266467397116429, + -0.010223710072863377, + -0.014991574919852809, + 0.07821162327823956, + 0.059255403532592794, + 0.0046808983555065185, + -0.16926610160313862, + ], + upper: [ + 12.068588756010806, + -0.05320482494928357, + 0.06313298260370558, + -0.15157121758954228, + -0.04457075214341936, + -0.00525921048748719, + -0.00469827180321417, + 0.053429486870360574, + -0.003065210953815426, + -0.1170463120141785, + 0.0024890621194557138, + 0.012482006834372618, + 0.014671341805076553, + 0.020702418247172033, + -0.03075376238119675, + -0.01835549989417961, + 0.07350561505793092, + 0.08129301706816137, + 0.0036466903063132336, + -0.15396948678804814, + ], + beta: [ + -0.2946423116142096, + -0.038269239125313985, + 0.3544965872374229, + -0.813234037882281, + -0.20364398146863608, + -0.02841583069683718, + 0.6638402356517423, + 0.0155425400805581, + -0.017248888193020114, + -0.0413544204655544, + -0.0007077226572284477, + 0.027678918068457018, + 3.604402143431094e-05, + 0.017872920354533223, + 0.028382408330727544, + -0.05328027163908902, + 0.05645081430217192, + -0.05734065113534405, + -0.050635552274912655, + 0.030131288157661782, + ], + saturation: [ + -6.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, + 0.0, 0.0, 0.0, + ], + }, +]; diff --git a/JPXL/crates/jpxl-encode-policy/src/quantize.rs b/JPXL/crates/jpxl-encode-policy/src/quantize.rs index 63382226..22565ba7 100644 --- a/JPXL/crates/jpxl-encode-policy/src/quantize.rs +++ b/JPXL/crates/jpxl-encode-policy/src/quantize.rs @@ -613,11 +613,12 @@ impl HfQuantizer { /// Phase-2: quantize a contiguous coefficient lane (one channel of one /// varblock) into `out`, skipping LLF cells when `skip_llf` is set. - /// Each row's HF span goes through [`Self::choose_run`] (Phase 36; before - /// that, [`Self::choose_lane4`] batches with scalar tails), so batches - /// never cross the top-left LLF boundary. /// - /// Same integer rules as [`Self::choose`], applied cell-by-cell. + /// The top-left `n_blocks × n_blocks` LLF is still handled per row so a + /// run never crosses that boundary. Every remaining row is one contiguous + /// HF rectangle in coefficient order, so it is one [`Self::choose_run`] + /// (Phase 36's AVX2 8-wide kernel, padded tail) instead of one call per + /// row. Same integer rules as [`Self::choose`], applied cell-by-cell. /// /// # Errors /// @@ -661,14 +662,16 @@ impl HfQuantizer { what: "a zero-width HF coefficient lane", }); } - for row_start in (0..cells).step_by(side) { - let row = row_start / side; + // Rows that contain the LLF prefix. Remaining rows are a contiguous + // HF rectangle (`first_cell = n_blocks * side`) and share one run. + let llf_rows = if skip_llf { n_blocks } else { 0 }; + for row in 0..llf_rows { + let row_start = row.saturating_mul(side); + if row_start >= cells { + break; + } let row_end = row_start.saturating_add(side).min(cells); - let first_hf = if skip_llf && row < n_blocks { - row_start.saturating_add(n_blocks).min(row_end) - } else { - row_start - }; + let first_hf = row_start.saturating_add(n_blocks).min(row_end); if let Some(llf) = out.get_mut(row_start..first_hf) { llf.fill(0); } @@ -679,9 +682,6 @@ impl HfQuantizer { *slot = self.reconstruct(0, channel, row_start + offset); } } - // Phase 36: the whole HF span of the row goes through the run - // kernel (vector chunks plus a padded final chunk), which is - // cell-for-cell identical to the scalar `choose` loop it replaces. let (Some(targets), Some(slots)) = ( coeffs.get(first_hf..row_end), out.get_mut(first_hf..row_end), @@ -693,6 +693,17 @@ impl HfQuantizer { .and_then(|r| r.get_mut(first_hf..row_end)); self.choose_run(channel, first_hf, targets, slots, recon_row)?; } + let bulk_start = llf_rows.saturating_mul(side); + if bulk_start < cells { + let (Some(targets), Some(slots)) = ( + coeffs.get(bulk_start..cells), + out.get_mut(bulk_start..cells), + ) else { + return Ok(()); + }; + let recon_bulk = recon_out.and_then(|r| r.get_mut(bulk_start..cells)); + self.choose_run(channel, bulk_start, targets, slots, recon_bulk)?; + } Ok(()) } @@ -1852,10 +1863,45 @@ mod tests { }) .collect::>() .expect("scalar reference"); + let expected_recon: Vec = expected + .iter() + .enumerate() + .map(|(cell, &qi)| q.reconstruct(qi, channel, cell).to_bits()) + .collect(); let mut actual = vec![i32::MIN; cells]; - q.quantize_lane(channel, &coeffs, &mut actual, side, n, true) - .expect("lane quantization"); + let mut recon = vec![f32::NAN; cells]; + q.quantize_lane_with_recon( + channel, + &coeffs, + &mut actual, + Some(&mut recon), + side, + n, + true, + ) + .expect("lane quantization"); assert_eq!(actual, expected, "{transform:?} channel {channel}"); + let actual_recon: Vec = recon.iter().map(|v| v.to_bits()).collect(); + assert_eq!( + actual_recon, expected_recon, + "{transform:?} channel {channel} recon" + ); + + // skip_llf = false: the whole lane is one HF run, including + // the top-left cells that skip_llf would have forced to zero. + let expected_full: Vec = coeffs + .iter() + .enumerate() + .map(|(cell, &target)| q.choose(target, channel, cell)) + .collect::>() + .expect("full-lane scalar reference"); + let mut actual_full = vec![i32::MIN; cells]; + q.quantize_lane(channel, &coeffs, &mut actual_full, side, n, false) + .expect("full-lane quantization"); + assert_eq!( + actual_full, expected_full, + "{transform:?} channel {channel} skip_llf=false" + ); } } } diff --git a/JPXL/crates/jpxl-encode-policy/src/rate.rs b/JPXL/crates/jpxl-encode-policy/src/rate.rs index f55281b0..a1fe521b 100644 --- a/JPXL/crates/jpxl-encode-policy/src/rate.rs +++ b/JPXL/crates/jpxl-encode-policy/src/rate.rs @@ -418,7 +418,7 @@ pub struct RateProbeStats { pub candidate_cache_entries: u64, /// Retained f32 coefficient payload, excluding collection metadata. pub candidate_payload_bytes: u64, - /// Dense coefficient-arena allocations (one per populated transform bank). + /// Dense coefficient-arena allocations (one per reserved transform bank). pub candidate_allocations: u64, /// Cover/CfL builds on the bounded anchored path. pub structural_builds: u32, @@ -700,14 +700,14 @@ fn cold_step(current: Rung, up: bool) -> Rung { /// lower segment. Interpolating on the index therefore aims badly across that /// kink — which is exactly where high-rate targets live. This quantity is /// smooth and strictly increasing across the whole ladder. -pub(crate) fn effective_scale(rung: Rung) -> u64 { +pub fn effective_scale(rung: Rung) -> u64 { let (scale, mul) = rung_fields(rung); u64::from(scale) * u64::from(mul) } /// The inverse of [`effective_scale`], rounded down to the finest /// representable rung whose effective scale does not exceed `scale`. -pub(crate) fn rung_for_effective_scale(scale: u64) -> Rung { +pub fn rung_for_effective_scale(scale: u64) -> Rung { let max = u64::from(MAX_GLOBAL_SCALE); if scale <= max { return Rung::new(u32::try_from(scale.saturating_sub(1)).unwrap_or(u32::MAX)); diff --git a/JPXL/crates/jpxl-encode-policy/src/source.rs b/JPXL/crates/jpxl-encode-policy/src/source.rs index 1d153b1a..8e137653 100644 --- a/JPXL/crates/jpxl-encode-policy/src/source.rs +++ b/JPXL/crates/jpxl-encode-policy/src/source.rs @@ -384,6 +384,14 @@ impl PreparedFrame { let max = f32::from(u16::MAX).min(((1u32 << bits_per_sample) - 1) as f32); let divisor = if max > 0.0 { max } else { 1.0 }; + // One transfer-curve evaluation per representable sample value, not + // per sample: each entry is exactly the `srgb_to_linear(v / divisor)` + // the per-sample path evaluated (dividing, per the note above), so the + // planes are bit-identical to it. + let lut: Vec = (0..=u32::from(u16::MAX)) + .map(|v| jpxl_core::color::srgb_to_linear(v as f32 / divisor)) + .collect(); + let convert_band = |src: &[u16], r: &mut [f32], g: &mut [f32], b: &mut [f32]| -> bool { let mut grayscale = true; for (((px, rs), gs), bs) in src @@ -398,9 +406,9 @@ impl PreparedFrame { px.get(2).copied().unwrap_or(0), ); grayscale &= cr == cg && cg == cb; - *rs = jpxl_core::color::srgb_to_linear(f32::from(cr) / divisor); - *gs = jpxl_core::color::srgb_to_linear(f32::from(cg) / divisor); - *bs = jpxl_core::color::srgb_to_linear(f32::from(cb) / divisor); + *rs = lut.get(usize::from(cr)).copied().unwrap_or(0.0); + *gs = lut.get(usize::from(cg)).copied().unwrap_or(0.0); + *bs = lut.get(usize::from(cb)).copied().unwrap_or(0.0); } linear_srgb_to_xyb_planes(r, g, b); grayscale diff --git a/JPXL/crates/jpxl-encode-policy/tests/rate_loop.rs b/JPXL/crates/jpxl-encode-policy/tests/rate_loop.rs index 7d56cbb6..91901e33 100644 --- a/JPXL/crates/jpxl-encode-policy/tests/rate_loop.rs +++ b/JPXL/crates/jpxl-encode-policy/tests/rate_loop.rs @@ -158,8 +158,10 @@ fn target_rate_is_byte_identical_across_executor_widths() { serial.stats.candidate_cache_entries, parallel.stats.candidate_cache_entries ); - assert_eq!(serial.stats.candidate_allocations, 3); - assert_eq!(parallel.stats.candidate_allocations, 3); + // Fast's fixed DCT8x8 cover reserves one bank per LF group, not the + // three-family hierarchical set. + assert_eq!(serial.stats.candidate_allocations, 1); + assert_eq!(parallel.stats.candidate_allocations, 1); #[cfg(feature = "anchor-sketch")] for outcome in [&serial, ¶llel] { assert_eq!(outcome.status, RateStatus::RescuedFreshStructure); @@ -206,6 +208,7 @@ fn balanced_anchor_is_exact_and_decodable() { assert!(outcome.stats.structural_builds <= 2); assert!(outcome.stats.full_prices <= 4); assert!(outcome.stats.exact_candidates <= 6); + assert_eq!(outcome.stats.candidate_allocations, 3); } #[cfg(feature = "anchor-sketch")] diff --git a/JPXL/crates/jpxl-encode/src/container.rs b/JPXL/crates/jpxl-encode/src/container.rs index 019f2a59..7ca68082 100644 --- a/JPXL/crates/jpxl-encode/src/container.rs +++ b/JPXL/crates/jpxl-encode/src/container.rs @@ -93,6 +93,27 @@ pub fn wrap_fragmented(codestream: &[u8], level: u8, fragment_size: usize) -> Ve out } +/// Appends an `Exif` box (18181-2 9.5) to an already-wrapped container. +/// +/// The payload must be the raw Exif block as JEITA CP-3451E / CP-3461B +/// define it — beginning with the TIFF header — and the box's +/// `tiff_header_offset` field is therefore written as zero. Clause 5 leaves +/// box order free after the file type box, so appending after the codestream +/// boxes is conforming; [`wrap`] writes the `jxlc` box with an explicit +/// length precisely so a box can follow it. +/// +/// Per 9.5, codestream fields (orientation, dimensions) take precedence over +/// Exif equivalents at decode time; the caller is responsible for not +/// contradicting them. +pub fn append_exif(container: &mut Vec, exif_payload: &[u8]) { + // LBox counts the 8-byte header and the u32 tiff_header_offset. + let explicit = u32::try_from(exif_payload.len() + 12).ok(); + container.extend_from_slice(&explicit.unwrap_or(0).to_be_bytes()); + container.extend_from_slice(b"Exif"); + container.extend_from_slice(&0u32.to_be_bytes()); // tiff_header_offset + container.extend_from_slice(exif_payload); +} + /// The signature box, the file type box and — when the level is not the /// default — the level box, i.e. everything before the codestream boxes. fn preamble(payload_hint: usize, level: u8) -> Vec { @@ -197,6 +218,32 @@ mod tests { } } + /// The claim the archive pipeline needs: an appended `Exif` box survives + /// the decoder's box walk with its payload intact, and does not disturb + /// codestream extraction — for a whole `jxlc` file and a fragmented one. + #[test] + fn an_appended_exif_box_round_trips_and_leaves_the_codestream_alone() { + let codestream: Vec = (0..500u32).map(|i| (i % 251) as u8).collect(); + // A plausible payload head: little-endian TIFF header, then filler. + let mut exif = vec![0x49, 0x49, 0x2A, 0x00]; + exif.extend((0..64u32).map(|i| (i % 7) as u8)); + + for mut file in [ + wrap(&codestream, DEFAULT_LEVEL), + wrap_fragmented(&codestream, EXTENDED_LEVEL, 128), + ] { + append_exif(&mut file, &exif); + let mut guard = AllocGuard::new(&Limits::relaxed()); + let tree = jpxl_decode::container::BoxTree::parse(&file, &mut guard).expect("parses"); + tree.validate().expect("conforming with an Exif box"); + assert_eq!(tree.codestream(&mut guard).expect("codestream"), codestream); + let parsed = tree.exif().expect("well-formed Exif boxes"); + assert_eq!(parsed.len(), 1); + assert_eq!(parsed[0].tiff_header_offset, 0); + assert_eq!(parsed[0].payload, &exif[..]); + } + } + #[test] fn the_last_fragment_is_the_only_one_carrying_the_final_marker() { // Written out by hand rather than trusting the decoder, because the diff --git a/JPXL/crates/jpxl-encode/src/headers.rs b/JPXL/crates/jpxl-encode/src/headers.rs index 51504d55..03a6641f 100644 --- a/JPXL/crates/jpxl-encode/src/headers.rs +++ b/JPXL/crates/jpxl-encode/src/headers.rs @@ -78,10 +78,20 @@ const ENUM_SPEC: U32Spec = U32Spec::new([ }, ]); +/// Table E.3 `ColourSpace`: `kRGB`, the bundle default. +const COLOUR_SPACE_RGB: u32 = 0; /// Table E.3 `ColourSpace`: `kGrey`. const COLOUR_SPACE_GREY: u32 = 1; /// Table E.4 `WhitePoint`: `kD65`, the bundle default. const WHITE_POINT_D65: u32 = 1; +/// Table E.5 `Primaries`: `kSRGB`, the bundle default. +const PRIMARIES_SRGB: u32 = 1; +/// Table E.5 `Primaries`: ITU-R BT.2100-2 (the Rec.2020 gamut). +const PRIMARIES_2100: u32 = 9; +/// Table E.5 `Primaries`: SMPTE ST 428-1 (P3). +const PRIMARIES_P3: u32 = 11; +/// Table E.6 `TransferFunction`: gamma exponent 1. +const TRANSFER_FUNCTION_LINEAR: u32 = 8; /// Table E.6 `TransferFunction`: `kSrgb`, the bundle default. const TRANSFER_FUNCTION_SRGB: u32 = 13; /// Table E.8 `RenderingIntent`: `kRelative`, the bundle default. @@ -147,14 +157,59 @@ impl ColourShape { } } +/// The colour space the caller's samples are in, signalled declaratively in +/// the `ColourEncoding` bundle (18181-1 E.2). +/// +/// The Modular path stores samples untouched, so this is a pure header claim: +/// it changes how a colour-managed viewer interprets the decoded samples and +/// nothing else. Every variant is a named row combination of Tables E.3–E.8 — +/// D65 white point and relative-colorimetric intent throughout, which is what +/// every real capture pipeline this encoder feeds produces. +/// +/// The XYB-encoded VarDCT path is not covered by this enum: its forward +/// transform and its perceptual metric are defined on sRGB input, so the +/// policy layer rejects a non-sRGB lossy request instead of mis-tagging it. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +#[non_exhaustive] +pub enum ColourSpace { + /// IEC 61966-2-1 sRGB — the Table E.1 `all_default` bundle. + #[default] + Srgb, + /// sRGB primaries with a linear transfer function (gamma 1). + LinearSrgb, + /// Display P3: SMPTE ST 428-1 primaries, D65, the sRGB transfer function. + DisplayP3, + /// Rec.2020 gamut: ITU-R BT.2100-2 primaries, D65, the sRGB transfer + /// function (the shape a "rec2020 working space" still image carries). + Rec2020, +} + +impl ColourSpace { + /// `(primaries, transfer_function)` Table E.5/E.6 rows for an RGB image. + const fn rgb_rows(self) -> (u32, u32) { + match self { + Self::Srgb => (PRIMARIES_SRGB, TRANSFER_FUNCTION_SRGB), + Self::LinearSrgb => (PRIMARIES_SRGB, TRANSFER_FUNCTION_LINEAR), + Self::DisplayP3 => (PRIMARIES_P3, TRANSFER_FUNCTION_SRGB), + Self::Rec2020 => (PRIMARIES_2100, TRANSFER_FUNCTION_SRGB), + } + } +} + /// Writes an `ImageMetadata` bundle describing a non-XYB integer still image /// (18181-1 D.3, Table D.3). /// /// # Errors /// -/// [`EncodeError::Unsupported`] if `bits_per_sample` is outside the 1..=16 -/// range this encoder handles, or a bit writer error. -pub fn write_metadata(w: &mut BitWriter, shape: ColourShape, bits_per_sample: u32) -> Result<()> { +/// [`EncodeError::Unsupported`] for a greyscale image in a non-sRGB colour +/// space, [`EncodeError::ValueOutOfRange`] if `bits_per_sample` is outside +/// the 1..=16 range this encoder handles, or a bit writer error. +pub fn write_metadata( + w: &mut BitWriter, + shape: ColourShape, + bits_per_sample: u32, + colour: ColourSpace, +) -> Result<()> { if bits_per_sample == 0 || bits_per_sample > 16 { return Err(EncodeError::ValueOutOfRange { what: "bits_per_sample", @@ -174,7 +229,7 @@ pub fn write_metadata(w: &mut BitWriter, shape: ColourShape, bits_per_sample: u3 w.write_u32(&NUM_EXTRA_SPEC, 0)?; // no extra channels w.write_bool(false); // xyb_encoded: samples are stored as-is - write_colour_encoding(w, shape)?; + write_colour_encoding(w, shape, colour)?; // tone_mapping is guarded by extra_fields, which is false. w.write_u64(0)?; // extensions (B.3) @@ -190,29 +245,56 @@ pub fn write_metadata(w: &mut BitWriter, shape: ColourShape, bits_per_sample: u3 /// /// Only through the bit writer. pub fn write_grey8_metadata(w: &mut BitWriter) -> Result<()> { - write_metadata(w, ColourShape::Grey, 8) + write_metadata(w, ColourShape::Grey, 8, ColourSpace::Srgb) } /// Writes a `ColourEncoding` bundle (18181-1 E.2, Table E.1). /// /// The Table E.1 defaults are exactly sRGB: `want_icc` false, `kRGB`, `kD65`, -/// `kSRGB` primaries, the sRGB transfer function and `kRelative`. An RGB image -/// is therefore one `all_default` bit. `kGrey` differs from the default, so -/// the greyscale form is written out — with `has_primaries` false, which skips -/// the primaries rows but not the white point. -fn write_colour_encoding(w: &mut BitWriter, shape: ColourShape) -> Result<()> { - if shape == ColourShape::Rgb { +/// `kSRGB` primaries, the sRGB transfer function and `kRelative`. An sRGB RGB +/// image is therefore one `all_default` bit. Every other case writes the +/// bundle out: `kGrey` with `has_primaries` false (which skips the primaries +/// rows but not the white point), and a non-default RGB colour space with the +/// primaries and transfer-function rows of [`ColourSpace::rgb_rows`]. +/// +/// # Errors +/// +/// [`EncodeError::Unsupported`] for a greyscale image in a non-sRGB colour +/// space — a combination nothing feeds this encoder — or a bit writer error. +fn write_colour_encoding(w: &mut BitWriter, shape: ColourShape, colour: ColourSpace) -> Result<()> { + if shape == ColourShape::Grey { + if colour != ColourSpace::Srgb { + return Err(EncodeError::unsupported( + "a greyscale image can only be signalled as sRGB", + "E.2", + )); + } + w.write_bool(false); // all_default (the default is kRGB) + w.write_bool(false); // want_icc + w.write_u32(&ENUM_SPEC, COLOUR_SPACE_GREY)?; + w.write_u32(&ENUM_SPEC, WHITE_POINT_D65)?; + // primaries / red / green / blue: skipped for kGrey. + // CustomTransferFunction (Table E.7). + w.write_bool(false); // have_gamma + w.write_u32(&ENUM_SPEC, TRANSFER_FUNCTION_SRGB)?; + w.write_u32(&ENUM_SPEC, RENDERING_INTENT_RELATIVE)?; + return Ok(()); + } + if colour == ColourSpace::Srgb { w.write_bool(true); // all_default return Ok(()); } - w.write_bool(false); // all_default (the default is kRGB) + let (primaries, transfer) = colour.rgb_rows(); + w.write_bool(false); // all_default w.write_bool(false); // want_icc - w.write_u32(&ENUM_SPEC, COLOUR_SPACE_GREY)?; + w.write_u32(&ENUM_SPEC, COLOUR_SPACE_RGB)?; w.write_u32(&ENUM_SPEC, WHITE_POINT_D65)?; - // primaries / red / green / blue: skipped for kGrey. + // white: skipped, the white point is not kCustom. + w.write_u32(&ENUM_SPEC, primaries)?; + // red / green / blue: skipped, the primaries are not kCustom. // CustomTransferFunction (Table E.7). w.write_bool(false); // have_gamma - w.write_u32(&ENUM_SPEC, TRANSFER_FUNCTION_SRGB)?; + w.write_u32(&ENUM_SPEC, transfer)?; w.write_u32(&ENUM_SPEC, RENDERING_INTENT_RELATIVE)?; Ok(()) } @@ -225,10 +307,20 @@ mod tests { use jpxl_decode::headers::decode_image_headers; fn headers(width: u32, height: u32, shape: ColourShape, bits: u32) -> Vec { + headers_in(width, height, shape, bits, ColourSpace::Srgb) + } + + fn headers_in( + width: u32, + height: u32, + shape: ColourShape, + bits: u32, + colour: ColourSpace, + ) -> Vec { let mut w = BitWriter::new(); write_signature(&mut w).expect("signature"); write_size_header(&mut w, width, height).expect("size"); - write_metadata(&mut w, shape, bits).expect("metadata"); + write_metadata(&mut w, shape, bits, colour).expect("metadata"); w.zero_pad_to_byte(); w.into_bytes() } @@ -270,7 +362,7 @@ mod tests { let mut w = BitWriter::new(); write_signature(&mut w).expect("signature"); write_size_header(&mut w, 64, 64).expect("size"); - write_metadata(&mut w, shape, bits).expect("metadata"); + write_metadata(&mut w, shape, bits, ColourSpace::Srgb).expect("metadata"); let written = w.bit_len(); w.zero_pad_to_byte(); let bytes = w.into_bytes(); @@ -282,15 +374,64 @@ mod tests { } } + /// Each declarative colour space comes back from the decoder as the + /// Table E.5/E.6 rows it names, with the header ending exactly where the + /// writer says — the bit count is what catches a wrong E.1 conditional. + #[test] + fn non_default_colour_spaces_round_trip_through_the_decoder() { + use jpxl_decode::headers::colour::CustomTransferFunction; + use jpxl_decode::headers::enums::{Primaries, TransferFunction, WhitePoint}; + + for (colour, primaries, tf) in [ + ( + ColourSpace::LinearSrgb, + Primaries::KSrgb, + TransferFunction::KLinear, + ), + ( + ColourSpace::DisplayP3, + Primaries::KP3, + TransferFunction::KSrgb, + ), + ( + ColourSpace::Rec2020, + Primaries::K2100, + TransferFunction::KSrgb, + ), + ] { + let bytes = headers_in(64, 64, ColourShape::Rgb, 16, colour); + let mut r = BitReader::new(&bytes); + let parsed = decode_image_headers(&mut r, &Limits::default()) + .unwrap_or_else(|e| panic!("{colour:?}: {e}")); + + let ce = parsed.metadata.colour_encoding; + assert!(!ce.all_default, "{colour:?}"); + assert!(!ce.want_icc, "{colour:?}"); + assert!(!ce.is_grey(), "{colour:?}"); + assert_eq!(ce.white_point, WhitePoint::KD65, "{colour:?}"); + assert_eq!(ce.primaries, primaries, "{colour:?}"); + assert_eq!(ce.tf, CustomTransferFunction::Enumerated(tf), "{colour:?}"); + } + } + + #[test] + fn a_greyscale_image_rejects_a_non_srgb_colour_space() { + let mut w = BitWriter::new(); + assert!(matches!( + write_metadata(&mut w, ColourShape::Grey, 8, ColourSpace::Rec2020), + Err(EncodeError::Unsupported { .. }) + )); + } + #[test] fn an_unrepresentable_bit_depth_is_rejected() { let mut w = BitWriter::new(); assert!(matches!( - write_metadata(&mut w, ColourShape::Grey, 0), + write_metadata(&mut w, ColourShape::Grey, 0, ColourSpace::Srgb), Err(EncodeError::ValueOutOfRange { .. }) )); assert!(matches!( - write_metadata(&mut w, ColourShape::Grey, 17), + write_metadata(&mut w, ColourShape::Grey, 17, ColourSpace::Srgb), Err(EncodeError::ValueOutOfRange { .. }) )); } diff --git a/JPXL/crates/jpxl-encode/src/lib.rs b/JPXL/crates/jpxl-encode/src/lib.rs index b3a1cbf4..94ad3308 100644 --- a/JPXL/crates/jpxl-encode/src/lib.rs +++ b/JPXL/crates/jpxl-encode/src/lib.rs @@ -76,6 +76,7 @@ use modular::{ModularSource, Plane, Rect}; use section::SectionStore; pub use error::{EncodeError, Result}; +pub use headers::ColourSpace; pub use lossless::Effort; pub use resources::{EncodeExecutor, EncodeResources, ParallelAxis}; @@ -152,12 +153,12 @@ impl Image { /// # Errors /// /// As [`Image::new`]. - pub fn from_interleaved( + pub fn from_interleaved>( width: u32, height: u32, channels: usize, bits_per_sample: u32, - samples: &[u16], + samples: &[S], ) -> Result { if channels == 0 { return Err(EncodeError::unsupported( @@ -174,7 +175,7 @@ impl Image { let planes: Vec = (0..channels) .map(|c| { (0..per_plane) - .map(|i| samples.get(i * channels + c).map_or(0, |&s| i32::from(s))) + .map(|i| samples.get(i * channels + c).map_or(0, |&s| s.into())) .collect() }) .collect(); @@ -251,6 +252,13 @@ pub struct EncodeOptions { /// Every level is exact-lossless, so this changes only the byte count and /// the encode time, never the decoded pixels. pub effort: Effort, + /// The colour space the caller's samples are in, signalled declaratively + /// in the image header (18181-1 E.2). Default: sRGB, the Table E.1 + /// `all_default` bundle. + /// + /// The Modular path stores samples untouched, so this changes only how a + /// colour-managed viewer interprets them — never the decoded values. + pub colour_space: headers::ColourSpace, /// Measurement escape hatch: override individual levers of the search /// budget [`EncodeOptions::effort`] would have selected. /// @@ -402,7 +410,13 @@ fn encode_codestream(image: &Image, options: &EncodeOptions) -> Result> } else { source_planes }; - encode_codestream_with_plan_resources(image, &planes, &plan, options.resources) + encode_codestream_with_plan_resources( + image, + &planes, + &plan, + options.resources, + options.colour_space, + ) } /// Emits the codestream a validated plan describes. @@ -418,10 +432,17 @@ pub fn encode_codestream_with_plan( planes: &[Plane], plan: &ValidatedLosslessPlan, ) -> Result> { - encode_codestream_with_plan_resources(image, planes, plan, EncodeResources::serial()) + encode_codestream_with_plan_resources( + image, + planes, + plan, + EncodeResources::serial(), + ColourSpace::Srgb, + ) } -/// As [`encode_codestream_with_plan`], with an explicit resource policy. +/// As [`encode_codestream_with_plan`], with an explicit resource policy and +/// an explicit declarative colour space. /// /// # Errors /// @@ -431,6 +452,7 @@ pub fn encode_codestream_with_plan_resources( planes: &[Plane], plan: &ValidatedLosslessPlan, resources: EncodeResources, + colour_space: ColourSpace, ) -> Result> { let (width, height) = (image.width(), image.height()); let plan = plan.plan(); @@ -456,7 +478,7 @@ pub fn encode_codestream_with_plan_resources( let mut w = BitWriter::new(); headers::write_signature(&mut w)?; headers::write_size_header(&mut w, width, height)?; - headers::write_metadata(&mut w, image.shape(), image.bits_per_sample())?; + headers::write_metadata(&mut w, image.shape(), image.bits_per_sample(), colour_space)?; // F.1: every frame starts on a byte boundary. w.zero_pad_to_byte(); diff --git a/JPXL/crates/jpxl-jpeg/Cargo.toml b/JPXL/crates/jpxl-jpeg/Cargo.toml new file mode 100644 index 00000000..c41b41db --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/Cargo.toml @@ -0,0 +1,15 @@ +[package] +name = "jpxl-jpeg" +description = "Clean-room ISO/IEC 10918-1 (JPEG1) coefficient-level codec: parse a JPEG bitstream into typed structures and re-emit it byte-for-byte" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true + +# Phase A is a self-contained 10918-1 codec: it deliberately depends on +# nothing in the workspace so all JPEG-1 subtleties are isolated before any +# JPEG XL codestream work (see JPXL/docs/jpeg-recompression-plan.md §3). +[dependencies] + +[lints] +workspace = true diff --git a/JPXL/crates/jpxl-jpeg/examples/roundtrip_sweep.rs b/JPXL/crates/jpxl-jpeg/examples/roundtrip_sweep.rs new file mode 100644 index 00000000..86460123 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/examples/roundtrip_sweep.rs @@ -0,0 +1,151 @@ +//! Archive roundtrip sweep: `serialize(parse(x)) == x` over a directory of +//! JPEG files, with deterministic ordering and a coverage/refusal summary. +//! +//! ```text +//! cargo run --release -p jpxl-jpeg --example roundtrip_sweep -- [stride] +//! ``` +//! +//! Files are collected recursively, sorted by path, and (optionally) sampled +//! by the given stride, so the sample is a pure function of the archive +//! contents. Every file is classified as byte-identical, refused (typed +//! `JpegError::Unsupported`), parse error, or mismatch; the exit code is +//! nonzero if any mismatch occurs. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +fn collect(dir: &Path, out: &mut Vec) -> std::io::Result<()> { + for entry in std::fs::read_dir(dir)? { + let path = entry?.path(); + if path.is_dir() { + collect(&path, out)?; + } else if path + .extension() + .and_then(|e| e.to_str()) + .is_some_and(|e| e.eq_ignore_ascii_case("jpg") || e.eq_ignore_ascii_case("jpeg")) + { + out.push(path); + } + } + Ok(()) +} + +/// Coverage attributes of one parsed JPEG, for the summary table. +fn classify(jpeg: &jpxl_jpeg::Jpeg, tally: &mut BTreeMap<&'static str, usize>) { + let mut bump = |key: &'static str| *tally.entry(key).or_insert(0) += 1; + let mut sof_code = None; + for segment in &jpeg.segments { + match segment { + jpxl_jpeg::Segment::Sof(frame) => { + sof_code = Some(frame.code); + match frame.components.len() { + 1 => bump("grayscale"), + 3 => { + let factors: Vec<(u8, u8)> = + frame.components.iter().map(|c| (c.h, c.v)).collect(); + match factors.first() { + Some(&(2, 2)) => bump("chroma-420"), + Some(&(2, 1)) => bump("chroma-422"), + Some(&(1, 2)) => bump("chroma-440"), + Some(&(1, 1)) => bump("chroma-444"), + _ => bump("chroma-other"), + } + } + _ => bump("components-other"), + } + } + jpxl_jpeg::Segment::Dri(interval) if *interval > 0 => bump("restart-interval"), + jpxl_jpeg::Segment::App(app) => { + if app.code == 0xE1 && app.payload.starts_with(b"Exif\0") { + bump("exif"); + } else if app.code == 0xE1 && app.payload.starts_with(b"http://ns.adobe.com/xap/") { + bump("xmp"); + } else if app.code == 0xE2 && app.payload.starts_with(b"ICC_PROFILE\0") { + bump("icc"); + } + } + _ => {} + } + } + match sof_code { + Some(0xC0) => bump("baseline"), + Some(0xC1) => bump("extended-sequential"), + Some(0xC2) => bump("progressive"), + Some(_) => bump("sof-other"), + None => bump("no-sof"), + } + if !jpeg.tail.is_empty() { + bump("trailing-data"); + } +} + +fn main() -> std::process::ExitCode { + let mut args = std::env::args().skip(1); + let Some(dir) = args.next() else { + eprintln!("usage: roundtrip_sweep [stride]"); + return std::process::ExitCode::FAILURE; + }; + let stride: usize = args.next().and_then(|s| s.parse().ok()).unwrap_or(1).max(1); + let mut files = Vec::new(); + if let Err(e) = collect(Path::new(&dir), &mut files) { + eprintln!("walk failed: {e}"); + return std::process::ExitCode::FAILURE; + } + files.sort(); + let sample: Vec<&PathBuf> = files.iter().step_by(stride).collect(); + let (mut identical, mut mismatches, mut parse_errors) = (0usize, 0usize, 0usize); + let mut refusals: BTreeMap = BTreeMap::new(); + let mut coverage: BTreeMap<&'static str, usize> = BTreeMap::new(); + for path in &sample { + let data = match std::fs::read(path) { + Ok(data) => data, + Err(e) => { + eprintln!("READ-ERROR {} {e}", path.display()); + parse_errors += 1; + continue; + } + }; + match jpxl_jpeg::parse(&data) { + Ok(jpeg) => match jpxl_jpeg::serialize(&jpeg) { + Ok(bytes) if bytes == data => { + identical += 1; + classify(&jpeg, &mut coverage); + } + Ok(_) => { + mismatches += 1; + eprintln!("MISMATCH {}", path.display()); + } + Err(e) => { + mismatches += 1; + eprintln!("SERIALIZE-ERROR {} {e}", path.display()); + } + }, + Err(jpxl_jpeg::JpegError::Unsupported(reason)) => { + *refusals.entry(reason).or_insert(0) += 1; + } + Err(e) => { + parse_errors += 1; + eprintln!("PARSE-ERROR {} {e}", path.display()); + } + } + } + println!( + "swept {} of {} files (stride {stride}): {identical} byte-identical, \ + {mismatches} mismatches, {parse_errors} parse errors, {} refused", + sample.len(), + files.len(), + refusals.values().sum::() + ); + for (reason, count) in &refusals { + println!(" refused[{count}]: {reason}"); + } + println!("coverage of the byte-identical set:"); + for (key, count) in &coverage { + println!(" {key}: {count}"); + } + if mismatches == 0 { + std::process::ExitCode::SUCCESS + } else { + std::process::ExitCode::FAILURE + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/bitio.rs b/JPXL/crates/jpxl-jpeg/src/bitio.rs new file mode 100644 index 00000000..0a1a8807 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/bitio.rs @@ -0,0 +1,292 @@ +//! Bit-level I/O for entropy-coded JPEG data (10918-1 F.1.2.1 / E.1.3). +//! +//! Entropy-coded segments are read most-significant-bit first. Two wire +//! conventions live here: +//! +//! * **Byte stuffing** — a `0xFF` data byte is written as `0xFF 0x00`, so a +//! `0xFF` in the entropy stream can only be the start of a marker. +//! * **Marker termination** — a `0xFF` followed by any non-zero code ends the +//! current entropy segment; the reader stops without consuming the marker. +//! +//! Both the reader and writer track just enough state to reproduce the +//! stream's padding bits exactly, which is what makes a byte-for-byte +//! round-trip possible. + +// Every `as u32`/`as u8` narrowing in this module is immediately masked to the +// bits actually requested (`read_bits`/`put_bits` take `n <= 32`), so the +// truncation is intentional and lossless within the requested width. +#![allow(clippy::cast_possible_truncation)] + +use crate::error::{JpegError, Result}; + +/// The padding written after the real bits of an entropy segment to reach a +/// byte boundary (10918-1 F.1.2.3): `nbits` bits with value `bits` +/// (right-aligned). Standard encoders pad with 1-bits; capturing the exact +/// value keeps the round-trip faithful to non-standard encoders too. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct Padding { + /// Number of padding bits (`0..=7`). + pub nbits: u8, + /// The padding bit pattern, right-aligned in the low `nbits` bits. + pub bits: u8, +} + +/// Reads bits MSB-first from an entropy-coded region, unstuffing `0xFF 0x00` +/// and stopping at the first marker. +pub struct EntropyReader<'a> { + data: &'a [u8], + /// Index of the next byte to consider. + pos: usize, + /// Pending bits, right-aligned: the next bit to emit is bit `nbits - 1`. + acc: u64, + /// Count of valid pending bits in `acc`. + nbits: u32, + /// Set once a marker terminated the segment; holds the marker code byte. + marker: Option, +} + +impl<'a> EntropyReader<'a> { + /// Creates a reader over `data`, starting at byte `start`. + #[must_use] + pub fn new(data: &'a [u8], start: usize) -> Self { + Self { + data, + pos: start, + acc: 0, + nbits: 0, + marker: None, + } + } + + /// The marker code that terminated the segment, if one has been reached. + #[must_use] + pub fn marker(&self) -> Option { + self.marker + } + + /// The byte offset of the next unconsumed byte (points at the terminating + /// marker's `0xFF` once a marker has been reached). + #[must_use] + pub fn byte_pos(&self) -> usize { + self.pos + } + + /// Pulls one raw entropy byte, handling stuffing and marker detection. + /// + /// Returns `Ok(None)` when a marker is reached (recording its code); + /// otherwise the unstuffed data byte. + fn next_data_byte(&mut self) -> Result> { + if self.marker.is_some() { + return Ok(None); + } + let Some(&b) = self.data.get(self.pos) else { + return Err(JpegError::UnexpectedEof { + while_reading: "entropy-coded data", + }); + }; + if b != 0xFF { + self.pos += 1; + return Ok(Some(b)); + } + // b == 0xFF: skip any run of fill 0xFF bytes to find the code byte. + let mut j = self.pos + 1; + while self.data.get(j) == Some(&0xFF) { + j += 1; + } + let Some(&code) = self.data.get(j) else { + return Err(JpegError::UnexpectedEof { + while_reading: "marker code after 0xFF", + }); + }; + if code == 0x00 { + // Stuffed 0xFF: consume `FF 00`, yield a literal 0xFF byte. + self.pos = j + 1; + return Ok(Some(0xFF)); + } + // A real marker. Leave `pos` on the 0xFF immediately preceding the + // code so the caller sees a contiguous `FF code`, and record it. + self.pos = j - 1; + self.marker = Some(code); + Ok(None) + } + + /// Refills `acc` up to at least `need` valid bits, or until a marker. + fn fill(&mut self, need: u32) -> Result<()> { + while self.nbits < need { + match self.next_data_byte()? { + Some(byte) => { + self.acc = (self.acc << 8) | u64::from(byte); + self.nbits += 8; + } + None => break, + } + } + Ok(()) + } + + /// Reads `n` bits (`0..=32`) MSB-first as an unsigned value. + pub fn read_bits(&mut self, n: u32) -> Result { + if n == 0 { + return Ok(0); + } + self.fill(n)?; + if self.nbits < n { + return Err(JpegError::Malformed(format!( + "F.1.2.1: entropy data exhausted needing {n} bits (only {} available before marker)", + self.nbits + ))); + } + self.nbits -= n; + let mask = if n == 32 { u32::MAX } else { (1u32 << n) - 1 }; + Ok(((self.acc >> self.nbits) as u32) & mask) + } + + /// Reads a single bit. + pub fn read_bit(&mut self) -> Result { + self.read_bits(1) + } + + /// Consumes the remaining sub-byte bits as segment padding and returns + /// them (10918-1 F.1.2.3). Leaves the reader byte-aligned. + pub fn take_padding(&mut self) -> Result { + // Any whole bytes still buffered would mean the decoder under-read the + // segment — a bug worth surfacing rather than silently dropping. + if self.nbits >= 8 { + return Err(JpegError::Malformed(format!( + "F.1.2.3: {} bits buffered at a byte-alignment point (expected < 8)", + self.nbits + ))); + } + let nbits = self.nbits; + let bits = (self.acc & ((1u64 << nbits) - 1)) as u8; + self.nbits = 0; + self.acc = 0; + Ok(Padding { + nbits: nbits as u8, + bits, + }) + } + + /// Ensures the terminating marker is detected while byte-aligned, and + /// returns its code. Call only after [`Self::take_padding`] (i.e. no + /// pending bits): the next bytes must be a marker. + pub fn probe_marker(&mut self) -> Result { + if let Some(c) = self.marker { + return Ok(c); + } + let Some(&b) = self.data.get(self.pos) else { + return Err(JpegError::UnexpectedEof { + while_reading: "marker after entropy segment", + }); + }; + if b != crate::marker::MARKER_PREFIX { + return Err(JpegError::Malformed(format!( + "expected a marker prefix 0xFF after an entropy segment, found 0x{b:02X}" + ))); + } + let mut j = self.pos + 1; + while self.data.get(j) == Some(&0xFF) { + j += 1; + } + let Some(&code) = self.data.get(j) else { + return Err(JpegError::UnexpectedEof { + while_reading: "marker code after entropy segment", + }); + }; + self.pos = j - 1; + self.marker = Some(code); + Ok(code) + } + + /// After a restart marker, advances past its two bytes and resumes reading + /// a fresh (byte-aligned) entropy segment. + pub fn resume_after_restart(&mut self) -> Result<()> { + let code = self.marker.ok_or_else(|| { + JpegError::Malformed("resume_after_restart called with no marker pending".into()) + })?; + if !crate::marker::is_restart(code) { + return Err(JpegError::Malformed(format!( + "expected a restart marker to resume past, found 0xFF{code:02X}" + ))); + } + // `pos` is on the marker's 0xFF; skip the two marker bytes. + self.pos += 2; + self.marker = None; + self.acc = 0; + self.nbits = 0; + Ok(()) + } +} + +/// Writes bits MSB-first into a byte buffer, stuffing `0xFF` data bytes as +/// `0xFF 0x00`. +pub struct EntropyWriter<'a> { + out: &'a mut Vec, + /// Pending bits, right-aligned in `acc`. + acc: u64, + nbits: u32, +} + +impl<'a> EntropyWriter<'a> { + /// Creates a writer appending to `out`. + pub fn new(out: &'a mut Vec) -> Self { + Self { + out, + acc: 0, + nbits: 0, + } + } + + /// Writes the low `n` bits (`0..=32`) of `value`, MSB-first. + pub fn put_bits(&mut self, value: u32, n: u32) { + if n == 0 { + return; + } + let mask = if n == 32 { u32::MAX } else { (1u32 << n) - 1 }; + self.acc = (self.acc << n) | u64::from(value & mask); + self.nbits += n; + while self.nbits >= 8 { + self.nbits -= 8; + let byte = ((self.acc >> self.nbits) & 0xFF) as u8; + self.out.push(byte); + if byte == 0xFF { + self.out.push(0x00); + } + } + } + + /// Writes a single bit. + pub fn put_bit(&mut self, bit: u32) { + self.put_bits(bit & 1, 1); + } + + /// Writes raw bytes directly (e.g. a restart marker) while byte-aligned. + /// + /// Must be called only when no partial byte is pending — i.e. right after + /// [`Self::flush_padding`]. Errors otherwise rather than corrupting the + /// bit stream. + pub fn write_aligned_bytes(&mut self, bytes: &[u8]) -> Result<()> { + if self.nbits != 0 { + return Err(JpegError::Encode( + "write_aligned_bytes called with a partial byte pending".into(), + )); + } + self.out.extend_from_slice(bytes); + Ok(()) + } + + /// Emits the captured segment [`Padding`] to reach a byte boundary. + /// + /// Returns an error if the writer is not byte-aligned afterwards, which can + /// only happen if the re-encoded real bits diverged from the original. + pub fn flush_padding(&mut self, padding: Padding) -> Result<()> { + self.put_bits(u32::from(padding.bits), u32::from(padding.nbits)); + if self.nbits != 0 { + return Err(JpegError::Encode(format!( + "F.1.2.3: {} bits pending after padding — re-encoded bit count diverged", + self.nbits + ))); + } + Ok(()) + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/codec.rs b/JPXL/crates/jpxl-jpeg/src/codec.rs new file mode 100644 index 00000000..05d72399 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/codec.rs @@ -0,0 +1,448 @@ +//! Entropy decode and encode of scan data (10918-1 F.1.2 / F.2.2 baseline; +//! G.1.2 progressive). +//! +//! Decode and encode walk the *same* block enumeration ([`ScanGeom`]) in the +//! same order, so the encoder is a faithful inverse of the decoder rather than +//! a separate interpretation of the scan geometry. That shared layout is what +//! lets `serialize(parse(x)) == x` hold: the coefficients, the Huffman codes, +//! the restart cadence and the padding all line up bit-for-bit. The +//! progressive coder in [`crate::progressive`] reuses this same geometry. + +// Coefficient indices come from `ZIGZAG_TO_NATURAL[k]` with `k` bounded to +// `1..=63`, component/prediction indices are bounded by the component count +// asserted when the vectors were sized, and every `as u8` narrows a value the +// code has just checked to `0..=15` (magnitude categories, run lengths). Bulk +// plane/block access still goes through checked `.get()`. +#![allow(clippy::indexing_slicing, clippy::cast_possible_truncation)] + +use crate::bitio::{EntropyReader, EntropyWriter, Padding}; +use crate::coeff::{extend, magnitude_category, mantissa_bits, to_coeff}; +use crate::error::{JpegError, Result}; +use crate::frame::{FrameGeometry, FrameHeader}; +use crate::huffman::HuffmanTable; +use crate::marker::is_restart; +use crate::scan::ScanHeader; +use crate::segment::ComponentPlane; +use crate::units::ZIGZAG_TO_NATURAL; + +/// Per-scan-component geometry and table bindings. +pub(crate) struct CompGeom<'a> { + /// Index into `frame.components` / `planes`. + pub frame_idx: usize, + /// Horizontal sampling factor. + pub h: usize, + /// Vertical sampling factor. + pub v: usize, + /// DC Huffman table for this component (None for AC-only scans). + pub dc: Option<&'a HuffmanTable>, + /// AC Huffman table for this component (None for DC-only scans). + pub ac: Option<&'a HuffmanTable>, +} + +/// The block-visiting order for one scan. +pub(crate) struct ScanGeom<'a> { + pub interleaved: bool, + pub mcus_per_line: usize, + /// Number of units (MCUs if interleaved, else blocks). + pub num_units: usize, + pub comps: Vec>, + /// For non-interleaved scans, the single component's block-line width. + pub ni_blocks_per_line: usize, +} + +/// A single block reference: which plane and where in it. +pub(crate) struct BlockRef { + pub frame_idx: usize, + pub bx: usize, + pub by: usize, +} + +impl<'a> ScanGeom<'a> { + /// Resolves the scan geometry and Huffman-table bindings. + pub(crate) fn resolve( + frame: &FrameHeader, + geom: &FrameGeometry, + header: &ScanHeader, + dc_tables: &'a [Option; 4], + ac_tables: &'a [Option; 4], + ) -> Result { + let interleaved = header.is_interleaved(); + let need_dc = header.spectral_start == 0; + let need_ac = header.spectral_end != 0; + let mut comps = Vec::with_capacity(header.components.len()); + for sc in &header.components { + let frame_idx = frame + .components + .iter() + .position(|fc| fc.id == sc.id) + .ok_or_else(|| { + JpegError::Malformed(format!( + "B.2.3: scan component id {} not present in frame", + sc.id.0 + )) + })?; + let fc = &frame.components[frame_idx]; + let dc = if need_dc { + Some(pick_table(dc_tables, sc.dc_table, "DC")?) + } else { + None + }; + let ac = if need_ac { + Some(pick_table(ac_tables, sc.ac_table, "AC")?) + } else { + None + }; + comps.push(CompGeom { + frame_idx, + h: fc.h as usize, + v: fc.v as usize, + dc, + ac, + }); + } + + let (num_units, ni_blocks_per_line) = if interleaved { + (geom.mcus_per_line * geom.mcu_rows, 0) + } else { + let idx = comps + .first() + .ok_or_else(|| JpegError::Malformed("B.2.3: scan has no components".into()))? + .frame_idx; + let dims = frame.component_dims(idx, geom)?; + ( + dims.blocks_per_line_noninterleaved * dims.block_rows_noninterleaved, + dims.blocks_per_line_noninterleaved, + ) + }; + + Ok(Self { + interleaved, + mcus_per_line: geom.mcus_per_line, + num_units, + comps, + ni_blocks_per_line, + }) + } + + /// Appends the block references for unit `u` to `out` (in scan order). + pub(crate) fn unit_blocks(&self, u: usize, out: &mut Vec) { + out.clear(); + if self.interleaved { + let mx = u % self.mcus_per_line; + let my = u / self.mcus_per_line; + for c in &self.comps { + for vv in 0..c.v { + for hh in 0..c.h { + out.push(BlockRef { + frame_idx: c.frame_idx, + bx: mx * c.h + hh, + by: my * c.v + vv, + }); + } + } + } + } else if let Some(c) = self.comps.first() { + let bx = u % self.ni_blocks_per_line; + let by = u / self.ni_blocks_per_line; + out.push(BlockRef { + frame_idx: c.frame_idx, + bx, + by, + }); + } + } + + /// The `CompGeom` a block belongs to (by matching `frame_idx`). + pub(crate) fn comp_for(&self, frame_idx: usize) -> Result<&CompGeom<'a>> { + self.comps + .iter() + .find(|c| c.frame_idx == frame_idx) + .ok_or_else(|| JpegError::Malformed("no scan component for block".into())) + } +} + +fn pick_table<'a>( + tables: &'a [Option; 4], + id: u8, + which: &str, +) -> Result<&'a HuffmanTable> { + tables + .get(id as usize) + .and_then(|t| t.as_ref()) + .ok_or_else(|| { + JpegError::Malformed(format!( + "B.2.3: scan selects undefined {which} Huffman table {id}" + )) + }) +} + +pub(crate) fn block_mut( + planes: &mut [ComponentPlane], + frame_idx: usize, + bx: usize, + by: usize, +) -> Result<&mut [i16; 64]> { + let plane = planes + .get_mut(frame_idx) + .ok_or_else(|| JpegError::Malformed("plane index out of range".into()))?; + let stride = plane.blocks_per_line; + plane + .blocks + .get_mut(by * stride + bx) + .ok_or_else(|| JpegError::Malformed("block position out of range".into())) +} + +/// Whether this scan is baseline/sequential (a single full-band scan). +fn is_sequential(frame: &FrameHeader, header: &ScanHeader) -> bool { + !frame.kind.is_progressive() + && header.spectral_start == 0 + && header.spectral_end == 63 + && header.approx_high == 0 + && header.approx_low == 0 +} + +/// Decodes one scan's entropy data into `planes`, returning the per-segment +/// padding and (for a progressive AC scan) the observed EOB-run lengths. +#[allow(clippy::too_many_arguments)] +pub fn decode_scan( + frame: &FrameHeader, + geom: &FrameGeometry, + planes: &mut [ComponentPlane], + header: &ScanHeader, + dc_tables: &[Option; 4], + ac_tables: &[Option; 4], + restart_interval: u16, + reader: &mut EntropyReader<'_>, +) -> Result<(Vec, Vec)> { + if frame.kind.is_progressive() { + return crate::progressive::decode_progressive_scan( + frame, + geom, + planes, + header, + dc_tables, + ac_tables, + restart_interval, + reader, + ); + } + if !is_sequential(frame, header) { + return Err(JpegError::Malformed(format!( + "B.2.3: sequential frame with progressive scan parameters (Ss={}, Se={}, Ah={}, Al={})", + header.spectral_start, header.spectral_end, header.approx_high, header.approx_low + ))); + } + + let sg = ScanGeom::resolve(frame, geom, header, dc_tables, ac_tables)?; + let mut dc_pred = vec![0i32; frame.components.len()]; + let mut padding = Vec::new(); + let mut refs: Vec = Vec::new(); + let ri = restart_interval as usize; + + for u in 0..sg.num_units { + if ri != 0 && u != 0 && u % ri == 0 { + padding.push(finish_restart_segment(reader)?); + for p in dc_pred.iter_mut() { + *p = 0; + } + } + sg.unit_blocks(u, &mut refs); + for r in &refs { + let comp = sg.comp_for(r.frame_idx)?; + let dc_tbl = comp.dc.ok_or_else(|| miss("DC"))?; + let ac_tbl = comp.ac.ok_or_else(|| miss("AC"))?; + let block = block_mut(planes, r.frame_idx, r.bx, r.by)?; + decode_block_sequential(reader, dc_tbl, ac_tbl, &mut dc_pred[r.frame_idx], block)?; + } + } + padding.push(reader.take_padding()?); + Ok((padding, Vec::new())) +} + +pub(crate) fn miss(which: &str) -> JpegError { + JpegError::Malformed(format!("scan component lacks a {which} Huffman table")) +} + +/// Consumes segment padding and the restart marker, resuming a fresh segment. +pub(crate) fn finish_restart_segment(reader: &mut EntropyReader<'_>) -> Result { + let pad = reader.take_padding()?; + let code = reader.probe_marker()?; + if !is_restart(code) { + return Err(JpegError::Malformed(format!( + "expected a restart marker between intervals, found 0xFF{code:02X}" + ))); + } + reader.resume_after_restart()?; + Ok(pad) +} + +fn decode_block_sequential( + reader: &mut EntropyReader<'_>, + dc: &HuffmanTable, + ac: &HuffmanTable, + dc_pred: &mut i32, + block: &mut [i16; 64], +) -> Result<()> { + let t = u32::from(dc.decode(reader)?); + if t > 15 { + return Err(JpegError::Malformed(format!( + "F.1.2: DC magnitude category {t} > 15" + ))); + } + let diff = extend(reader.read_bits(t)?, t); + *dc_pred = dc_pred.wrapping_add(diff); + block[0] = to_coeff(*dc_pred)?; + + let mut k = 1usize; + while k <= 63 { + let rs = ac.decode(reader)?; + let r = (rs >> 4) as usize; + let s = u32::from(rs & 0x0F); + if s == 0 { + if r == 15 { + k += 16; + continue; + } + break; // EOB + } + k += r; + if k > 63 { + return Err(JpegError::Malformed( + "F.1.2: AC run overruns the block (k > 63)".into(), + )); + } + let val = extend(reader.read_bits(s)?, s); + block[ZIGZAG_TO_NATURAL[k] as usize] = to_coeff(val)?; + k += 1; + } + Ok(()) +} + +/// Encodes one scan's entropy data from `planes` into `out`, reproducing the +/// captured padding. `out` receives the raw entropy bytes and any restart +/// markers, but not the terminating marker. +#[allow(clippy::too_many_arguments)] +pub fn encode_scan( + out: &mut Vec, + frame: &FrameHeader, + geom: &FrameGeometry, + planes: &[ComponentPlane], + header: &ScanHeader, + dc_tables: &[Option; 4], + ac_tables: &[Option; 4], + restart_interval: u16, + padding: &[Padding], + eob_runs: &[u32], +) -> Result<()> { + if frame.kind.is_progressive() { + return crate::progressive::encode_progressive_scan( + out, + frame, + geom, + planes, + header, + dc_tables, + ac_tables, + restart_interval, + padding, + eob_runs, + ); + } + if !is_sequential(frame, header) { + return Err(JpegError::Encode( + "sequential frame with progressive scan parameters".into(), + )); + } + + let sg = ScanGeom::resolve(frame, geom, header, dc_tables, ac_tables)?; + let mut dc_pred = vec![0i32; frame.components.len()]; + let ri = restart_interval as usize; + let mut pad_iter = padding.iter(); + let mut refs: Vec = Vec::new(); + let mut restart_counter = 0u8; + + let mut writer = EntropyWriter::new(out); + for u in 0..sg.num_units { + if ri != 0 && u != 0 && u % ri == 0 { + let pad = *pad_iter + .next() + .ok_or_else(|| JpegError::Encode("missing padding for restart segment".into()))?; + writer.flush_padding(pad)?; + writer.write_aligned_bytes(&[0xFF, crate::marker::RST0 + (restart_counter & 7)])?; + restart_counter = restart_counter.wrapping_add(1); + for p in dc_pred.iter_mut() { + *p = 0; + } + } + sg.unit_blocks(u, &mut refs); + for r in &refs { + let comp = sg.comp_for(r.frame_idx)?; + let dc_tbl = comp.dc.ok_or_else(|| miss("DC"))?; + let ac_tbl = comp.ac.ok_or_else(|| miss("AC"))?; + let plane = planes + .get(r.frame_idx) + .ok_or_else(|| JpegError::Encode("plane index out of range".into()))?; + let block = plane + .block(r.bx, r.by) + .ok_or_else(|| JpegError::Encode("block position out of range".into()))?; + encode_block_sequential( + &mut writer, + dc_tbl, + ac_tbl, + &mut dc_pred[r.frame_idx], + block, + )?; + } + } + let pad = *pad_iter + .next() + .ok_or_else(|| JpegError::Encode("missing padding for final segment".into()))?; + writer.flush_padding(pad)?; + Ok(()) +} + +fn encode_block_sequential( + writer: &mut EntropyWriter<'_>, + dc: &HuffmanTable, + ac: &HuffmanTable, + dc_pred: &mut i32, + block: &[i16; 64], +) -> Result<()> { + let diff = i32::from(block[0]) - *dc_pred; + *dc_pred = i32::from(block[0]); + let s = magnitude_category(diff); + if s > 15 { + return Err(JpegError::Encode(format!( + "DC diff {diff} needs category {s} > 15" + ))); + } + dc.encode(writer, s as u8)?; + writer.put_bits(mantissa_bits(diff, s), s); + + let mut run = 0usize; + for k in 1..=63usize { + let coeff = i32::from(block[ZIGZAG_TO_NATURAL[k] as usize]); + if coeff == 0 { + run += 1; + continue; + } + while run >= 16 { + ac.encode(writer, 0xF0)?; // ZRL + run -= 16; + } + let s = magnitude_category(coeff); + if s == 0 || s > 15 { + return Err(JpegError::Encode(format!( + "AC coefficient {coeff} has category {s}" + ))); + } + let rs = ((run as u8) << 4) | (s as u8); + ac.encode(writer, rs)?; + writer.put_bits(mantissa_bits(coeff, s), s); + run = 0; + } + if run > 0 { + ac.encode(writer, 0x00)?; // EOB + } + Ok(()) +} diff --git a/JPXL/crates/jpxl-jpeg/src/coeff.rs b/JPXL/crates/jpxl-jpeg/src/coeff.rs new file mode 100644 index 00000000..458ebef9 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/coeff.rs @@ -0,0 +1,82 @@ +//! Coefficient magnitude coding helpers (10918-1 F.1.2.1, F.2.2.1). +//! +//! JPEG codes a signed coefficient (or DC difference) as a *magnitude +//! category* `S` — the number of significant bits — plus `S` mantissa bits. +//! These are the two directions of that mapping, shared by the baseline and +//! progressive scan coders. + +use crate::error::{JpegError, Result}; + +/// Rebuilds a signed value from `S` received bits (`EXTEND`, F.2.2.1). +#[must_use] +pub fn extend(received: u32, size: u32) -> i32 { + if size == 0 { + return 0; + } + let v = received as i32; + let threshold = 1i32 << (size - 1); + if v < threshold { + v + (-1i32 << size) + 1 + } else { + v + } +} + +/// The magnitude category `S` (number of significant bits) of a value. +#[must_use] +pub fn magnitude_category(value: i32) -> u32 { + let m = value.unsigned_abs(); + 32 - m.leading_zeros() +} + +/// The `S` mantissa bits JPEG appends for `value` (F.1.2.1): the low `size` +/// bits of `value` when non-negative, or of `value - 1` when negative. +#[must_use] +pub fn mantissa_bits(value: i32, size: u32) -> u32 { + if size == 0 { + return 0; + } + let mask = if size >= 32 { + u32::MAX + } else { + (1u32 << size) - 1 + }; + let raw = if value < 0 { value - 1 } else { value } as u32; + raw & mask +} + +/// Narrows a decoded coefficient to `i16`, rejecting values that do not fit — +/// which for an 8-bit frame signals corruption or an out-of-scope precision. +pub fn to_coeff(value: i32) -> Result { + i16::try_from(value).map_err(|_| { + JpegError::Malformed(format!( + "coefficient {value} exceeds the 16-bit range of an 8-bit-precision frame" + )) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn extend_is_inverse_of_magnitude_coding() { + // For every value, its (category, mantissa) must decode back exactly. + for v in -4096i32..=4096 { + let s = magnitude_category(v); + let bits = mantissa_bits(v, s); + assert_eq!(extend(bits, s), v, "roundtrip failed for {v} (S={s})"); + } + } + + #[test] + fn magnitude_category_matches_spec_table() { + assert_eq!(magnitude_category(0), 0); + assert_eq!(magnitude_category(1), 1); + assert_eq!(magnitude_category(-1), 1); + assert_eq!(magnitude_category(2), 2); + assert_eq!(magnitude_category(-3), 2); + assert_eq!(magnitude_category(1023), 10); + assert_eq!(magnitude_category(-2047), 11); + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/error.rs b/JPXL/crates/jpxl-jpeg/src/error.rs new file mode 100644 index 00000000..6a3c2f0d --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/error.rs @@ -0,0 +1,69 @@ +//! Error type for the JPEG-1 codec. +//! +//! Follows the project error style (AGENTS.md §6, mirroring +//! `jpxl_core::error`): a hand-rolled enum, no `thiserror`/`anyhow`, with a +//! human-readable message that names the 10918-1 clause and the invariant that +//! failed. Message text is for humans and is not stable. + +use core::fmt; + +/// Anything that can go wrong parsing or re-emitting a JPEG-1 bitstream. +#[derive(Debug)] +#[non_exhaustive] +pub enum JpegError { + /// The stream is truncated: more bytes were required than are present. + UnexpectedEof { + /// What the parser was reading when the bytes ran out. + while_reading: &'static str, + }, + /// The stream violates a requirement of ISO/IEC 10918-1. + /// + /// The message names the clause and the invariant that failed. + Malformed(String), + /// The stream is a well-formed JPEG the codec deliberately refuses. + /// + /// Arithmetic-coded, hierarchical, 12-bit-sample and lossless JPEG modes + /// are out of Phase-A scope; each is rejected here rather than mis-parsed. + Unsupported(String), + /// A resource limit (allocation cap) was exceeded while parsing. + /// + /// Decode paths are attacker-facing (AGENTS.md §6): oversized dimensions or + /// table counts are rejected before any large allocation. + LimitExceeded(String), + /// The re-emitter was asked for something it cannot serialize. + /// + /// A caller/logic error, not a stream error: a coefficient outside the + /// range its magnitude category can express, a Huffman symbol absent from + /// the table it must be coded with. + Encode(String), +} + +impl fmt::Display for JpegError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::UnexpectedEof { while_reading } => { + write!( + f, + "unexpected end of JPEG data while reading {while_reading}" + ) + } + Self::Malformed(msg) => write!(f, "malformed JPEG stream: {msg}"), + Self::Unsupported(msg) => write!(f, "unsupported JPEG feature: {msg}"), + Self::LimitExceeded(msg) => write!(f, "JPEG resource limit exceeded: {msg}"), + Self::Encode(msg) => write!(f, "cannot re-emit JPEG stream: {msg}"), + } + } +} + +impl std::error::Error for JpegError {} + +/// Shorthand for a result carrying a [`JpegError`]. +pub type Result = core::result::Result; + +/// Builds a [`JpegError::Unsupported`] with a formatted message. +macro_rules! unsupported { + ($($arg:tt)*) => { + $crate::error::JpegError::Unsupported(format!($($arg)*)) + }; +} +pub(crate) use unsupported; diff --git a/JPXL/crates/jpxl-jpeg/src/frame.rs b/JPXL/crates/jpxl-jpeg/src/frame.rs new file mode 100644 index 00000000..dc04a124 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/frame.rs @@ -0,0 +1,104 @@ +//! Frame header (`SOF`, 10918-1 B.2.2) and the frame geometry derived from it. + +use crate::error::{JpegError, Result}; +use crate::marker::SofKind; +use crate::units::ComponentId; + +/// One component's parameters from the frame header. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct FrameComponent { + /// Component identifier `Ci` (arbitrary label). + pub id: ComponentId, + /// Horizontal sampling factor `Hi`, `1..=4`. + pub h: u8, + /// Vertical sampling factor `Vi`, `1..=4`. + pub v: u8, + /// Quantization-table destination selector `Tqi`. + pub quant_id: u8, +} + +/// A parsed start-of-frame header. +#[derive(Clone, Debug)] +pub struct FrameHeader { + /// The `SOF` marker code byte (e.g. `0xC0`), preserved for re-emission. + pub code: u8, + /// The coding process / entropy class this code selects. + pub kind: SofKind, + /// Sample precision `P` in bits (Phase A supports 8 only). + pub precision: u8, + /// Number of image lines `Y`. + pub height: u16, + /// Number of samples per line `X`. + pub width: u16, + /// Per-component parameters, in header order. + pub components: Vec, +} + +/// The block/MCU geometry a frame implies (10918-1 A.2). +#[derive(Clone, Debug)] +pub struct FrameGeometry { + /// Maximum horizontal sampling factor `Hmax`. + pub hmax: u8, + /// Maximum vertical sampling factor `Vmax`. + pub vmax: u8, + /// Minimum-coded-units per line. + pub mcus_per_line: usize, + /// Minimum-coded-unit rows. + pub mcu_rows: usize, +} + +/// Per-component derived block dimensions. +#[derive(Clone, Copy, Debug)] +pub struct ComponentDims { + /// Blocks per line when the component is scanned interleaved (padded to + /// complete MCUs): `mcus_per_line * Hi`. + pub blocks_per_line_interleaved: usize, + /// Block rows when interleaved: `mcu_rows * Vi`. + pub block_rows_interleaved: usize, + /// Blocks per line when scanned non-interleaved (single-component scan): + /// `ceil(ceil(X * Hi / Hmax) / 8)`. + pub blocks_per_line_noninterleaved: usize, + /// Block rows when non-interleaved: `ceil(ceil(Y * Vi / Vmax) / 8)`. + pub block_rows_noninterleaved: usize, +} + +const fn div_ceil_usize(a: usize, b: usize) -> usize { + a.div_ceil(b) +} + +impl FrameHeader { + /// Computes the frame's MCU geometry. + pub fn geometry(&self) -> Result { + let hmax = self + .components + .iter() + .map(|c| c.h) + .max() + .ok_or_else(|| JpegError::Malformed("B.2.2: frame has no components".into()))?; + let vmax = self.components.iter().map(|c| c.v).max().unwrap_or(1); + let mcus_per_line = div_ceil_usize(self.width as usize, 8 * hmax as usize); + let mcu_rows = div_ceil_usize(self.height as usize, 8 * vmax as usize); + Ok(FrameGeometry { + hmax, + vmax, + mcus_per_line, + mcu_rows, + }) + } + + /// Computes block dimensions for the component at index `idx`. + pub fn component_dims(&self, idx: usize, geom: &FrameGeometry) -> Result { + let c = self + .components + .get(idx) + .ok_or_else(|| JpegError::Malformed(format!("component index {idx} out of range")))?; + let x_i = div_ceil_usize(self.width as usize * c.h as usize, geom.hmax as usize); + let y_i = div_ceil_usize(self.height as usize * c.v as usize, geom.vmax as usize); + Ok(ComponentDims { + blocks_per_line_interleaved: geom.mcus_per_line * c.h as usize, + block_rows_interleaved: geom.mcu_rows * c.v as usize, + blocks_per_line_noninterleaved: div_ceil_usize(x_i, 8), + block_rows_noninterleaved: div_ceil_usize(y_i, 8), + }) + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/huffman.rs b/JPXL/crates/jpxl-jpeg/src/huffman.rs new file mode 100644 index 00000000..2f03d6a1 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/huffman.rs @@ -0,0 +1,225 @@ +//! Huffman tables (10918-1 B.2.4.2 storage; Annex C code assignment; Annex F +//! decode). +//! +//! A `DHT` segment stores, per table, a class/id byte, sixteen count bytes +//! `L_i` (codes of each length 1..=16) and `sum(L_i)` value bytes `V`. The +//! canonical code assignment of Annex C is a pure function of the counts, so +//! the table is re-emitted byte-for-byte from `(class, id, counts, values)` +//! alone. This module derives both the decode tables (Annex F.2.2.1) and the +//! per-symbol encode table from that stored form. + +// Indexing here is into fixed `[_; 16]` / `[_; 17]` / `[_; 256]` tables and +// into `Vec`s whose length is established just above the access, always by an +// index the surrounding loop bounds to range (code length `1..=16`, symbol +// value `0..=255`). The `as` narrowings are range-checked first (a code of +// length `si <= 16` is `< 2^si`, and `l + 1 <= 16`). +#![allow(clippy::indexing_slicing, clippy::cast_possible_truncation)] + +use crate::error::{JpegError, Result}; + +/// A single Huffman table, as stored in a `DHT` segment plus the decode/encode +/// lookups derived from it. +#[derive(Clone, Debug)] +pub struct HuffmanTable { + /// Table class `Tc`: 0 = DC / lossless, 1 = AC. + pub class: u8, + /// Table destination identifier `Th`, `0..=3`. + pub id: u8, + /// `L_i`: the number of codes of each length `1..=16`. + pub counts: [u8; 16], + /// `V`: the symbol values, in canonical order. + pub values: Vec, + /// Per length `l` (indexed `1..=16`): smallest code of that length. + mincode: [i32; 17], + /// Per length `l`: largest code of that length, or `-1` if none. + maxcode: [i32; 17], + /// Per length `l`: index into `values` of the first symbol of that length. + valptr: [usize; 17], + /// Per symbol value: its code length (0 = value not present). + enc_size: [u8; 256], + /// Per symbol value: its code word, right-aligned. + enc_code: [u16; 256], +} + +impl HuffmanTable { + /// Builds a table from its stored form, deriving the lookups. + /// + /// Rejects tables whose counts describe more codes than fit in their length + /// (an over-subscribed / non-prefix code, 10918-1 C.2). + pub fn new(class: u8, id: u8, counts: [u8; 16], values: Vec) -> Result { + let total: usize = counts.iter().map(|&c| c as usize).sum(); + if total != values.len() { + return Err(JpegError::Malformed(format!( + "B.2.4.2: Huffman table sum(L_i)={total} but {} values present", + values.len() + ))); + } + if total == 0 { + return Err(JpegError::Malformed( + "B.2.4.2: empty Huffman table (sum(L_i)=0)".into(), + )); + } + if total > 256 { + return Err(JpegError::Malformed(format!( + "B.2.4.2: Huffman table has {total} codes (>256)" + ))); + } + + // Annex C: HUFFSIZE, then HUFFCODE (canonical, monotone by length). + let mut huffsize: Vec = Vec::with_capacity(total); + for (l, &n) in counts.iter().enumerate() { + for _ in 0..n { + huffsize.push((l + 1) as u8); + } + } + let mut huffcode = vec![0u16; total]; + let mut code: u32 = 0; + let mut si = huffsize[0]; + let mut k = 0usize; + loop { + while k < total && huffsize[k] == si { + if code >= (1u32 << si) { + return Err(JpegError::Malformed(format!( + "C.2: Huffman code of length {si} overflows (over-subscribed table)" + ))); + } + huffcode[k] = code as u16; + code += 1; + k += 1; + } + if k >= total { + break; + } + // Advance to the next used length, shifting the code left each step. + while k < total && huffsize[k] != si { + code <<= 1; + si += 1; + } + } + + // Annex F.2.2.1 decode tables. + let mut mincode = [0i32; 17]; + let mut maxcode = [-1i32; 17]; + let mut valptr = [0usize; 17]; + let mut p = 0usize; + for l in 1..=16usize { + let n = counts[l - 1] as usize; + if n == 0 { + maxcode[l] = -1; + continue; + } + valptr[l] = p; + mincode[l] = i32::from(huffcode[p]); + p += n; + maxcode[l] = i32::from(huffcode[p - 1]); + } + + // Encode lookup, keyed by symbol value. + let mut enc_size = [0u8; 256]; + let mut enc_code = [0u16; 256]; + for (idx, &v) in values.iter().enumerate() { + enc_size[v as usize] = huffsize[idx]; + enc_code[v as usize] = huffcode[idx]; + } + + Ok(Self { + class, + id, + counts, + values, + mincode, + maxcode, + valptr, + enc_size, + enc_code, + }) + } + + /// Decodes one symbol value from `reader` (Annex F.2.2.3, `DECODE`). + pub fn decode(&self, reader: &mut crate::bitio::EntropyReader<'_>) -> Result { + let mut code: i32 = reader.read_bit()? as i32; + let mut l = 1usize; + while code > self.maxcode[l] { + code = (code << 1) | (reader.read_bit()? as i32); + l += 1; + if l > 16 { + return Err(JpegError::Malformed( + "F.2.2.3: no Huffman code of length <= 16 matched".into(), + )); + } + } + let idx = self.valptr[l] + (code - self.mincode[l]) as usize; + self.values.get(idx).copied().ok_or_else(|| { + JpegError::Malformed(format!("F.2.2.3: decoded Huffman index {idx} out of range")) + }) + } + + /// The code length in bits for symbol `value`, or 0 if it is not in the + /// table. + #[must_use] + pub fn code_len(&self, value: u8) -> u8 { + self.enc_size[value as usize] + } + + /// Emits the code word for symbol `value` (Annex F.1.2.2, `ENCODE`). + pub fn encode(&self, writer: &mut crate::bitio::EntropyWriter<'_>, value: u8) -> Result<()> { + let size = self.enc_size[value as usize]; + if size == 0 { + return Err(JpegError::Encode(format!( + "symbol 0x{value:02X} is absent from Huffman table (class {}, id {})", + self.class, self.id + ))); + } + writer.put_bits(u32::from(self.enc_code[value as usize]), u32::from(size)); + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bitio::{EntropyReader, EntropyWriter}; + + #[test] + fn rejects_over_subscribed_table() { + // Three codes of length 1 is impossible (only two 1-bit codes exist). + let mut counts = [0u8; 16]; + counts[0] = 3; + assert!(HuffmanTable::new(0, 0, counts, vec![0, 1, 2]).is_err()); + } + + #[test] + fn encode_decode_symbol_stream_roundtrips() { + // One code each of lengths 1..=4: canonical codes 0, 10, 110, 1110. + let mut counts = [0u8; 16]; + counts[0] = 1; + counts[1] = 1; + counts[2] = 1; + counts[3] = 1; + let table = + HuffmanTable::new(1, 0, counts, vec![0x05, 0x11, 0xF0, 0x00]).expect("valid table"); + let symbols = [0x05u8, 0x11, 0x00, 0xF0, 0x05, 0xF0, 0x11, 0x00]; + + let mut bytes = Vec::new(); + let mut total_bits = 0u32; + { + let mut w = EntropyWriter::new(&mut bytes); + for &s in &symbols { + table.encode(&mut w, s).expect("symbol in table"); + total_bits += u32::from(table.code_len(s)); + } + let pad = u8::try_from((8 - (total_bits % 8)) % 8).expect("pad < 8"); + let bits = if pad == 0 { 0 } else { (1u8 << pad) - 1 }; + w.flush_padding(crate::bitio::Padding { nbits: pad, bits }) + .expect("aligned"); + } + // Terminate the entropy region with a marker so the reader has a bound. + bytes.push(0xFF); + bytes.push(0xD9); + + let mut r = EntropyReader::new(&bytes, 0); + for &expected in &symbols { + assert_eq!(table.decode(&mut r).expect("decode"), expected); + } + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/lib.rs b/JPXL/crates/jpxl-jpeg/src/lib.rs new file mode 100644 index 00000000..760a71a9 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/lib.rs @@ -0,0 +1,57 @@ +//! Clean-room ISO/IEC 10918-1 (JPEG-1 / ITU-T T.81) coefficient-level codec. +//! +//! This crate parses a JPEG-1 bitstream into typed structures — markers, +//! quantization tables, Huffman tables, frame and scan headers, and the +//! entropy-decoded DCT coefficient planes — and re-emits a **byte-for-byte +//! identical** JPEG from those structures. It has no pixel path and no IDCT: it +//! stops at quantized coefficients, which is exactly the surface JPEG XL's +//! lossless recompression needs (see `JPXL/docs/jpeg-recompression-plan.md`, +//! Phase A). +//! +//! ```no_run +//! # fn demo(bytes: &[u8]) -> Result<(), jpxl_jpeg::JpegError> { +//! let jpeg = jpxl_jpeg::parse(bytes)?; // -> typed model + coefficients +//! let round = jpxl_jpeg::serialize(&jpeg)?; // exact inverse +//! assert_eq!(bytes, round.as_slice()); // bit-exact round-trip +//! # Ok(()) +//! # } +//! ``` +//! +//! # Scope +//! +//! Supported: baseline sequential (SOF0), extended sequential Huffman (SOF1) +//! and progressive Huffman (SOF2) JPEGs at 8-bit precision, with 1–4 +//! components, any 4:4:4 / 4:2:2 / 4:2:0 / 4:4:0 subsampling, restart +//! intervals, and arbitrary `APPn` / `COM` / trailing-garbage payloads. +//! +//! Refused with a typed [`JpegError::Unsupported`], never mis-decoded: +//! arithmetic coding (SOF9–15, DAC), hierarchical mode (SOF5–7, DHP/EXP), +//! lossless mode (SOF3), and 12-bit samples. +//! +//! # Clean-room +//! +//! Every field order, table layout and entropy-coding rule here derives from +//! ITU-T T.81 / ISO-IEC 10918-1 and this workspace's own reasoning. libjxl is +//! not consulted as a source. + +pub mod bitio; +pub mod codec; +pub mod coeff; +pub mod error; +pub mod frame; +pub mod huffman; +pub mod limits; +pub mod marker; +pub mod parse; +pub mod progressive; +pub mod quant; +pub mod scan; +pub mod segment; +pub mod serialize; +pub mod units; + +pub use error::{JpegError, Result}; +pub use limits::Limits; +pub use parse::{parse, parse_with_limits}; +pub use segment::{ComponentPlane, Jpeg, Segment}; +pub use serialize::serialize; diff --git a/JPXL/crates/jpxl-jpeg/src/limits.rs b/JPXL/crates/jpxl-jpeg/src/limits.rs new file mode 100644 index 00000000..c7547213 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/limits.rs @@ -0,0 +1,31 @@ +//! Resource limits for attacker-facing JPEG parsing (AGENTS.md §6). +//! +//! A JPEG header can claim enormous dimensions or table counts in a handful of +//! bytes. These caps bound every allocation the parser makes on the size of +//! *validated* fields, rejecting hostile inputs before they can exhaust memory. + +/// Caps applied while parsing. +#[derive(Clone, Copy, Debug)] +pub struct Limits { + /// Maximum width or height in samples. + pub max_dimension: u32, + /// Maximum total blocks across all component planes. + pub max_total_blocks: u64, + /// Maximum input length in bytes. + pub max_input_len: usize, +} + +impl Default for Limits { + fn default() -> Self { + Self { + // 65535 is the largest a 16-bit SOF field can express anyway. + max_dimension: 65_535, + // ~1.07e9 blocks ≈ 68 gigapixels of luma; far above any real + // photo, far below anything that could OOM a 64-bit host at + // 128 bytes/block. + max_total_blocks: 1 << 30, + // 512 MiB: larger than any single JPEG this codec targets. + max_input_len: 512 << 20, + } + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/marker.rs b/JPXL/crates/jpxl-jpeg/src/marker.rs new file mode 100644 index 00000000..cd39ab67 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/marker.rs @@ -0,0 +1,110 @@ +//! JPEG marker bytes and start-of-frame classification (10918-1 Table B.1). +//! +//! A marker is two bytes: `0xFF` then a code byte. This module names the codes +//! this codec cares about and, crucially, decides which start-of-frame codes +//! are *in scope* for Phase A and which are refused. + +/// The marker-prefix byte. Every marker is `MARKER_PREFIX` then a code byte. +pub const MARKER_PREFIX: u8 = 0xFF; + +// Standalone markers (no length, no payload). +/// Start of image. +pub const SOI: u8 = 0xD8; +/// End of image. +pub const EOI: u8 = 0xD9; +/// Temporary (arithmetic), standalone. +pub const TEM: u8 = 0x01; + +/// First restart marker `RST0`; restart markers are `RST0..=RST7`. +pub const RST0: u8 = 0xD0; +/// Last restart marker `RST7`. +pub const RST7: u8 = 0xD7; + +// Segment markers (a 2-byte big-endian length follows the code). +/// Define quantization table(s). +pub const DQT: u8 = 0xDB; +/// Define Huffman table(s). +pub const DHT: u8 = 0xC4; +/// Define restart interval. +pub const DRI: u8 = 0xDD; +/// Start of scan. +pub const SOS: u8 = 0xDA; +/// Comment. +pub const COM: u8 = 0xFE; +/// First application segment `APP0`. +pub const APP0: u8 = 0xE0; +/// Last application segment `APP15`. +pub const APP15: u8 = 0xEF; +/// Define arithmetic conditioning (arithmetic coding — refused). +pub const DAC: u8 = 0xCC; +/// Define hierarchical progression (hierarchical mode — refused). +pub const DHP: u8 = 0xDE; +/// Expand reference component (hierarchical mode — refused). +pub const EXP: u8 = 0xDF; +/// Define number of lines. +pub const DNL: u8 = 0xDC; + +/// Whether `code` is a restart marker `RST0..=RST7`. +#[must_use] +pub fn is_restart(code: u8) -> bool { + (RST0..=RST7).contains(&code) +} + +/// Whether `code` names a start-of-frame segment (`SOF0..=SOF15`, excluding the +/// codes `0xC4` DHT, `0xC8` JPG and `0xCC` DAC that fall in the same range). +#[must_use] +pub fn is_sof(code: u8) -> bool { + matches!(code, 0xC0..=0xC3 | 0xC5..=0xC7 | 0xC9..=0xCB | 0xCD..=0xCF) +} + +/// The entropy-coding and process class a start-of-frame code selects, used to +/// decide Phase-A support (10918-1 Table B.1). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum SofKind { + /// Baseline sequential DCT, Huffman (SOF0). Supported. + BaselineSequential, + /// Extended sequential DCT, Huffman (SOF1). Supported (8-bit only). + ExtendedSequential, + /// Progressive DCT, Huffman (SOF2). Supported. + Progressive, + /// Lossless (sequential), Huffman (SOF3). Refused. + Lossless, + /// Differential/hierarchical Huffman (SOF5/6/7). Refused. + Hierarchical, + /// Arithmetic-coded (SOF9/10/11/13/14/15). Refused. + Arithmetic, +} + +impl SofKind { + /// Classifies a start-of-frame code byte. + /// + /// Returns `None` for codes that are not start-of-frame markers. + #[must_use] + pub fn from_code(code: u8) -> Option { + Some(match code { + 0xC0 => Self::BaselineSequential, + 0xC1 => Self::ExtendedSequential, + 0xC2 => Self::Progressive, + 0xC3 => Self::Lossless, + 0xC5..=0xC7 => Self::Hierarchical, + 0xC9..=0xCB | 0xCD..=0xCF => Self::Arithmetic, + _ => return None, + }) + } + + /// Whether this frame kind is decoded by Phase A (Huffman DCT modes only). + #[must_use] + pub fn is_supported(self) -> bool { + matches!( + self, + Self::BaselineSequential | Self::ExtendedSequential | Self::Progressive + ) + } + + /// Whether this scan is coded progressively (spectral selection / successive + /// approximation apply). + #[must_use] + pub fn is_progressive(self) -> bool { + matches!(self, Self::Progressive) + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/parse.rs b/JPXL/crates/jpxl-jpeg/src/parse.rs new file mode 100644 index 00000000..fb0b30bf --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/parse.rs @@ -0,0 +1,497 @@ +//! Parse a JPEG-1 bitstream into the typed [`Jpeg`] document model. +//! +//! The parser walks marker segments (B.1), builds the typed tables and +//! headers, and entropy-decodes every scan into coefficient planes. Everything +//! Annex-A reconstruction needs — segment order, table grouping, restart +//! cadence, padding bits, trailing bytes — is captured so the companion +//! [`crate::serialize`] can reproduce the input byte-for-byte. + +use crate::bitio::EntropyReader; +use crate::codec::decode_scan; +use crate::error::{JpegError, Result, unsupported}; +use crate::frame::{FrameComponent, FrameGeometry, FrameHeader}; +use crate::huffman::HuffmanTable; +use crate::limits::Limits; +use crate::marker::{self, SofKind}; +use crate::quant::QuantTable; +use crate::scan::{ScanComponent, ScanHeader}; +use crate::segment::{AppSegment, ComponentPlane, Jpeg, OtherSegment, ScanSegment, Segment}; +use crate::units::ComponentId; + +/// A forward-only cursor over the byte stream, for reading segment headers. +struct Cursor<'a> { + data: &'a [u8], + pos: usize, +} + +impl<'a> Cursor<'a> { + fn u8(&mut self, ctx: &'static str) -> Result { + let b = self + .data + .get(self.pos) + .copied() + .ok_or(JpegError::UnexpectedEof { while_reading: ctx })?; + self.pos += 1; + Ok(b) + } + + fn u16(&mut self, ctx: &'static str) -> Result { + let hi = self.u8(ctx)?; + let lo = self.u8(ctx)?; + Ok((u16::from(hi) << 8) | u16::from(lo)) + } + + fn bytes(&mut self, n: usize, ctx: &'static str) -> Result<&'a [u8]> { + let end = self + .pos + .checked_add(n) + .ok_or(JpegError::UnexpectedEof { while_reading: ctx })?; + let slice = self + .data + .get(self.pos..end) + .ok_or(JpegError::UnexpectedEof { while_reading: ctx })?; + self.pos = end; + Ok(slice) + } +} + +/// Parses `data` with default [`Limits`]. +pub fn parse(data: &[u8]) -> Result { + parse_with_limits(data, &Limits::default()) +} + +/// Parses `data`, enforcing `limits` on all size-bearing fields. +pub fn parse_with_limits(data: &[u8], limits: &Limits) -> Result { + if data.len() > limits.max_input_len { + return Err(JpegError::LimitExceeded(format!( + "input is {} bytes (cap {})", + data.len(), + limits.max_input_len + ))); + } + let mut cur = Cursor { data, pos: 0 }; + // SOI. + if cur.u8("SOI prefix")? != marker::MARKER_PREFIX || cur.u8("SOI code")? != marker::SOI { + return Err(JpegError::Malformed( + "stream does not start with SOI (0xFFD8)".into(), + )); + } + + let mut segments: Vec = Vec::new(); + let mut frame: Option = None; + let mut geom: Option = None; + let mut planes: Vec = Vec::new(); + let mut dc_tables: [Option; 4] = Default::default(); + let mut ac_tables: [Option; 4] = Default::default(); + let mut restart_interval: u16 = 0; + let tail; + + loop { + let code = read_marker(&mut cur)?; + match code { + marker::EOI => { + tail = data.get(cur.pos..).unwrap_or(&[]).to_vec(); + break; + } + marker::SOI => return Err(JpegError::Malformed("unexpected second SOI".into())), + marker::TEM => { + // Standalone, no payload. + segments.push(Segment::Other(OtherSegment { + code, + payload: Vec::new(), + })); + } + c if marker::is_restart(c) => { + return Err(JpegError::Malformed( + "restart marker outside an entropy-coded segment".into(), + )); + } + marker::DAC => { + return Err(unsupported!( + "arithmetic coding (DAC, 0xFFCC) is out of scope" + )); + } + marker::DHP | marker::EXP => { + return Err(unsupported!("hierarchical mode (DHP/EXP) is out of scope")); + } + c if marker::is_sof(c) => { + if frame.is_some() { + return Err(JpegError::Malformed("more than one SOF in a frame".into())); + } + let fh = parse_sof(&mut cur, c)?; + let g = fh.geometry()?; + planes = allocate_planes(&fh, &g, limits)?; + geom = Some(g); + segments.push(Segment::Sof(fh.clone())); + frame = Some(fh); + } + marker::DHT => { + let tables = parse_dht(&mut cur)?; + for t in &tables { + let slot = if t.class == 0 { + &mut dc_tables + } else { + &mut ac_tables + }; + // `id` is validated to 0..=3 in `parse_dht`, so this slot + // always exists. + if let Some(dst) = slot.get_mut(t.id as usize) { + *dst = Some(t.clone()); + } + } + segments.push(Segment::Dht(tables)); + } + marker::DQT => { + let tables = parse_dqt(&mut cur)?; + segments.push(Segment::Dqt(tables)); + } + marker::DRI => { + let len = cur.u16("DRI length")?; + if len != 4 { + return Err(JpegError::Malformed(format!( + "B.2.4.4: DRI length {len} != 4" + ))); + } + restart_interval = cur.u16("DRI interval")?; + segments.push(Segment::Dri(restart_interval)); + } + marker::SOS => { + let header = parse_sos(&mut cur)?; + let fh = frame + .as_ref() + .ok_or_else(|| JpegError::Malformed("SOS before SOF".into()))?; + let g = geom + .as_ref() + .ok_or_else(|| JpegError::Malformed("SOS before SOF".into()))?; + let mut reader = EntropyReader::new(data, cur.pos); + let (padding, eob_runs) = decode_scan( + fh, + g, + &mut planes, + &header, + &dc_tables, + &ac_tables, + restart_interval, + &mut reader, + )?; + cur.pos = reader.byte_pos(); + segments.push(Segment::Sos(ScanSegment { + header, + padding, + eob_runs, + })); + } + c @ (marker::APP0..=marker::APP15) => { + let payload = read_length_payload(&mut cur, "APPn")?; + segments.push(Segment::App(AppSegment { code: c, payload })); + } + marker::COM => { + let payload = read_length_payload(&mut cur, "COM")?; + segments.push(Segment::Com(payload)); + } + other => { + // Any other length-bearing marker (e.g. DNL): preserve verbatim. + let payload = read_length_payload(&mut cur, "marker segment")?; + segments.push(Segment::Other(OtherSegment { + code: other, + payload, + })); + } + } + } + + if frame.is_none() { + return Err(JpegError::Malformed("no SOF segment before EOI".into())); + } + + Ok(Jpeg { + segments, + frame, + planes, + tail, + }) +} + +/// Reads the next marker code, skipping any leading `0xFF` fill bytes. +fn read_marker(cur: &mut Cursor<'_>) -> Result { + let mut saw_ff = false; + loop { + let b = cur.u8("marker")?; + if b == marker::MARKER_PREFIX { + saw_ff = true; + continue; + } + if !saw_ff { + return Err(JpegError::Malformed(format!( + "expected a marker prefix 0xFF, found 0x{b:02X}" + ))); + } + if b == 0x00 { + return Err(JpegError::Malformed( + "0xFF00 where a marker was expected".into(), + )); + } + return Ok(b); + } +} + +/// Reads a `[length][payload]` segment body, returning the payload. +fn read_length_payload(cur: &mut Cursor<'_>, ctx: &'static str) -> Result> { + let len = cur.u16(ctx)?; + if len < 2 { + return Err(JpegError::Malformed(format!( + "{ctx} segment length {len} < 2" + ))); + } + Ok(cur.bytes(len as usize - 2, ctx)?.to_vec()) +} + +fn parse_sof(cur: &mut Cursor<'_>, code: u8) -> Result { + let kind = SofKind::from_code(code) + .ok_or_else(|| JpegError::Malformed(format!("0xFF{code:02X} is not a SOF marker")))?; + match kind { + SofKind::Arithmetic => { + return Err(unsupported!("arithmetic-coded JPEG (SOF 0x{code:02X})")); + } + SofKind::Hierarchical => { + return Err(unsupported!("hierarchical JPEG (SOF 0x{code:02X})")); + } + SofKind::Lossless => { + return Err(unsupported!("lossless JPEG (SOF3)")); + } + _ => {} + } + let len = cur.u16("SOF length")?; + let precision = cur.u8("SOF precision")?; + if precision != 8 { + return Err(unsupported!( + "{precision}-bit samples (only 8-bit precision is in scope)" + )); + } + let height = cur.u16("SOF Y")?; + let width = cur.u16("SOF X")?; + let nf = cur.u8("SOF Nf")?; + if nf == 0 || nf > 4 { + return Err(JpegError::Malformed(format!( + "B.2.2: component count Nf={nf} not in 1..=4" + ))); + } + let expected_len = 8 + 3 * nf as u16; + if len != expected_len { + return Err(JpegError::Malformed(format!( + "B.2.2: SOF length {len} != {expected_len} for Nf={nf}" + ))); + } + if width == 0 { + return Err(JpegError::Malformed("B.2.2: SOF X (width) is 0".into())); + } + // Y == 0 is legal (lines defined later by DNL); reject it as out of scope + // rather than mis-parse, since DNL handling is not implemented. + if height == 0 { + return Err(unsupported!("SOF Y (height) = 0 with deferred DNL height")); + } + let mut components = Vec::with_capacity(nf as usize); + for _ in 0..nf { + let id = ComponentId(cur.u8("SOF component id")?); + let hv = cur.u8("SOF sampling")?; + let h = hv >> 4; + let v = hv & 0x0F; + if !(1..=4).contains(&h) || !(1..=4).contains(&v) { + return Err(JpegError::Malformed(format!( + "B.2.2: sampling factors H={h} V={v} not in 1..=4" + ))); + } + let quant_id = cur.u8("SOF Tq")?; + if quant_id > 3 { + return Err(JpegError::Malformed(format!( + "B.2.2: quant selector Tq={quant_id} > 3" + ))); + } + components.push(FrameComponent { id, h, v, quant_id }); + } + // Component ids must be distinct (they are matched by SOS). + let mut seen: Vec = Vec::with_capacity(components.len()); + for c in &components { + if seen.contains(&c.id.0) { + return Err(JpegError::Malformed(format!( + "B.2.2: duplicate component id {}", + c.id.0 + ))); + } + seen.push(c.id.0); + } + Ok(FrameHeader { + code, + kind, + precision, + height, + width, + components, + }) +} + +fn allocate_planes( + fh: &FrameHeader, + geom: &FrameGeometry, + limits: &Limits, +) -> Result> { + let mut planes = Vec::with_capacity(fh.components.len()); + let mut total: u64 = 0; + for (idx, fc) in fh.components.iter().enumerate() { + let dims = fh.component_dims(idx, geom)?; + let bpl = dims.blocks_per_line_interleaved; + let bh = dims.block_rows_interleaved; + let count = bpl + .checked_mul(bh) + .ok_or_else(|| JpegError::LimitExceeded("block count overflow".into()))?; + total = total.saturating_add(count as u64); + if total > limits.max_total_blocks { + return Err(JpegError::LimitExceeded(format!( + "total blocks {total} exceed cap {}", + limits.max_total_blocks + ))); + } + planes.push(ComponentPlane { + id: fc.id, + h: fc.h, + v: fc.v, + blocks_per_line: bpl, + block_rows: bh, + blocks: vec![[0i16; 64]; count], + }); + } + Ok(planes) +} + +fn parse_dht(cur: &mut Cursor<'_>) -> Result> { + let len = cur.u16("DHT length")?; + if len < 2 { + return Err(JpegError::Malformed("DHT length < 2".into())); + } + let end = cur.pos + len as usize - 2; + let mut tables = Vec::new(); + while cur.pos < end { + let tc_th = cur.u8("DHT Tc/Th")?; + let class = tc_th >> 4; + let id = tc_th & 0x0F; + if class > 1 { + return Err(JpegError::Malformed(format!( + "B.2.4.2: Huffman table class Tc={class} > 1" + ))); + } + if id > 3 { + return Err(JpegError::Malformed(format!( + "B.2.4.2: Huffman table id Th={id} > 3" + ))); + } + let mut counts = [0u8; 16]; + let count_bytes = cur.bytes(16, "DHT counts")?; + counts.copy_from_slice(count_bytes); + let total: usize = counts.iter().map(|&c| c as usize).sum(); + let values = cur.bytes(total, "DHT values")?.to_vec(); + tables.push(HuffmanTable::new(class, id, counts, values)?); + } + if cur.pos != end { + return Err(JpegError::Malformed( + "DHT segment length does not match its tables".into(), + )); + } + Ok(tables) +} + +fn parse_dqt(cur: &mut Cursor<'_>) -> Result> { + let len = cur.u16("DQT length")?; + if len < 2 { + return Err(JpegError::Malformed("DQT length < 2".into())); + } + let end = cur.pos + len as usize - 2; + let mut tables = Vec::new(); + while cur.pos < end { + let pq_tq = cur.u8("DQT Pq/Tq")?; + let precision = pq_tq >> 4; + let id = pq_tq & 0x0F; + if precision > 1 { + return Err(JpegError::Malformed(format!( + "B.2.4.1: quant precision Pq={precision} > 1" + ))); + } + if id > 3 { + return Err(JpegError::Malformed(format!( + "B.2.4.1: quant table id Tq={id} > 3" + ))); + } + let mut values = [0u16; 64]; + if precision == 0 { + let raw = cur.bytes(64, "DQT 8-bit values")?; + for (dst, &src) in values.iter_mut().zip(raw) { + *dst = u16::from(src); + } + } else { + let raw = cur.bytes(128, "DQT 16-bit values")?; + for (dst, pair) in values.iter_mut().zip(raw.chunks_exact(2)) { + *dst = u16::from_be_bytes(pair.try_into().expect("chunks_exact(2)")); + } + } + tables.push(QuantTable { + precision, + id, + values, + }); + } + if cur.pos != end { + return Err(JpegError::Malformed( + "DQT segment length does not match its tables".into(), + )); + } + Ok(tables) +} + +fn parse_sos(cur: &mut Cursor<'_>) -> Result { + let len = cur.u16("SOS length")?; + let ns = cur.u8("SOS Ns")?; + if ns == 0 || ns > 4 { + return Err(JpegError::Malformed(format!( + "B.2.3: scan component count Ns={ns} not in 1..=4" + ))); + } + let expected_len = 6 + 2 * ns as u16; + if len != expected_len { + return Err(JpegError::Malformed(format!( + "B.2.3: SOS length {len} != {expected_len} for Ns={ns}" + ))); + } + let mut components = Vec::with_capacity(ns as usize); + for _ in 0..ns { + let id = ComponentId(cur.u8("SOS Cs")?); + let td_ta = cur.u8("SOS Td/Ta")?; + let dc_table = td_ta >> 4; + let ac_table = td_ta & 0x0F; + if dc_table > 3 || ac_table > 3 { + return Err(JpegError::Malformed(format!( + "B.2.3: table selectors Td={dc_table} Ta={ac_table} > 3" + ))); + } + components.push(ScanComponent { + id, + dc_table, + ac_table, + }); + } + let spectral_start = cur.u8("SOS Ss")?; + let spectral_end = cur.u8("SOS Se")?; + let approx = cur.u8("SOS Ah/Al")?; + let approx_high = approx >> 4; + let approx_low = approx & 0x0F; + if spectral_start > 63 || spectral_end > 63 { + return Err(JpegError::Malformed(format!( + "B.2.3: spectral selection Ss={spectral_start} Se={spectral_end} out of 0..=63" + ))); + } + Ok(ScanHeader { + components, + spectral_start, + spectral_end, + approx_high, + approx_low, + }) +} diff --git a/JPXL/crates/jpxl-jpeg/src/progressive.rs b/JPXL/crates/jpxl-jpeg/src/progressive.rs new file mode 100644 index 00000000..f9a64bc1 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/progressive.rs @@ -0,0 +1,544 @@ +//! Progressive DCT scan coding (10918-1 G.1.2): DC and AC, first and +//! refinement scans, with spectral selection and successive approximation. +//! +//! Each coefficient is coded across several scans. The decoder accumulates the +//! full coefficient into the plane; the encoder re-derives every scan's symbols +//! from that final coefficient with the matching *point transform*: an +//! arithmetic (floor) shift for DC, a toward-zero shift for AC. Because the +//! plane holds the fully-refined value, `serialize(parse(x)) == x` reduces to +//! each scan's coder being an exact inverse of the other — no intermediate +//! per-scan state is stored. +//! +//! The EOB-run batching and the interleaving of correction bits with +//! zero-run-length (`ZRL`) codes follow the sequence the reference procedures +//! produce, so a stream from a standard encoder is reproduced bit-for-bit. + +// Coefficient indices come from `nat(k)` (masked to `0..=63`); `run`/`rem` are +// bounded to `0..=15` before the `as u8` run-length narrowings, and EOB-run and +// correction-bit buffers are indexed by cursors the loops keep in range. Bulk +// plane/block access goes through checked `.get()`. +#![allow(clippy::indexing_slicing, clippy::cast_possible_truncation)] + +use crate::bitio::{EntropyReader, EntropyWriter, Padding}; +use crate::codec::{ScanGeom, block_mut, finish_restart_segment, miss}; +use crate::coeff::{extend, magnitude_category, mantissa_bits, to_coeff}; +use crate::error::{JpegError, Result}; +use crate::frame::{FrameGeometry, FrameHeader}; +use crate::huffman::HuffmanTable; +use crate::scan::ScanHeader; +use crate::segment::ComponentPlane; +use crate::units::ZIGZAG_TO_NATURAL; + +#[inline] +fn nat(k: usize) -> usize { + ZIGZAG_TO_NATURAL[k & 63] as usize +} + +/// DC point transform (encoder side): arithmetic floor shift by `al`. +#[inline] +fn dc_point(v: i32, al: u8) -> i32 { + v >> al +} + +/// AC point transform (encoder side): division by `2^al` rounded toward zero. +#[inline] +fn ac_point(v: i32, al: u8) -> i32 { + if v >= 0 { v >> al } else { -((-v) >> al) } +} + +/// Decodes one progressive scan into `planes`, returning per-segment padding +/// and the observed EOB-run lengths (empty for DC scans). +#[allow(clippy::too_many_arguments)] +pub fn decode_progressive_scan( + frame: &FrameHeader, + geom: &FrameGeometry, + planes: &mut [ComponentPlane], + header: &ScanHeader, + dc_tables: &[Option; 4], + ac_tables: &[Option; 4], + restart_interval: u16, + reader: &mut EntropyReader<'_>, +) -> Result<(Vec, Vec)> { + let sg = ScanGeom::resolve(frame, geom, header, dc_tables, ac_tables)?; + let ss = header.spectral_start as usize; + let se = header.spectral_end as usize; + let ah = header.approx_high; + let al = header.approx_low; + let ri = restart_interval as usize; + let mut padding = Vec::new(); + let mut eob_runs: Vec = Vec::new(); + let mut refs = Vec::new(); + + if ss == 0 { + // DC scan (may be interleaved). + let mut dc_pred = vec![0i32; frame.components.len()]; + for u in 0..sg.num_units { + if ri != 0 && u != 0 && u % ri == 0 { + padding.push(finish_restart_segment(reader)?); + for p in dc_pred.iter_mut() { + *p = 0; + } + } + sg.unit_blocks(u, &mut refs); + for r in &refs { + let comp = sg.comp_for(r.frame_idx)?; + let dc_tbl = comp.dc.ok_or_else(|| miss("DC"))?; + let block = block_mut(planes, r.frame_idx, r.bx, r.by)?; + if ah == 0 { + let t = u32::from(dc_tbl.decode(reader)?); + if t > 15 { + return Err(JpegError::Malformed(format!( + "G.1.2.1: DC magnitude category {t} > 15" + ))); + } + let diff = extend(reader.read_bits(t)?, t); + let pred = &mut dc_pred[r.frame_idx]; + *pred = pred.wrapping_add(diff); + block[0] = to_coeff(*pred << al)?; + } else if reader.read_bit()? == 1 { + let v = i32::from(block[0]) | (1i32 << al); + block[0] = to_coeff(v)?; + } + } + } + } else { + // AC scan (single component, non-interleaved). + let mut eobrun: u32 = 0; + for u in 0..sg.num_units { + if ri != 0 && u != 0 && u % ri == 0 { + padding.push(finish_restart_segment(reader)?); + eobrun = 0; + } + sg.unit_blocks(u, &mut refs); + let r = &refs[0]; + let comp = sg.comp_for(r.frame_idx)?; + let ac_tbl = comp.ac.ok_or_else(|| miss("AC"))?; + let block = block_mut(planes, r.frame_idx, r.bx, r.by)?; + if ah == 0 { + decode_ac_first( + reader, + ac_tbl, + block, + ss, + se, + al, + &mut eobrun, + &mut eob_runs, + )?; + } else { + decode_ac_refine( + reader, + ac_tbl, + block, + ss, + se, + al, + &mut eobrun, + &mut eob_runs, + )?; + } + } + } + padding.push(reader.take_padding()?); + Ok((padding, eob_runs)) +} + +#[allow(clippy::too_many_arguments)] +fn decode_ac_first( + reader: &mut EntropyReader<'_>, + ac: &HuffmanTable, + block: &mut [i16; 64], + ss: usize, + se: usize, + al: u8, + eobrun: &mut u32, + eob_runs: &mut Vec, +) -> Result<()> { + if *eobrun > 0 { + *eobrun -= 1; + return Ok(()); + } + let mut k = ss; + while k <= se { + let rs = ac.decode(reader)?; + let r = (rs >> 4) as usize; + let s = u32::from(rs & 0x0F); + if s == 0 { + if r != 15 { + let mut run = 1u32 << r; + if r > 0 { + run += reader.read_bits(r as u32)?; + } + eob_runs.push(run); + *eobrun = run - 1; + break; + } + k += 16; // ZRL: 16 zero coefficients + } else { + k += r; + if k > se { + return Err(JpegError::Malformed( + "G.1.2.2: AC-first run overruns the band".into(), + )); + } + let val = extend(reader.read_bits(s)?, s); + block[nat(k)] = to_coeff(val << al)?; + k += 1; + } + } + Ok(()) +} + +#[allow(clippy::too_many_arguments)] +fn decode_ac_refine( + reader: &mut EntropyReader<'_>, + ac: &HuffmanTable, + block: &mut [i16; 64], + ss: usize, + se: usize, + al: u8, + eobrun: &mut u32, + eob_runs: &mut Vec, +) -> Result<()> { + let p1: i32 = 1 << al; + let m1: i32 = -(1 << al); + let mut k = ss; + + if *eobrun == 0 { + while k <= se { + let rs = ac.decode(reader)?; + let mut r = (rs >> 4) as i32; + let s = u32::from(rs & 0x0F); + let mut newval: i32 = 0; + if s == 0 { + if r != 15 { + let mut run = 1u32 << r; + if r > 0 { + run += reader.read_bits(r as u32)?; + } + eob_runs.push(run); + *eobrun = run; + break; + } + // r == 15: ZRL, skip 16 zero-history coefficients. + } else { + if s != 1 { + return Err(JpegError::Malformed( + "G.1.2.3: refinement coefficient size != 1".into(), + )); + } + let bit = reader.read_bit()?; + newval = if bit == 1 { p1 } else { m1 }; + } + // Advance over zero-history coefficients, correcting nonzero ones. + loop { + let coef = i32::from(block[nat(k)]); + if coef != 0 { + let cb = reader.read_bit()?; + if cb == 1 && (coef & p1) == 0 { + block[nat(k)] = to_coeff(coef + if coef >= 0 { p1 } else { m1 })?; + } + } else { + r -= 1; + if r < 0 { + break; + } + } + k += 1; + if k > se { + break; + } + } + if s != 0 { + if k > se { + return Err(JpegError::Malformed( + "G.1.2.3: refinement coefficient position past band end".into(), + )); + } + block[nat(k)] = to_coeff(newval)?; + } + k += 1; + } + } + + if *eobrun > 0 { + while k <= se { + let coef = i32::from(block[nat(k)]); + if coef != 0 { + let cb = reader.read_bit()?; + if cb == 1 && (coef & p1) == 0 { + block[nat(k)] = to_coeff(coef + if coef >= 0 { p1 } else { m1 })?; + } + } + k += 1; + } + *eobrun -= 1; + } + Ok(()) +} + +/// Encodes one progressive scan from `planes` into `out`. +#[allow(clippy::too_many_arguments)] +pub fn encode_progressive_scan( + out: &mut Vec, + frame: &FrameHeader, + geom: &FrameGeometry, + planes: &[ComponentPlane], + header: &ScanHeader, + dc_tables: &[Option; 4], + ac_tables: &[Option; 4], + restart_interval: u16, + padding: &[Padding], + eob_runs: &[u32], +) -> Result<()> { + let sg = ScanGeom::resolve(frame, geom, header, dc_tables, ac_tables)?; + let ss = header.spectral_start as usize; + let se = header.spectral_end as usize; + let ah = header.approx_high; + let al = header.approx_low; + let ri = restart_interval as usize; + let mut pad_iter = padding.iter(); + let mut refs = Vec::new(); + let mut restart_counter = 0u8; + let mut writer = EntropyWriter::new(out); + + /// Closes the current entropy segment at a restart boundary: emits padding + /// and the cycling `RSTn` marker. + fn emit_restart( + writer: &mut EntropyWriter<'_>, + pad_iter: &mut std::slice::Iter<'_, Padding>, + restart_counter: &mut u8, + ) -> Result<()> { + let pad = *pad_iter + .next() + .ok_or_else(|| JpegError::Encode("missing padding for restart segment".into()))?; + writer.flush_padding(pad)?; + writer.write_aligned_bytes(&[0xFF, crate::marker::RST0 + (*restart_counter & 7)])?; + *restart_counter = restart_counter.wrapping_add(1); + Ok(()) + } + + if ss == 0 { + let mut dc_pred = vec![0i32; frame.components.len()]; + for u in 0..sg.num_units { + if ri != 0 && u != 0 && u % ri == 0 { + emit_restart(&mut writer, &mut pad_iter, &mut restart_counter)?; + for p in dc_pred.iter_mut() { + *p = 0; + } + } + sg.unit_blocks(u, &mut refs); + for r in &refs { + let comp = sg.comp_for(r.frame_idx)?; + let plane = planes + .get(r.frame_idx) + .ok_or_else(|| JpegError::Encode("plane index out of range".into()))?; + let block = plane + .block(r.bx, r.by) + .ok_or_else(|| JpegError::Encode("block position out of range".into()))?; + if ah == 0 { + let dc_tbl = comp.dc.ok_or_else(|| miss("DC"))?; + let v = dc_point(i32::from(block[0]), al); + let diff = v - dc_pred[r.frame_idx]; + dc_pred[r.frame_idx] = v; + let s = magnitude_category(diff); + if s > 15 { + return Err(JpegError::Encode(format!( + "DC diff {diff} needs category {s} > 15" + ))); + } + dc_tbl.encode(&mut writer, s as u8)?; + writer.put_bits(mantissa_bits(diff, s), s); + } else { + let bit = (i32::from(block[0]) >> al) & 1; + writer.put_bit(bit as u32); + } + } + } + } else { + let ac_tbl = sg + .comps + .first() + .and_then(|c| c.ac) + .ok_or_else(|| miss("AC"))?; + // EOB runs are replayed at exactly the lengths the original encoder + // used (`eob_runs`, recorded at decode), rather than re-derived: an + // encoder may split a run into several `EOBn` codes, and that split is + // not recoverable from the coefficients alone. + let mut eobrun: u32 = 0; + let mut be: Vec = Vec::new(); + let mut ei = 0usize; + for u in 0..sg.num_units { + if ri != 0 && u != 0 && u % ri == 0 { + // A restart resets the decoder's EOB run, so any pending run was + // flushed before the marker; `eobrun` is already 0 here after the + // match-flush below, but guard defensively. + if eobrun > 0 { + emit_eobrun(&mut writer, ac_tbl, &mut eobrun, &mut be)?; + ei += 1; + } + emit_restart(&mut writer, &mut pad_iter, &mut restart_counter)?; + } + sg.unit_blocks(u, &mut refs); + let r = &refs[0]; + let plane = planes + .get(r.frame_idx) + .ok_or_else(|| JpegError::Encode("plane index out of range".into()))?; + let block = plane + .block(r.bx, r.by) + .ok_or_else(|| JpegError::Encode("block position out of range".into()))?; + let contributes = if ah == 0 { + encode_ac_first(&mut writer, ac_tbl, block, ss, se, al)? + } else { + encode_ac_refine(&mut writer, ac_tbl, block, ss, se, al, &mut be)? + }; + if contributes { + eobrun += 1; + // Flush at exactly the recorded run length (handles both natural + // ends at the next significant block and mid-run encoder splits). + if ei < eob_runs.len() && eobrun == eob_runs[ei] { + emit_eobrun(&mut writer, ac_tbl, &mut eobrun, &mut be)?; + ei += 1; + } + } + } + // Flush any final run (its length is the last recorded value). + if eobrun > 0 { + emit_eobrun(&mut writer, ac_tbl, &mut eobrun, &mut be)?; + } + } + + let pad = *pad_iter + .next() + .ok_or_else(|| JpegError::Encode("missing padding for final segment".into()))?; + writer.flush_padding(pad)?; + Ok(()) +} + +/// Emits a pending EOB run code (if any) and any buffered correction bits. +fn emit_eobrun( + writer: &mut EntropyWriter<'_>, + ac: &HuffmanTable, + eobrun: &mut u32, + be: &mut Vec, +) -> Result<()> { + if *eobrun > 0 { + let e = *eobrun; + let n = 31 - e.leading_zeros(); // floor(log2(e)) + ac.encode(writer, (n as u8) << 4)?; + if n > 0 { + writer.put_bits(e - (1 << n), n); + } + *eobrun = 0; + } + for &b in be.iter() { + writer.put_bit(u32::from(b)); + } + be.clear(); + Ok(()) +} + +/// Encodes one AC-first block. Returns whether the band ends in a run of zeros +/// (so the caller extends the pending EOB run). EOB-run flushing is the +/// caller's job (it replays the recorded run lengths). +fn encode_ac_first( + writer: &mut EntropyWriter<'_>, + ac: &HuffmanTable, + block: &[i16; 64], + ss: usize, + se: usize, + al: u8, +) -> Result { + let mut run = 0usize; + for k in ss..=se { + let coef = ac_point(i32::from(block[nat(k)]), al); + if coef == 0 { + run += 1; + continue; + } + while run >= 16 { + ac.encode(writer, 0xF0)?; // ZRL + run -= 16; + } + let s = magnitude_category(coef); + if s == 0 || s > 15 { + return Err(JpegError::Encode(format!( + "AC-first coefficient {coef} has category {s}" + ))); + } + ac.encode(writer, ((run as u8) << 4) | (s as u8))?; + writer.put_bits(mantissa_bits(coef, s), s); + run = 0; + } + // Trailing zeros (or a wholly-zero band) begin/extend an EOB run. + Ok(run > 0) +} + +/// Encodes one AC-refinement block. Returns whether the band ends in a run +/// (trailing zeros or trailing already-significant corrections), so the caller +/// extends the pending EOB run; trailing correction bits are appended to `be` +/// for the caller to emit when it flushes that run. +/// +/// `run` counts consecutive zero-history (still-insignificant) coefficients. +/// Already-significant coefficients are transparent to the run; their correction +/// bits are buffered with the zero-count that preceded them, so that when a +/// newly-significant coefficient forces `ZRL` codes, the corrections replay in +/// the same 16-zero spans the decoder walks (10918-1 G.1.2.3). Only +/// newly-significant coefficients emit `ZRL`s; a block with none is folded into +/// the EOB run. +fn encode_ac_refine( + writer: &mut EntropyWriter<'_>, + ac: &HuffmanTable, + block: &[i16; 64], + ss: usize, + se: usize, + al: u8, + be: &mut Vec, +) -> Result { + let mut run = 0usize; + let mut br: Vec<(usize, u8)> = Vec::new(); // (zeros seen before this correction, bit) + for k in ss..=se { + let coef = i32::from(block[nat(k)]); + let mag = coef.abs() >> al; // magnitude at this scan + if mag == 0 { + run += 1; + continue; + } + if mag > 1 { + // Already significant: buffer its correction bit; the run continues. + br.push((run, (mag & 1) as u8)); + continue; + } + // Newly significant (mag == 1). + let num_zrl = run / 16; + let rem = run % 16; + let mut bi = 0usize; + for j in 0..num_zrl { + ac.encode(writer, 0xF0)?; // ZRL + while bi < br.len() && br[bi].0 / 16 == j { + writer.put_bit(u32::from(br[bi].1)); + bi += 1; + } + } + ac.encode(writer, ((rem as u8) << 4) | 1)?; + let sign = if coef < 0 { 0 } else { 1 }; + writer.put_bit(sign); + // Corrections in the final (remainder) span follow the coefficient code. + while bi < br.len() { + writer.put_bit(u32::from(br[bi].1)); + bi += 1; + } + br.clear(); + run = 0; + } + if run > 0 || !br.is_empty() { + // Trailing zeros / already-significant corrections extend an EOB run; + // their correction bits are emitted when that run is flushed. + for &(_, bit) in &br { + be.push(bit); + } + Ok(true) + } else { + Ok(false) + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/quant.rs b/JPXL/crates/jpxl-jpeg/src/quant.rs new file mode 100644 index 00000000..dba94d9a --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/quant.rs @@ -0,0 +1,20 @@ +//! Quantization tables (10918-1 B.2.4.1). +//! +//! A `DQT` segment stores, per table, a precision/id byte then 64 element +//! bytes in zig-zag order (`Pq = 0`: 8-bit elements; `Pq = 1`: 16-bit +//! big-endian elements). The elements are kept in the exact zig-zag order they +//! appear on the wire so re-emission is byte-identical; Phase B, which needs +//! them in natural order, can permute with [`ZigZag`]. +//! +//! [`ZigZag`]: crate::units::ZigZag + +/// One quantization table as stored in a `DQT` segment. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct QuantTable { + /// Element precision `Pq`: 0 = 8-bit, 1 = 16-bit. + pub precision: u8, + /// Destination identifier `Tq`, `0..=3`. + pub id: u8, + /// The 64 quantization step sizes in zig-zag order. + pub values: [u16; 64], +} diff --git a/JPXL/crates/jpxl-jpeg/src/scan.rs b/JPXL/crates/jpxl-jpeg/src/scan.rs new file mode 100644 index 00000000..23623ae3 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/scan.rs @@ -0,0 +1,40 @@ +//! Scan header (`SOS`, 10918-1 B.2.3). + +use crate::units::ComponentId; + +/// One component's table selectors within a scan. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ScanComponent { + /// Component selector `Csj` — matches a [`FrameComponent`] identifier. + /// + /// [`FrameComponent`]: crate::frame::FrameComponent + pub id: ComponentId, + /// DC entropy-table destination selector `Tdj`, `0..=3`. + pub dc_table: u8, + /// AC entropy-table destination selector `Taj`, `0..=3`. + pub ac_table: u8, +} + +/// A parsed start-of-scan header. +#[derive(Clone, Debug)] +pub struct ScanHeader { + /// Components in this scan, in scan order (interleave order). + pub components: Vec, + /// Start of spectral selection `Ss`. + pub spectral_start: u8, + /// End of spectral selection `Se`. + pub spectral_end: u8, + /// Successive approximation high bit position `Ah`. + pub approx_high: u8, + /// Successive approximation low bit position `Al`. + pub approx_low: u8, +} + +impl ScanHeader { + /// Whether this scan interleaves more than one component (MCU geometry) or + /// is a single-component scan (non-interleaved block geometry). + #[must_use] + pub fn is_interleaved(&self) -> bool { + self.components.len() > 1 + } +} diff --git a/JPXL/crates/jpxl-jpeg/src/segment.rs b/JPXL/crates/jpxl-jpeg/src/segment.rs new file mode 100644 index 00000000..7a3c0806 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/segment.rs @@ -0,0 +1,123 @@ +//! The parsed JPEG document model. +//! +//! A JPEG is `SOI`, an ordered list of marker segments (some carrying entropy +//! data), `EOI`, then optional trailing bytes. Byte-exact re-emission depends +//! on preserving that order exactly, including how multiple tables were grouped +//! into a single `DQT`/`DHT` segment, so segments are kept as an ordered list +//! rather than folded into a normalized structure. + +use crate::bitio::Padding; +use crate::frame::FrameHeader; +use crate::huffman::HuffmanTable; +use crate::quant::QuantTable; +use crate::scan::ScanHeader; +use crate::units::ComponentId; + +/// An application segment `APPn` (`0xE0..=0xEF`). +#[derive(Clone, Debug)] +pub struct AppSegment { + /// The marker code byte (`0xE0` = APP0 … `0xEF` = APP15). + pub code: u8, + /// The raw payload (everything after the 2-byte length field). + pub payload: Vec, +} + +/// A start-of-scan segment: the header plus the artifacts needed to reproduce +/// its entropy-coded data bit-for-bit. +#[derive(Clone, Debug)] +pub struct ScanSegment { + /// The `SOS` header. + pub header: ScanHeader, + /// The end-of-segment padding bits for each entropy segment, in order: + /// one entry per restart interval plus one for the final segment + /// (10918-1 F.1.2.3). + pub padding: Vec, + /// For a progressive AC scan, the end-of-band run lengths exactly as the + /// original encoder emitted them, in decode order (10918-1 G.1.2.2). + /// + /// An encoder is free to split a long EOB run into several `EOBn` codes + /// (libjpeg flushes when its correction-bit buffer fills), and that split + /// is *not* recoverable from the coefficients alone. Recording the observed + /// run lengths and replaying them is what keeps the re-encode bit-exact. + /// Empty for DC and baseline scans. + pub eob_runs: Vec, +} + +/// A marker segment the codec does not model in detail but must preserve +/// verbatim (e.g. `DNL`). +#[derive(Clone, Debug)] +pub struct OtherSegment { + /// The marker code byte. + pub code: u8, + /// The raw payload (everything after the 2-byte length field). + pub payload: Vec, +} + +/// One element of the ordered segment list between `SOI` and `EOI`. +#[derive(Clone, Debug)] +pub enum Segment { + /// `APPn` application data. + App(AppSegment), + /// `COM` comment payload. + Com(Vec), + /// `DQT` — one or more quantization tables, in segment order. + Dqt(Vec), + /// `DHT` — one or more Huffman tables, in segment order. + Dht(Vec), + /// `DRI` — restart interval in MCUs. + Dri(u16), + /// `SOF` — the frame header. + Sof(FrameHeader), + /// `SOS` — a scan header and its entropy segment padding. + Sos(ScanSegment), + /// Any other length-bearing marker segment, preserved verbatim. + Other(OtherSegment), +} + +/// Decoded quantized coefficients for one frame component. +/// +/// Blocks are stored row-major at the *interleaved* (MCU-padded) dimensions, +/// so a single plane serves both interleaved and non-interleaved scans of the +/// same component. Each block holds 64 coefficients in **natural** (raster) +/// order; element 0 is DC. +#[derive(Clone, Debug)] +pub struct ComponentPlane { + /// Component identifier `Ci`. + pub id: ComponentId, + /// Horizontal sampling factor `Hi`. + pub h: u8, + /// Vertical sampling factor `Vi`. + pub v: u8, + /// Blocks per line (interleaved / MCU-padded). + pub blocks_per_line: usize, + /// Block rows (interleaved / MCU-padded). + pub block_rows: usize, + /// The blocks, row-major: block `(bx, by)` is at `by * blocks_per_line + bx`. + pub blocks: Vec<[i16; 64]>, +} + +impl ComponentPlane { + /// Immutable access to block `(bx, by)`. + #[must_use] + pub fn block(&self, bx: usize, by: usize) -> Option<&[i16; 64]> { + if bx >= self.blocks_per_line || by >= self.block_rows { + return None; + } + self.blocks.get(by * self.blocks_per_line + bx) + } +} + +/// A fully parsed JPEG: its segment structure, its decoded coefficients, and +/// its trailing bytes. +#[derive(Clone, Debug)] +pub struct Jpeg { + /// The ordered segments between `SOI` and `EOI`. + pub segments: Vec, + /// The frame header (`None` only for a malformed frame-less stream, which + /// parsing rejects before returning). + pub frame: Option, + /// Decoded coefficient planes, indexed like `frame.components`. + pub planes: Vec, + /// Bytes after `EOI` (the "tail data" Annex A preserves). + pub tail: Vec, +} diff --git a/JPXL/crates/jpxl-jpeg/src/serialize.rs b/JPXL/crates/jpxl-jpeg/src/serialize.rs new file mode 100644 index 00000000..9669a246 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/serialize.rs @@ -0,0 +1,179 @@ +//! Re-emit a [`Jpeg`] document model back to a byte stream. +//! +//! This is the exact inverse of [`crate::parse`]: it walks the ordered segment +//! list, re-emits every marker segment from its typed form, and re-encodes each +//! scan's coefficients through [`crate::codec::encode_scan`], reproducing the +//! captured padding and restart cadence. For a stream this codec parsed, +//! `serialize(parse(x)) == x` byte-for-byte. + +use crate::codec::encode_scan; +use crate::error::{JpegError, Result}; +use crate::huffman::HuffmanTable; +use crate::marker; +use crate::quant::QuantTable; +use crate::segment::{Jpeg, Segment}; + +/// Serializes `jpeg` to bytes. +pub fn serialize(jpeg: &Jpeg) -> Result> { + let frame = jpeg + .frame + .as_ref() + .ok_or_else(|| JpegError::Encode("cannot serialize a frame-less JPEG".into()))?; + let geom = frame.geometry()?; + + let mut out = Vec::new(); + write_marker(&mut out, marker::SOI); + + let mut dc_tables: [Option; 4] = Default::default(); + let mut ac_tables: [Option; 4] = Default::default(); + let mut restart_interval: u16 = 0; + + for seg in &jpeg.segments { + match seg { + Segment::App(a) => { + write_marker(&mut out, a.code); + write_length_payload(&mut out, &a.payload)?; + } + Segment::Com(payload) => { + write_marker(&mut out, marker::COM); + write_length_payload(&mut out, payload)?; + } + Segment::Dqt(tables) => { + write_marker(&mut out, marker::DQT); + write_dqt(&mut out, tables)?; + } + Segment::Dht(tables) => { + for t in tables { + let slot = if t.class == 0 { + &mut dc_tables + } else { + &mut ac_tables + }; + if let Some(dst) = slot.get_mut(t.id as usize) { + *dst = Some(t.clone()); + } + } + write_marker(&mut out, marker::DHT); + write_dht(&mut out, tables)?; + } + Segment::Dri(ri) => { + restart_interval = *ri; + write_marker(&mut out, marker::DRI); + write_u16(&mut out, 4); + write_u16(&mut out, *ri); + } + Segment::Sof(fh) => { + write_marker(&mut out, fh.code); + write_sof(&mut out, fh)?; + } + Segment::Sos(scan) => { + write_marker(&mut out, marker::SOS); + write_sos(&mut out, &scan.header)?; + encode_scan( + &mut out, + frame, + &geom, + &jpeg.planes, + &scan.header, + &dc_tables, + &ac_tables, + restart_interval, + &scan.padding, + &scan.eob_runs, + )?; + } + Segment::Other(o) => { + write_marker(&mut out, o.code); + if o.code != marker::TEM { + write_length_payload(&mut out, &o.payload)?; + } + } + } + } + + write_marker(&mut out, marker::EOI); + out.extend_from_slice(&jpeg.tail); + Ok(out) +} + +fn write_marker(out: &mut Vec, code: u8) { + out.push(marker::MARKER_PREFIX); + out.push(code); +} + +fn write_u16(out: &mut Vec, value: u16) { + out.extend_from_slice(&value.to_be_bytes()); +} + +fn segment_len(payload_bytes: usize, ctx: &'static str) -> Result { + u16::try_from(payload_bytes + 2) + .map_err(|_| JpegError::Encode(format!("{ctx} segment exceeds 65535 bytes"))) +} + +fn write_length_payload(out: &mut Vec, payload: &[u8]) -> Result<()> { + write_u16(out, segment_len(payload.len(), "segment")?); + out.extend_from_slice(payload); + Ok(()) +} + +fn write_dqt(out: &mut Vec, tables: &[QuantTable]) -> Result<()> { + let mut body = Vec::new(); + for t in tables { + body.push((t.precision << 4) | t.id); + if t.precision == 0 { + for &v in &t.values { + body.push(v.to_be_bytes()[1]); + } + } else { + for &v in &t.values { + body.extend_from_slice(&v.to_be_bytes()); + } + } + } + write_u16(out, segment_len(body.len(), "DQT")?); + out.extend_from_slice(&body); + Ok(()) +} + +fn write_dht(out: &mut Vec, tables: &[HuffmanTable]) -> Result<()> { + let mut body = Vec::new(); + for t in tables { + body.push((t.class << 4) | t.id); + body.extend_from_slice(&t.counts); + body.extend_from_slice(&t.values); + } + write_u16(out, segment_len(body.len(), "DHT")?); + out.extend_from_slice(&body); + Ok(()) +} + +fn write_sof(out: &mut Vec, fh: &crate::frame::FrameHeader) -> Result<()> { + let nf = u8::try_from(fh.components.len()) + .map_err(|_| JpegError::Encode("SOF component count > 255".into()))?; + write_u16(out, 8 + 3 * u16::from(nf)); + out.push(fh.precision); + write_u16(out, fh.height); + write_u16(out, fh.width); + out.push(nf); + for c in &fh.components { + out.push(c.id.0); + out.push((c.h << 4) | c.v); + out.push(c.quant_id); + } + Ok(()) +} + +fn write_sos(out: &mut Vec, header: &crate::scan::ScanHeader) -> Result<()> { + let ns = u8::try_from(header.components.len()) + .map_err(|_| JpegError::Encode("SOS component count > 255".into()))?; + write_u16(out, 6 + 2 * u16::from(ns)); + out.push(ns); + for c in &header.components { + out.push(c.id.0); + out.push((c.dc_table << 4) | c.ac_table); + } + out.push(header.spectral_start); + out.push(header.spectral_end); + out.push((header.approx_high << 4) | header.approx_low); + Ok(()) +} diff --git a/JPXL/crates/jpxl-jpeg/src/units.rs b/JPXL/crates/jpxl-jpeg/src/units.rs new file mode 100644 index 00000000..dbad266f --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/src/units.rs @@ -0,0 +1,57 @@ +//! Unit-bearing newtypes at the JPEG-1 coefficient/coordinate boundaries. +//! +//! AGENTS.md §6 bans bare `u8`/`usize` where a mix-up is a real bug. The traps +//! this codec must not fall into are the classic ones: a zig-zag stream index +//! read as a natural (raster) block position, a component's *identifier* (the +//! `Ci` byte from SOF, an arbitrary label) confused with its *index* in the +//! frame's component list, and a DC prediction difference confused with a DC +//! value. Each gets its own type. + +/// The natural (raster, row-major) position of a coefficient inside an 8×8 +/// block, `0..=63`. Element 0 is the DC term. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Natural(pub u8); + +/// A position `0..=63` along the zig-zag scan order in which a block's +/// coefficients travel on the wire (10918-1 Figure A.6 / A.3.6). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ZigZag(pub u8); + +/// The zig-zag → natural permutation (10918-1 Annex A, Figure A.6). +/// +/// `ZIGZAG_TO_NATURAL[k]` is the natural index of the coefficient that is +/// `k`-th in zig-zag order. +pub const ZIGZAG_TO_NATURAL: [u8; 64] = [ + 0, 1, 8, 16, 9, 2, 3, 10, // + 17, 24, 32, 25, 18, 11, 4, 5, // + 12, 19, 26, 33, 40, 48, 41, 34, // + 27, 20, 13, 6, 7, 14, 21, 28, // + 35, 42, 49, 56, 57, 50, 43, 36, // + 29, 22, 15, 23, 30, 37, 44, 51, // + 58, 59, 52, 45, 38, 31, 39, 46, // + 53, 60, 61, 54, 47, 55, 62, 63, // +]; + +impl ZigZag { + /// Maps this zig-zag position to its natural (raster) position. + #[must_use] + // The index is masked to 0..=63 and `ZIGZAG_TO_NATURAL` has exactly 64 + // entries, so it is always in range. + #[allow(clippy::indexing_slicing)] + pub fn to_natural(self) -> Natural { + Natural(ZIGZAG_TO_NATURAL[self.0 as usize & 63]) + } +} + +/// A component's identifier — the arbitrary `Ci` label byte from the SOF +/// header (10918-1 B.2.2). Not an index into any array. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ComponentId(pub u8); + +/// A component's *index* in the frame's component list, `0..Nf`. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ComponentIndex(pub usize); + +/// A restart interval in minimum-coded-units (10918-1 B.2.4.4). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RestartInterval(pub u16); diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_420.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_420.jpg new file mode 100644 index 00000000..9e0ff249 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_420.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_420.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_420.jpg.prov new file mode 100644 index 00000000..4b2ffeda --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_420.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_420.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: ea260f57d77586940f9938d427ba16c73d2ef2af7b4378b7f988cddf1fe45b4e +regenerate: cjpeg -quality 85 -sample 2x2 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_422.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_422.jpg new file mode 100644 index 00000000..556aca3d Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_422.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_422.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_422.jpg.prov new file mode 100644 index 00000000..b92504ff --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_422.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_422.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: ed4231510da7db2dbcbd4f28ffdffc36952134ea0530ca6fdfb9adf87d259da5 +regenerate: cjpeg -quality 85 -sample 2x1 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_440.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_440.jpg new file mode 100644 index 00000000..ae8c9e15 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_440.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_440.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_440.jpg.prov new file mode 100644 index 00000000..77a5a627 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_440.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_440.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 4c081c88785200558d5e3440d67bfc667434795ee01f6c46e86e974795fe02d0 +regenerate: cjpeg -quality 85 -sample 1x2 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_444.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_444.jpg new file mode 100644 index 00000000..5d65a99d Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_444.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_444.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_444.jpg.prov new file mode 100644 index 00000000..d6d9c822 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_444.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_444.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 478b2e1da74c5af83796cdcd5f86f6a0117d392692212e02124015168727a789 +regenerate: cjpeg -quality 85 -sample 1x1 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_gray.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_gray.jpg new file mode 100644 index 00000000..abbc1ad9 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_gray.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_gray.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_gray.jpg.prov new file mode 100644 index 00000000..c9186591 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_gray.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_gray.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 9ef80258bd425cd773d5e62863b3cab6097491f00a9f60316f0ffff9e25d37c7 +regenerate: cjpeg -quality 85 -grayscale diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q20.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q20.jpg new file mode 100644 index 00000000..640eaf42 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q20.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q20.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q20.jpg.prov new file mode 100644 index 00000000..4c5f5079 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q20.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_q20.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: fb65f1e9094878e38581a6c828040c0dcec7ea9d32e954c72d235604f7514d7d +regenerate: cjpeg -quality 20 -sample 2x2 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q98.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q98.jpg new file mode 100644 index 00000000..ab9c6adc Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q98.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q98.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q98.jpg.prov new file mode 100644 index 00000000..2d675f77 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_q98.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_q98.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 0f4769cf46a36d8e11bde1f4e4ce556784298ea3ec8e7fa4db19c7fc1cf9bb82 +regenerate: cjpeg -quality 98 -sample 1x1 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart.jpg new file mode 100644 index 00000000..fa39771e Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart.jpg.prov new file mode 100644 index 00000000..48469ad9 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_restart.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: ddfe7e6dd538ded71789ad4b4b7b94ce27bb5753aba1914ee27b8443e5ef3374 +regenerate: cjpeg -quality 90 -sample 2x2 -restart 5 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart_rows.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart_rows.jpg new file mode 100644 index 00000000..eab913fa Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart_rows.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart_rows.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart_rows.jpg.prov new file mode 100644 index 00000000..1bd5b273 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/base_restart_rows.jpg.prov @@ -0,0 +1,5 @@ +fixture: base_restart_rows.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: b07dbf9d47363145de93221a43ebbc36d28423eae9c065e8cfa98693b6b144d7 +regenerate: cjpeg -quality 90 -sample 1x1 -restart 2B diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/generate.py b/JPXL/crates/jpxl-jpeg/tests/fixtures/generate.py new file mode 100644 index 00000000..ea44ad12 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/generate.py @@ -0,0 +1,235 @@ +#!/usr/bin/env python3 +"""Deterministically generate the jpxl-jpeg round-trip fixtures. + +All source pixels are synthesised here from a fixed seed (no third-party image +is used), so the fixtures are original content, CC0 / public domain, and +regenerable byte-for-byte. Encoders: libjpeg-turbo `cjpeg`/`jpegtran` and +Pillow (libjpeg-turbo) for the Exif/ICC cases. + +Run from this directory: python3 generate.py +It writes each `*.jpg` plus a `*.jpg.prov` provenance sidecar. +""" + +import hashlib +import math +import os +import struct +import subprocess +import sys + +HERE = os.path.dirname(os.path.abspath(__file__)) + + +def synth_rgb(width, height, seed): + """A rich synthetic RGB image: gradients, sinusoids, and block noise, so the + DCT coefficients span many run/size categories rather than being trivially + sparse.""" + # xorshift32 for a dependency-free deterministic PRNG. + state = seed & 0xFFFFFFFF or 1 + + def rnd(): + nonlocal state + state ^= (state << 13) & 0xFFFFFFFF + state ^= state >> 17 + state ^= (state << 5) & 0xFFFFFFFF + return state & 0xFFFFFFFF + + data = bytearray(width * height * 3) + for y in range(height): + for x in range(width): + i = (y * width + x) * 3 + r = (x * 255) // (width - 1) + g = (y * 255) // (height - 1) + b = int(127 + 100 * math.sin(x / 11.0) * math.cos(y / 13.0)) + # Block-structured noise every 16 px to exercise sharp edges. + if ((x // 16) + (y // 16)) % 3 == 0: + n = rnd() % 64 + r = (r + n) & 0xFF + g = (g ^ (n << 1)) & 0xFF + b = (b + (rnd() % 48)) & 0xFF + data[i] = r & 0xFF + data[i + 1] = g & 0xFF + data[i + 2] = max(0, min(255, b)) & 0xFF + return bytes(data) + + +def write_ppm(path, width, height, rgb): + with open(path, "wb") as f: + f.write(f"P6\n{width} {height}\n255\n".encode()) + f.write(rgb) + + +def sha256(path): + with open(path, "rb") as f: + return hashlib.sha256(f.read()).hexdigest() + + +def provenance(jpg, how): + with open(jpg + ".prov", "w") as f: + f.write( + "fixture: {name}\n" + "origin: synthesised by tests/fixtures/generate.py (original content)\n" + "license: CC0-1.0 / public domain (no third-party image used)\n" + "sha256: {digest}\n" + "regenerate: {how}\n".format( + name=os.path.basename(jpg), digest=sha256(jpg), how=how + ) + ) + + +def cjpeg(ppm, out, args, how): + subprocess.run(["cjpeg", "-outfile", out] + args + [ppm], check=True) + provenance(out, how) + + +def main(): + # Odd dimensions (not multiples of 8/16) to exercise edge / padding blocks + # and the interleaved-vs-non-interleaved block-count subtlety. + W, H = 385, 259 + color = synth_rgb(W, H, 0xC0FFEE) + ppm = os.path.join(HERE, "_src_color.ppm") + write_ppm(ppm, W, H, color) + + j = lambda n: os.path.join(HERE, n) + + # Baseline sequential, all chroma subsamplings. + cjpeg(ppm, j("base_444.jpg"), ["-quality", "85", "-sample", "1x1"], + "cjpeg -quality 85 -sample 1x1") + cjpeg(ppm, j("base_422.jpg"), ["-quality", "85", "-sample", "2x1"], + "cjpeg -quality 85 -sample 2x1") + cjpeg(ppm, j("base_420.jpg"), ["-quality", "85", "-sample", "2x2"], + "cjpeg -quality 85 -sample 2x2") + cjpeg(ppm, j("base_440.jpg"), ["-quality", "85", "-sample", "1x2"], + "cjpeg -quality 85 -sample 1x2") + cjpeg(ppm, j("base_gray.jpg"), ["-quality", "85", "-grayscale"], + "cjpeg -quality 85 -grayscale") + # Restart intervals (MCU count, and MCU-row form). + cjpeg(ppm, j("base_restart.jpg"), + ["-quality", "90", "-sample", "2x2", "-restart", "5"], + "cjpeg -quality 90 -sample 2x2 -restart 5") + cjpeg(ppm, j("base_restart_rows.jpg"), + ["-quality", "90", "-sample", "1x1", "-restart", "2B"], + "cjpeg -quality 90 -sample 1x1 -restart 2B") + # High quality (dense coefficients) and low quality (sparse, long EOBs). + cjpeg(ppm, j("base_q98.jpg"), ["-quality", "98", "-sample", "1x1"], + "cjpeg -quality 98 -sample 1x1") + cjpeg(ppm, j("base_q20.jpg"), ["-quality", "20", "-sample", "2x2"], + "cjpeg -quality 20 -sample 2x2") + + # Progressive. + cjpeg(ppm, j("prog_444.jpg"), ["-quality", "85", "-progressive", "-sample", "1x1"], + "cjpeg -quality 85 -progressive -sample 1x1") + cjpeg(ppm, j("prog_420.jpg"), ["-quality", "85", "-progressive", "-sample", "2x2"], + "cjpeg -quality 85 -progressive -sample 2x2") + cjpeg(ppm, j("prog_gray.jpg"), ["-quality", "85", "-progressive", "-grayscale"], + "cjpeg -quality 85 -progressive -grayscale") + cjpeg(ppm, j("prog_restart.jpg"), + ["-quality", "90", "-progressive", "-sample", "2x2", "-restart", "4"], + "cjpeg -quality 90 -progressive -sample 2x2 -restart 4") + cjpeg(ppm, j("prog_q30.jpg"), ["-quality", "30", "-progressive", "-sample", "2x2"], + "cjpeg -quality 30 -progressive -sample 2x2") + + # Large, detailed progressive image: its AC refinement scans have long EOB + # runs that libjpeg splits across several EOBn codes (correction-bit buffer + # fills), which the re-encoder must replay from recorded run lengths rather + # than re-derive. Regression for the archive-sweep EOB-run failures. + bw, bh = 1024, 768 + bstate = 0x1234567 + + def brnd(): + nonlocal bstate + bstate ^= (bstate << 13) & 0xFFFFFFFF + bstate ^= bstate >> 17 + bstate ^= (bstate << 5) & 0xFFFFFFFF + return bstate & 0xFFFFFFFF + + big = bytearray(bw * bh * 3) + for y in range(bh): + for x in range(bw): + i = (y * bw + x) * 3 + r = int(127 + 90 * math.sin(x / 7.0) * math.cos(y / 9.0)) + g = (x ^ y) & 0xFF + b = int(127 + 80 * math.sin((x + y) / 5.0)) + if ((x // 8) + (y // 8)) % 2 == 0: + n = brnd() % 96 + r = (r + n) & 0xFF + g = (g ^ n) & 0xFF + b = (b + (brnd() % 64)) & 0xFF + big[i] = r & 0xFF + big[i + 1] = max(0, min(255, g)) & 0xFF + big[i + 2] = max(0, min(255, b)) & 0xFF + bigppm = os.path.join(HERE, "_big.ppm") + write_ppm(bigppm, bw, bh, bytes(big)) + cjpeg(bigppm, j("prog_large.jpg"), + ["-quality", "92", "-progressive", "-sample", "2x2"], + "cjpeg -quality 92 -progressive -sample 2x2 (1024x768 synthetic source)") + os.remove(bigppm) + + # Trailing garbage after EOI (Annex A "tail data"). + with open(j("base_444.jpg"), "rb") as f: + base = f.read() + with open(j("trailing_garbage.jpg"), "wb") as f: + f.write(base + b"\x00\x01\x02trailing-bytes-after-EOI\xff\xd9\xde\xad") + provenance(j("trailing_garbage.jpg"), + "base_444.jpg with literal trailing bytes appended after EOI") + + # Exif + ICC via Pillow (baseline). Kept optional so the core set does not + # depend on Pillow being importable. + try: + from PIL import Image + img = Image.frombytes("RGB", (W, H), color) + # A minimal but valid ICC profile and Exif blob. + exif = img.getexif() + exif[0x010E] = "jpxl-jpeg synthetic fixture" # ImageDescription + exif[0x0131] = "generate.py" # Software + # A tiny sRGB-ish ICC stub is not a valid profile; instead let Pillow + # attach a real one if littlecms is available, else skip ICC. + icc = b"" + try: + from PIL import ImageCms + prof = ImageCms.createProfile("sRGB") + icc = ImageCms.ImageCmsProfile(prof).tobytes() + except Exception: + icc = b"" + save_kwargs = dict(format="JPEG", quality=88, subsampling=0, + exif=exif.tobytes()) + if icc: + save_kwargs["icc_profile"] = icc + img.save(j("meta_exif_icc.jpg"), **save_kwargs) + provenance(j("meta_exif_icc.jpg"), + "Pillow img.save(quality=88, subsampling=0, exif=..., icc_profile=sRGB)") + except Exception as e: # pragma: no cover + sys.stderr.write(f"skipping Pillow Exif/ICC fixture: {e}\n") + + # XMP metadata (APP1 with the XMP namespace) via Pillow (baseline). + try: + from PIL import Image + img = Image.frombytes("RGB", (W, H), color) + xmp = ( + b'' + b'' + b'' + b"jpxl-jpeg synthetic fixture" + b"" + ) + img.save(j("meta_xmp.jpg"), format="JPEG", quality=88, subsampling=2, xmp=xmp) + provenance(j("meta_xmp.jpg"), + "Pillow img.save(quality=88, subsampling=2, xmp=)") + except Exception as e: # pragma: no cover + sys.stderr.write(f"skipping XMP fixture: {e}\n") + + # Arithmetic-coded JPEG — a *refusal* fixture (must be rejected, not decoded). + try: + cjpeg(ppm, j("refuse_arithmetic.jpg"), + ["-quality", "85", "-arithmetic", "-sample", "2x2"], + "cjpeg -quality 85 -arithmetic -sample 2x2") + except subprocess.CalledProcessError as e: # pragma: no cover + sys.stderr.write(f"skipping arithmetic fixture: {e}\n") + + os.remove(ppm) + print("fixtures generated in", HERE) + + +if __name__ == "__main__": + main() diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_exif_icc.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_exif_icc.jpg new file mode 100644 index 00000000..14e1e3ae Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_exif_icc.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_exif_icc.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_exif_icc.jpg.prov new file mode 100644 index 00000000..b5dcdbcd --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_exif_icc.jpg.prov @@ -0,0 +1,5 @@ +fixture: meta_exif_icc.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 63f9972564aada07559d2c3c5fc15e94d6e09f3d39b076defe5ce34de38ccf1b +regenerate: Pillow img.save(quality=88, subsampling=0, exif=..., icc_profile=sRGB) diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_xmp.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_xmp.jpg new file mode 100644 index 00000000..650053cf Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_xmp.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_xmp.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_xmp.jpg.prov new file mode 100644 index 00000000..655bd065 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/meta_xmp.jpg.prov @@ -0,0 +1,5 @@ +fixture: meta_xmp.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 6beeb6cba6d460af850d0103d3210e64765d670a87dbda5c5617deec8c482d83 +regenerate: Pillow img.save(quality=88, subsampling=2, xmp=<XMP packet>) diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_420.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_420.jpg new file mode 100644 index 00000000..3c22a866 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_420.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_420.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_420.jpg.prov new file mode 100644 index 00000000..7c4698bc --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_420.jpg.prov @@ -0,0 +1,5 @@ +fixture: prog_420.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 56a386d62f8f0102020db6c11d2ba97fc93f3f775a4653744b7057a08f02a159 +regenerate: cjpeg -quality 85 -progressive -sample 2x2 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_444.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_444.jpg new file mode 100644 index 00000000..090dc7ae Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_444.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_444.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_444.jpg.prov new file mode 100644 index 00000000..790c37f7 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_444.jpg.prov @@ -0,0 +1,5 @@ +fixture: prog_444.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: f19b6f578043f336808342fde14acce074206eff50284a7a4e6e03e989eba1b7 +regenerate: cjpeg -quality 85 -progressive -sample 1x1 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_gray.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_gray.jpg new file mode 100644 index 00000000..2d4ec3a2 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_gray.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_gray.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_gray.jpg.prov new file mode 100644 index 00000000..3c316a68 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_gray.jpg.prov @@ -0,0 +1,5 @@ +fixture: prog_gray.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: a865c757ef16dbc30281a7a81bd2be40f189564559664fec6bb7d1ebbda24a45 +regenerate: cjpeg -quality 85 -progressive -grayscale diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_large.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_large.jpg new file mode 100644 index 00000000..2221a350 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_large.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_large.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_large.jpg.prov new file mode 100644 index 00000000..00e3ceef --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_large.jpg.prov @@ -0,0 +1,5 @@ +fixture: prog_large.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 2b4cdeee8298264627c1ac4d4cd246bcdf399bb17df310714e185e4aab1be708 +regenerate: cjpeg -quality 92 -progressive -sample 2x2 (1024x768 synthetic source) diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_q30.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_q30.jpg new file mode 100644 index 00000000..700663b7 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_q30.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_q30.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_q30.jpg.prov new file mode 100644 index 00000000..cf2aee3d --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_q30.jpg.prov @@ -0,0 +1,5 @@ +fixture: prog_q30.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 8d81d5e4f2698388af0eb6854a27c49c19b85c9a6cb0d53ac57693e3b03c18c8 +regenerate: cjpeg -quality 30 -progressive -sample 2x2 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_restart.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_restart.jpg new file mode 100644 index 00000000..0f5432f1 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_restart.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_restart.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_restart.jpg.prov new file mode 100644 index 00000000..f0e744ac --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/prog_restart.jpg.prov @@ -0,0 +1,5 @@ +fixture: prog_restart.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 809213f5fd6990175934c25e03eeb4a2c43d924256878ef82a9e50e98473c801 +regenerate: cjpeg -quality 90 -progressive -sample 2x2 -restart 4 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/refuse_arithmetic.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/refuse_arithmetic.jpg new file mode 100644 index 00000000..5230957e Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/refuse_arithmetic.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/refuse_arithmetic.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/refuse_arithmetic.jpg.prov new file mode 100644 index 00000000..47289a89 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/refuse_arithmetic.jpg.prov @@ -0,0 +1,5 @@ +fixture: refuse_arithmetic.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: 2b41fb4ffef97ad0f978e0d4aa2cff0abc7abdfde1104692e4159199f6e7a989 +regenerate: cjpeg -quality 85 -arithmetic -sample 2x2 diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/trailing_garbage.jpg b/JPXL/crates/jpxl-jpeg/tests/fixtures/trailing_garbage.jpg new file mode 100644 index 00000000..08197187 Binary files /dev/null and b/JPXL/crates/jpxl-jpeg/tests/fixtures/trailing_garbage.jpg differ diff --git a/JPXL/crates/jpxl-jpeg/tests/fixtures/trailing_garbage.jpg.prov b/JPXL/crates/jpxl-jpeg/tests/fixtures/trailing_garbage.jpg.prov new file mode 100644 index 00000000..e3bbefcb --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/fixtures/trailing_garbage.jpg.prov @@ -0,0 +1,5 @@ +fixture: trailing_garbage.jpg +origin: synthesised by tests/fixtures/generate.py (original content) +license: CC0-1.0 / public domain (no third-party image used) +sha256: aa55e7e2147339bfb7d64d33bdfe35a273d055915db0d7083e8ef0e3d9982bcc +regenerate: base_444.jpg with literal trailing bytes appended after EOI diff --git a/JPXL/crates/jpxl-jpeg/tests/progressive_eob.rs b/JPXL/crates/jpxl-jpeg/tests/progressive_eob.rs new file mode 100644 index 00000000..5067ff24 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/progressive_eob.rs @@ -0,0 +1,74 @@ +//! Regression: progressive AC EOB-run lengths must be recorded at decode and +//! replayed at encode, not re-derived. +//! +//! An encoder may split one end-of-band run into several `EOBn` codes (libjpeg +//! flushes when its correction-bit buffer fills). That split is invisible in +//! the coefficients, so a re-encoder that greedily re-derives run lengths emits +//! a different bit count — the "N bits pending after padding" and +//! "symbol 0xN0 absent from Huffman table" failures seen on the archive sweep. +//! `prog_large.jpg` is large and detailed enough to force such splits. + +#![allow( + clippy::indexing_slicing, + clippy::unwrap_used, + reason = "test indexes its own just-loaded data; a panic is a failing test" +)] + +use std::path::PathBuf; + +use jpxl_jpeg::{Segment, parse, serialize}; + +fn fixture(name: &str) -> Vec<u8> { + let mut p = PathBuf::from(env!("CARGO_MANIFEST_DIR")); + p.push("tests/fixtures"); + p.push(name); + std::fs::read(&p).unwrap_or_else(|e| panic!("reading {name}: {e}")) +} + +#[test] +fn roundtrip_progressive_large_is_bit_exact() { + let original = fixture("prog_large.jpg"); + let jpeg = parse(&original).expect("parse"); + let round = serialize(&jpeg).expect("serialize"); + assert_eq!( + original, round, + "prog_large.jpg must round-trip byte-exactly" + ); +} + +#[test] +fn recorded_eob_runs_are_load_bearing() { + let original = fixture("prog_large.jpg"); + let jpeg = parse(&original).expect("parse"); + + // The fixture must actually contain a *split* EOB run for this test to + // prove anything: two consecutive recorded runs whose total the greedy + // re-derivation would have merged into one. Any scan with >1 recorded run + // has been split by the original encoder. + let split_seen = jpeg.segments.iter().any(|s| match s { + Segment::Sos(scan) => scan.eob_runs.len() > 1, + _ => false, + }); + assert!( + split_seen, + "prog_large.jpg should exercise a multi-EOBn (split) run" + ); + + // With the recorded runs cleared, the encoder must fall back to greedy + // re-derivation, which cannot reproduce the split — so the output must + // differ from the original (or fail outright). Either proves the recorded + // runs are necessary for bit-exactness. + let mut stripped = jpeg.clone(); + for seg in &mut stripped.segments { + if let Segment::Sos(scan) = seg { + scan.eob_runs.clear(); + } + } + match serialize(&stripped) { + Ok(bytes) => assert_ne!( + bytes, original, + "clearing EOB runs must change the output (they are load-bearing)" + ), + Err(_) => { /* greedy re-derivation produced an illegal symbol: also proves the point */ } + } +} diff --git a/JPXL/crates/jpxl-jpeg/tests/refusal.rs b/JPXL/crates/jpxl-jpeg/tests/refusal.rs new file mode 100644 index 00000000..4aec6638 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/refusal.rs @@ -0,0 +1,137 @@ +//! Graceful, typed refusal of out-of-scope JPEG modes, and non-panicking +//! rejection of malformed input. +//! +//! Phase A supports Huffman-coded 8-bit baseline/progressive DCT only. Every +//! other well-formed JPEG mode must be rejected with [`JpegError::Unsupported`] +//! — never mis-parsed into silent garbage — and every malformed byte string +//! must return an `Err`, never panic (decode paths are attacker-facing). + +#![allow( + clippy::indexing_slicing, + clippy::unwrap_used, + reason = "tests index their own just-loaded data; a panic here is a failing \ + test, which is the intended signal" +)] + +use jpxl_jpeg::{JpegError, parse}; + +/// Builds a minimal `SOI` + one marker stream (enough to reach the code path +/// that classifies the marker). +fn soi_then(marker_code: u8, tail: &[u8]) -> Vec<u8> { + let mut v = vec![0xFF, 0xD8, 0xFF, marker_code]; + v.extend_from_slice(tail); + v +} + +#[test] +fn refuses_arithmetic_coded_jpeg_fixture() { + let data = std::fs::read(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/fixtures/refuse_arithmetic.jpg" + )) + .expect("arithmetic fixture"); + match parse(&data) { + Err(JpegError::Unsupported(_)) => {} + other => panic!("expected Unsupported for arithmetic JPEG, got {other:?}"), + } +} + +#[test] +fn refuses_lossless_sof3() { + // SOF3 is classified before its body is read. + assert!(matches!( + parse(&soi_then(0xC3, &[])), + Err(JpegError::Unsupported(_)) + )); +} + +#[test] +fn refuses_hierarchical_sof5() { + assert!(matches!( + parse(&soi_then(0xC5, &[])), + Err(JpegError::Unsupported(_)) + )); +} + +#[test] +fn refuses_arithmetic_sof9() { + assert!(matches!( + parse(&soi_then(0xC9, &[])), + Err(JpegError::Unsupported(_)) + )); +} + +#[test] +fn refuses_arithmetic_conditioning_dac() { + assert!(matches!( + parse(&soi_then(0xCC, &[])), + Err(JpegError::Unsupported(_)) + )); +} + +#[test] +fn refuses_hierarchical_dhp() { + assert!(matches!( + parse(&soi_then(0xDE, &[])), + Err(JpegError::Unsupported(_)) + )); +} + +#[test] +fn refuses_twelve_bit_precision() { + // A well-formed baseline SOF0 header claiming 12-bit samples. + let sof0_12bit = soi_then( + 0xC0, + &[ + 0x00, 0x11, // Lf = 17 + 0x0C, // P = 12-bit precision + 0x00, 0x10, // Y = 16 + 0x00, 0x10, // X = 16 + 0x03, // Nf = 3 + 0x01, 0x11, 0x00, // component 1 + 0x02, 0x11, 0x00, // component 2 + 0x03, 0x11, 0x00, // component 3 + ], + ); + assert!(matches!(parse(&sof0_12bit), Err(JpegError::Unsupported(_)))); +} + +// ---- Malformed input must error, not panic -------------------------------- + +#[test] +fn rejects_non_jpeg_bytes() { + assert!(parse(b"not a jpeg at all").is_err()); + assert!(parse(&[]).is_err()); + assert!(parse(&[0xFF]).is_err()); + assert!(parse(&[0xFF, 0xD8]).is_err()); // SOI then nothing +} + +#[test] +fn rejects_truncated_baseline_without_panic() { + // Take a real baseline JPEG and truncate it at many offsets; none may panic. + let data = std::fs::read(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/fixtures/base_444.jpg" + )) + .unwrap(); + for cut in (2..data.len()).step_by(97) { + // Must return Ok or Err, but never unwind. + let _ = parse(&data[..cut]); + } +} + +#[test] +fn rejects_corrupted_entropy_without_panic() { + let mut data = std::fs::read(concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/fixtures/prog_420.jpg" + )) + .unwrap(); + // Flip bytes across the stream; parsing must stay panic-free. + for i in (0..data.len()).step_by(31) { + let saved = data[i]; + data[i] ^= 0xA5; + let _ = parse(&data); + data[i] = saved; + } +} diff --git a/JPXL/crates/jpxl-jpeg/tests/roundtrip.rs b/JPXL/crates/jpxl-jpeg/tests/roundtrip.rs new file mode 100644 index 00000000..1aeb29e2 --- /dev/null +++ b/JPXL/crates/jpxl-jpeg/tests/roundtrip.rs @@ -0,0 +1,189 @@ +//! Byte-exact `serialize(parse(x)) == x` round-trip over real JPEG fixtures. +//! +//! Every fixture is produced by libjpeg-turbo (`cjpeg`) or Pillow from +//! synthetic original content; see `tests/fixtures/generate.py` and the +//! per-file `*.jpg.prov` provenance sidecars. The images are 385×259 — many +//! MCUs, non-multiple-of-16 edges, and (for subsampled cases) differing +//! interleaved / non-interleaved block counts — so the round-trip exercises +//! edge padding blocks, restart cadence, and padding bits, not just a happy +//! single-block path. + +#![allow( + clippy::indexing_slicing, + clippy::unwrap_used, + reason = "tests index their own just-loaded data; a panic here is a failing \ + test, which is the intended signal" +)] + +use std::path::PathBuf; + +use jpxl_jpeg::{parse, serialize}; + +fn fixture(name: &str) -> Vec<u8> { + let mut p = PathBuf::from(env!("CARGO_MANIFEST_DIR")); + p.push("tests/fixtures"); + p.push(name); + std::fs::read(&p).unwrap_or_else(|e| panic!("reading fixture {name}: {e}")) +} + +/// Parses then re-serializes, asserting byte-for-byte identity, and returns the +/// parsed model so callers can additionally assert structural facts. +fn assert_roundtrip(name: &str) -> jpxl_jpeg::Jpeg { + let original = fixture(name); + let jpeg = parse(&original).unwrap_or_else(|e| panic!("parse {name}: {e}")); + let reemitted = serialize(&jpeg).unwrap_or_else(|e| panic!("serialize {name}: {e}")); + assert_eq!( + original.len(), + reemitted.len(), + "{name}: length differs ({} vs {})", + original.len(), + reemitted.len() + ); + if original != reemitted { + // Report the first divergence to make failures debuggable. + let at = original + .iter() + .zip(&reemitted) + .position(|(a, b)| a != b) + .unwrap_or(0); + panic!( + "{name}: byte mismatch at offset {at}: original 0x{:02X} != re-emitted 0x{:02X}", + original[at], reemitted[at] + ); + } + jpeg +} + +// ---- Baseline sequential --------------------------------------------------- + +#[test] +fn roundtrip_baseline_444_is_bit_exact() { + let j = assert_roundtrip("base_444.jpg"); + let f = j.frame.as_ref().expect("frame"); + assert_eq!((f.width, f.height), (385, 259)); + assert_eq!(f.components.len(), 3, "YCbCr"); + // Prove parsing produced real coefficients, not an empty pass-through. + let any_ac = j + .planes + .iter() + .any(|p| p.blocks.iter().any(|b| b.iter().skip(1).any(|&c| c != 0))); + assert!(any_ac, "expected some nonzero AC coefficients"); +} + +#[test] +fn roundtrip_baseline_422_is_bit_exact() { + let j = assert_roundtrip("base_422.jpg"); + let f = j.frame.as_ref().unwrap(); + assert_eq!((f.components[0].h, f.components[0].v), (2, 1)); +} + +#[test] +fn roundtrip_baseline_420_is_bit_exact() { + let j = assert_roundtrip("base_420.jpg"); + let f = j.frame.as_ref().unwrap(); + assert_eq!((f.components[0].h, f.components[0].v), (2, 2)); +} + +#[test] +fn roundtrip_baseline_440_is_bit_exact() { + assert_roundtrip("base_440.jpg"); +} + +#[test] +fn roundtrip_baseline_grayscale_is_bit_exact() { + let j = assert_roundtrip("base_gray.jpg"); + assert_eq!(j.frame.as_ref().unwrap().components.len(), 1); +} + +#[test] +fn roundtrip_baseline_high_quality_dense_coeffs_is_bit_exact() { + assert_roundtrip("base_q98.jpg"); +} + +#[test] +fn roundtrip_baseline_low_quality_sparse_coeffs_is_bit_exact() { + assert_roundtrip("base_q20.jpg"); +} + +// ---- Restart intervals ----------------------------------------------------- + +#[test] +fn roundtrip_baseline_restart_interval_is_bit_exact() { + let j = assert_roundtrip("base_restart.jpg"); + // The restart interval must have been captured as a DRI segment. + let has_dri = j + .segments + .iter() + .any(|s| matches!(s, jpxl_jpeg::Segment::Dri(_))); + assert!(has_dri, "expected a DRI segment"); +} + +#[test] +fn roundtrip_baseline_restart_rows_is_bit_exact() { + assert_roundtrip("base_restart_rows.jpg"); +} + +// ---- Progressive ----------------------------------------------------------- + +#[test] +fn roundtrip_progressive_444_is_bit_exact() { + let j = assert_roundtrip("prog_444.jpg"); + // A progressive frame has multiple SOS segments (spectral selection). + let scans = j + .segments + .iter() + .filter(|s| matches!(s, jpxl_jpeg::Segment::Sos(_))) + .count(); + assert!( + scans > 1, + "progressive stream should have several scans, got {scans}" + ); +} + +#[test] +fn roundtrip_progressive_420_is_bit_exact() { + assert_roundtrip("prog_420.jpg"); +} + +#[test] +fn roundtrip_progressive_grayscale_is_bit_exact() { + assert_roundtrip("prog_gray.jpg"); +} + +#[test] +fn roundtrip_progressive_restart_is_bit_exact() { + assert_roundtrip("prog_restart.jpg"); +} + +#[test] +fn roundtrip_progressive_low_quality_is_bit_exact() { + assert_roundtrip("prog_q30.jpg"); +} + +// ---- Metadata and trailing data ------------------------------------------- + +#[test] +fn roundtrip_exif_icc_metadata_is_bit_exact() { + let j = assert_roundtrip("meta_exif_icc.jpg"); + let has_app = j + .segments + .iter() + .any(|s| matches!(s, jpxl_jpeg::Segment::App(_))); + assert!(has_app, "expected APPn metadata segments"); +} + +#[test] +fn roundtrip_xmp_metadata_is_bit_exact() { + let j = assert_roundtrip("meta_xmp.jpg"); + let has_app = j + .segments + .iter() + .any(|s| matches!(s, jpxl_jpeg::Segment::App(_))); + assert!(has_app, "expected an APP1 XMP segment"); +} + +#[test] +fn roundtrip_trailing_garbage_after_eoi_is_bit_exact() { + let j = assert_roundtrip("trailing_garbage.jpg"); + assert!(!j.tail.is_empty(), "trailing bytes should be captured"); +} diff --git a/JPXL/crates/jpxl-perceptual/src/blur.rs b/JPXL/crates/jpxl-perceptual/src/blur.rs index 1fc17ae5..ab44ddc5 100644 --- a/JPXL/crates/jpxl-perceptual/src/blur.rs +++ b/JPXL/crates/jpxl-perceptual/src/blur.rs @@ -26,7 +26,8 @@ //! Each output is computed from its row (or column) alone, so any partition //! into rows or column strips gives bit-identical results; the horizontal //! pass runs its rows in fixed bands on the executor and the vertical pass -//! walks column strips on the calling thread. +//! runs its column strips there too, each strip writing its own disjoint +//! columns of the output in place. use crate::bands::{BAND_ROWS, band_of, mutable_bands}; use crate::executor::BandExecutor; @@ -126,7 +127,8 @@ impl Blur { if width == 0 || height == 0 { return; } - self.temp.clear(); + // The horizontal pass writes every temp sample, so a reused temp needs + // no re-zeroing; resize only grows (zeroed) or truncates. self.temp.resize(width * height, 0.0); let band_len = width.saturating_mul(BAND_ROWS); let bands = mutable_bands(vec![self.temp.as_mut_slice()], band_len); @@ -137,29 +139,13 @@ impl Blur { let Some(out_band) = outs.pop() else { return; }; - let mut padded = Vec::new(); - match input { - BlurInput::Plane(input) => { - let in_band = band_of(input, index, band_len); - for (row_in, row_out) in in_band - .chunks_exact(width) - .zip(out_band.chunks_exact_mut(width)) - { - horizontal_row(row_in, row_out, &mut padded); - } - } + let band_input = match input { + BlurInput::Plane(input) => BlurInput::Plane(band_of(input, index, band_len)), BlurInput::Product(a, b) => { - let a_band = band_of(a, index, band_len); - let b_band = band_of(b, index, band_len); - for ((a_row, b_row), row_out) in a_band - .chunks_exact(width) - .zip(b_band.chunks_exact(width)) - .zip(out_band.chunks_exact_mut(width)) - { - horizontal_product_row(a_row, b_row, row_out, &mut padded); - } + BlurInput::Product(band_of(a, index, band_len), band_of(b, index, band_len)) } - } + }; + horizontal_band(band_input, out_band, width); }); vertical_pass(&self.temp, output, width, height, executor); } @@ -188,6 +174,193 @@ fn step(sum: f32, prev: &mut [f64; 3], prev2: &mut [f64; 3]) -> f32 { const LEFT_PAD: usize = 2 * RADIUS_USIZE; const RADIUS_USIZE: usize = 5; +/// Rows processed together in the horizontal pass: each is one independent +/// recursion lane (the row analogue of the vertical pass's column lanes), so +/// the compiler vectorises the lane loop. +const ROW_LANES: usize = 4; + +/// Horizontal pass over one band: groups of [`ROW_LANES`] rows through the +/// lane recursion, remainder rows through the scalar one. Every row's output +/// depends on that row alone and the lane arithmetic is exactly [`step`]'s, +/// so the grouping cannot change a value. Dispatched to an AVX2 build where +/// the host supports it. +fn horizontal_band(input: BlurInput<'_>, out_band: &mut [f32], width: usize) { + #[cfg(target_arch = "x86_64")] + if jpxl_core::cpu::has_avx2() { + // SAFETY: `horizontal_band_avx2` only requires that the host support + // AVX2, which `has_avx2` has just confirmed. + #[allow(unsafe_code)] + unsafe { + horizontal_band_avx2(input, out_band, width); + } + return; + } + horizontal_band_impl(input, out_band, width); +} + +/// [`horizontal_band`] compiled for AVX2. +/// +/// Calling it is `unsafe` unless the host supports AVX2 (see +/// [`jpxl_core::cpu::has_avx2`]); that is the whole contract. +#[cfg(target_arch = "x86_64")] +#[target_feature(enable = "avx2,fma")] +fn horizontal_band_avx2(input: BlurInput<'_>, out_band: &mut [f32], width: usize) { + horizontal_band_impl(input, out_band, width); +} + +#[inline(always)] +fn horizontal_band_impl(input: BlurInput<'_>, out_band: &mut [f32], width: usize) { + if width == 0 { + return; + } + let group = width * ROW_LANES; + let mut padded_t = Vec::new(); + let mut padded = Vec::new(); + match input { + BlurInput::Plane(in_band) => { + let mut ins = in_band.chunks_exact(group); + let mut outs = out_band.chunks_exact_mut(group); + for (rows_in, rows_out) in ins.by_ref().zip(outs.by_ref()) { + build_padded_lanes(BlurInput::Plane(rows_in), width, &mut padded_t); + horizontal_padded_lanes(rows_out, width, &padded_t); + } + for (row_in, row_out) in ins + .remainder() + .chunks_exact(width) + .zip(outs.into_remainder().chunks_exact_mut(width)) + { + horizontal_row(row_in, row_out, &mut padded); + } + } + BlurInput::Product(a, b) => { + let mut a_groups = a.chunks_exact(group); + let mut b_groups = b.chunks_exact(group); + let mut outs = out_band.chunks_exact_mut(group); + for ((a_rows, b_rows), rows_out) in + a_groups.by_ref().zip(b_groups.by_ref()).zip(outs.by_ref()) + { + build_padded_lanes(BlurInput::Product(a_rows, b_rows), width, &mut padded_t); + horizontal_padded_lanes(rows_out, width, &padded_t); + } + for ((a_row, b_row), row_out) in a_groups + .remainder() + .chunks_exact(width) + .zip(b_groups.remainder().chunks_exact(width)) + .zip(outs.into_remainder().chunks_exact_mut(width)) + { + horizontal_product_row(a_row, b_row, row_out, &mut padded); + } + } + } +} + +/// Builds the lane-interleaved padded rows: sample `x` of lane `r` sits at +/// `(LEFT_PAD + x) * ROW_LANES + r`, so each recursion step loads one +/// contiguous [`ROW_LANES`]-wide window. The pads stay zero from the first +/// allocation and the body is fully overwritten, exactly as the scalar +/// padded row is. Product lanes round `a * b` into `f32` here, exactly as +/// [`horizontal_product_row`] does. +#[inline(always)] +fn build_padded_lanes(input: BlurInput<'_>, width: usize, padded_t: &mut Vec<f32>) { + let n = (width + 3 * RADIUS_USIZE) * ROW_LANES; + if padded_t.len() != n { + padded_t.clear(); + padded_t.resize(n, 0.0); + } + let Some(body) = padded_t.get_mut(LEFT_PAD * ROW_LANES..(LEFT_PAD + width) * ROW_LANES) else { + return; + }; + match input { + BlurInput::Plane(rows) => { + for (r, row) in rows.chunks_exact(width).enumerate() { + for (slot, &v) in body.iter_mut().skip(r).step_by(ROW_LANES).zip(row) { + *slot = v; + } + } + } + BlurInput::Product(a, b) => { + for (r, (a_row, b_row)) in a.chunks_exact(width).zip(b.chunks_exact(width)).enumerate() + { + for ((slot, &av), &bv) in body + .iter_mut() + .skip(r) + .step_by(ROW_LANES) + .zip(a_row) + .zip(b_row) + { + *slot = av * bv; + } + } + } + } +} + +/// The recursion over [`ROW_LANES`] padded rows at once: the same warm-up and +/// the same per-sample [`pole_step`] arithmetic as [`horizontal_padded`], +/// lane-parallel across the rows. +#[inline(always)] +fn horizontal_padded_lanes(out: &mut [f32], width: usize, padded_t: &[f32]) { + let mut prev = [[0.0f64; ROW_LANES]; 3]; + let mut prev2 = [[0.0f64; ROW_LANES]; 3]; + let warm = RADIUS_USIZE - 1; + let lane_sum = |i: usize| -> [f64; ROW_LANES] { + let mut sum = [0.0f64; ROW_LANES]; + let l = padded_t.get(i * ROW_LANES..(i + 1) * ROW_LANES); + let r = padded_t.get((i + LEFT_PAD) * ROW_LANES..(i + LEFT_PAD + 1) * ROW_LANES); + if let (Some(l), Some(r)) = (l, r) { + for ((s, &lv), &rv) in sum.iter_mut().zip(l).zip(r) { + *s = f64::from(lv + rv); + } + } + sum + }; + for i in 0..warm { + step_lanes(lane_sum(i), &mut prev, &mut prev2); + } + let mut rows = out.chunks_exact_mut(width); + let (Some(o0), Some(o1), Some(o2), Some(o3)) = + (rows.next(), rows.next(), rows.next(), rows.next()) + else { + return; + }; + for ((((s0, s1), s2), s3), n) in o0 + .iter_mut() + .zip(o1.iter_mut()) + .zip(o2.iter_mut()) + .zip(o3.iter_mut()) + .zip(0..width) + { + let [v0, v1, v2, v3] = step_lanes(lane_sum(n + warm), &mut prev, &mut prev2); + *s0 = v0; + *s1 = v1; + *s2 = v2; + *s3 = v3; + } +} + +/// One three-component recursion step over [`ROW_LANES`] independent lanes. +/// Per lane this is exactly [`step`]: the same expression order, the same +/// `o1 + o3 + o5` summation, one `f32` rounding of the summed output. +#[allow(clippy::cast_possible_truncation)] +#[inline(always)] +fn step_lanes( + sum: [f64; ROW_LANES], + prev: &mut [[f64; ROW_LANES]; 3], + prev2: &mut [[f64; ROW_LANES]; 3], +) -> [f32; ROW_LANES] { + let [p1, p3, p5] = prev; + let [q1, q3, q5] = prev2; + let mut acc = [0.0f64; ROW_LANES]; + pole_step(&sum, p1, q1, &mut acc, MUL_IN[0], MUL_PREV[0], true); + pole_step(&sum, p3, q3, &mut acc, MUL_IN[1], MUL_PREV[1], false); + pole_step(&sum, p5, q5, &mut acc, MUL_IN[2], MUL_PREV[2], false); + let mut out = [0.0f32; ROW_LANES]; + for (o, &a) in out.iter_mut().zip(&acc) { + *o = a as f32; + } + out +} + /// Horizontal recursive pass over one row, through a zero-padded copy so the /// window reads are plain slice walks with no per-sample bounds branch. fn horizontal_row(input: &[f32], output: &mut [f32], padded: &mut Vec<f32>) { @@ -236,8 +409,11 @@ fn horizontal_padded(output: &mut [f32], padded: &[f32]) { /// executor item writing its own buffer. const STRIP: usize = 64; -/// Vertical recursive pass over every column, strips in parallel, then one -/// ordered scatter into the row-major output. +/// Vertical recursive pass over every column, strips in parallel, each strip +/// writing its own disjoint column range of the row-major output directly. +/// The arithmetic per column is untouched — only the destination addressing +/// changed from a per-strip buffer plus scatter to in-place row windows — so +/// the output is bit-identical to the former scatter form. fn vertical_pass( input: &[f32], output: &mut [f32], @@ -246,29 +422,53 @@ fn vertical_pass( executor: &dyn BandExecutor, ) { let strips = width.div_ceil(STRIP); - let buffers: Vec<std::sync::Mutex<Vec<f32>>> = (0..strips) - .map(|_| std::sync::Mutex::new(Vec::new())) - .collect(); + let out = DisjointColumns::new(output); executor.run(strips, &|index| { let x0 = index * STRIP; let cols = (width - x0).min(STRIP); - let mut buffer = vec![0.0f32; cols * height]; - vertical_strip(input, &mut buffer, width, height, x0, cols); - if let Some(slot) = buffers.get(index) - && let Ok(mut guard) = slot.lock() - { - *guard = buffer; - } + vertical_strip(input, &out, width, height, x0, cols); }); - for (index, slot) in buffers.into_iter().enumerate() { - let buffer = slot.into_inner().unwrap_or_default(); - let x0 = index * STRIP; - let cols = (width - x0).min(STRIP); - for (y, src) in buffer.chunks_exact(cols).enumerate() { - if let Some(dst) = output.get_mut(y * width + x0..y * width + x0 + cols) { - dst.copy_from_slice(src); - } +} + +/// A row-major plane shared across strip workers, each writing row windows of +/// a column range no other worker touches. +struct DisjointColumns { + ptr: *mut f32, + len: usize, +} + +// SAFETY: every worker writes only row windows of its own `x0 .. x0 + cols` +// column range, and the strip ranges partition the columns, so no element is +// ever aliased by two workers. +#[allow(unsafe_code)] +unsafe impl Sync for DisjointColumns {} + +impl DisjointColumns { + fn new(plane: &mut [f32]) -> Self { + Self { + ptr: plane.as_mut_ptr(), + len: plane.len(), + } + } + + /// The `cols` samples starting at `offset`, as one mutable row window; + /// `None` when the window leaves the plane. + /// + /// # Safety + /// + /// No other thread may read or write this window for the returned + /// borrow's lifetime; the strip partition guarantees that here. + // The `&self`-to-`&mut` shape is the point of the type: it is a manual + // interior-mutability cell whose disjointness contract lives in `unsafe`. + #[allow(unsafe_code, clippy::mut_from_ref)] + unsafe fn window(&self, offset: usize, cols: usize) -> Option<&mut [f32]> { + let end = offset.checked_add(cols)?; + if end > self.len { + return None; } + // SAFETY: the range is in bounds (checked above) and unaliased (the + // caller's contract). + Some(unsafe { core::slice::from_raw_parts_mut(self.ptr.add(offset), cols) }) } } @@ -353,12 +553,12 @@ fn vertical_row(top: &[f32], bottom: &[f32], state: &mut StripState, out: Option } } -/// Vertical pass over columns `x0 .. x0 + cols` of `input`, writing the -/// strip row-major (`cols` per row) into `strip_out`. Dispatched to an AVX2 -/// build where the host supports it. +/// Vertical pass over columns `x0 .. x0 + cols` of `input`, writing each +/// row's window of `out` in place. Dispatched to an AVX2 build where the host +/// supports it. fn vertical_strip( input: &[f32], - strip_out: &mut [f32], + out: &DisjointColumns, width: usize, height: usize, x0: usize, @@ -370,11 +570,11 @@ fn vertical_strip( // AVX2, which `has_avx2` has just confirmed. #[allow(unsafe_code)] unsafe { - vertical_strip_avx2(input, strip_out, width, height, x0, cols); + vertical_strip_avx2(input, out, width, height, x0, cols); } return; } - vertical_strip_impl(input, strip_out, width, height, x0, cols); + vertical_strip_impl(input, out, width, height, x0, cols); } /// [`vertical_strip`] compiled for AVX2. @@ -385,19 +585,19 @@ fn vertical_strip( #[target_feature(enable = "avx2,fma")] fn vertical_strip_avx2( input: &[f32], - strip_out: &mut [f32], + out: &DisjointColumns, width: usize, height: usize, x0: usize, cols: usize, ) { - vertical_strip_impl(input, strip_out, width, height, x0, cols); + vertical_strip_impl(input, out, width, height, x0, cols); } #[inline(always)] fn vertical_strip_impl( input: &[f32], - strip_out: &mut [f32], + out: &DisjointColumns, width: usize, height: usize, x0: usize, @@ -419,9 +619,12 @@ fn vertical_strip_impl( while n < h { let top = row(n - RADIUS - 1); let bottom = row(n + RADIUS - 1); + // SAFETY: this strip's `x0 .. x0 + cols` columns are its own — see + // `DisjointColumns` — and row `n` is in bounds for `0 <= n < h`. + #[allow(unsafe_code)] let out_row = usize::try_from(n) .ok() - .and_then(|n| strip_out.get_mut(n * cols..n * cols + cols)); + .and_then(|n| unsafe { out.window(n * width + x0, cols) }); vertical_row(top, bottom, &mut state, out_row); n += 1; } @@ -464,14 +667,9 @@ mod tests { horizontal_row(row_in, row_out, &mut padded); } let mut strips = vec![0.0f32; w * h]; - let mut left = vec![0.0f32; 50 * h]; - let mut right = vec![0.0f32; 50 * h]; - vertical_strip(&temp, &mut left, w, h, 0, 50); - vertical_strip(&temp, &mut right, w, h, 50, 50); - for y in 0..h { - strips[y * w..y * w + 50].copy_from_slice(&left[y * 50..y * 50 + 50]); - strips[y * w + 50..y * w + 100].copy_from_slice(&right[y * 50..y * 50 + 50]); - } + let out = DisjointColumns::new(&mut strips); + vertical_strip(&temp, &out, w, h, 0, 50); + vertical_strip(&temp, &out, w, h, 50, 50); assert_eq!(whole, strips); } diff --git a/JPXL/crates/jpxl-perceptual/src/evaluator.rs b/JPXL/crates/jpxl-perceptual/src/evaluator.rs index 42a194bb..c4b48677 100644 --- a/JPXL/crates/jpxl-perceptual/src/evaluator.rs +++ b/JPXL/crates/jpxl-perceptual/src/evaluator.rs @@ -103,8 +103,13 @@ impl<'e> PlanRenderEvaluator<'e> { executor: &'e EncodeExecutor, ) -> Result<Self, EvaluatorError> { let max = f32::from(u16::MAX).min(((1u32 << bits_per_sample.clamp(1, 16)) - 1) as f32); + // The sample domain is u16, so the per-sample transfer function is a + // table of the same expression evaluated once per distinct value. + let lut: Vec<f32> = (0..=u32::from(u16::MAX)) + .map(|v| jpxl_core::color::srgb_to_linear(v as f32 / max)) + .collect(); let planes = deinterleave(rgb.chunks_exact(3), |v| { - jpxl_core::color::srgb_to_linear(f32::from(v) / max) + lut.get(usize::from(v)).copied().unwrap_or(0.0) }); Self::from_linear_planes(width, height, planes, bits_per_sample, executor) } @@ -153,18 +158,13 @@ impl PerceptualEvaluator for PlanRenderEvaluator<'_> { &mut self, candidate: &ValidatedPixelPlan, ) -> jpxl_encode_policy::Result<PerceptualObservation> { - let frame = self + let (width, height, linear) = self .renderer - .render_with(candidate, self.executor) + .render_linear_at_depth_with(candidate, self.bits_per_sample, self.executor) .map_err(|_| PolicyError::Unsupported { what: "a candidate plan the renderer could not reconstruct", })?; - let (width, height) = (frame.width(), frame.height()); - frame.linear_rgb_at_depth_into(self.bits_per_sample, &mut self.linear); - // The rendered frame is no longer needed — only its linearised planes - // are scored — so release it before the metric allocates its scratch, - // keeping both from being resident at once. - drop(frame); + self.linear = linear; let [r, g, b] = &self.linear; let view = LinearRgbView::new(width, height, r, g, b).map_err(|_| PolicyError::Unsupported { @@ -191,18 +191,16 @@ impl PerceptualEvaluator for PlanRenderEvaluator<'_> { return Ok((observation, Some(candidate))); } - let frame = self + let (width, height, linear) = self .renderer - .render_with(&candidate, self.executor) + .render_linear_at_depth_with(&candidate, self.bits_per_sample, self.executor) .map_err(|_| PolicyError::Unsupported { what: "a candidate plan the renderer could not reconstruct", })?; - let (width, height) = (frame.width(), frame.height()); // Once reconstruction is complete, the coefficient payload is not // needed for this score. Exact finalists are rebuilt deterministically // by the policy if this rung survives navigation. drop(candidate); - let linear = frame.into_linear_rgb_at_depth(self.bits_per_sample); let result = self .metric .score_owned(&self.reference, width, height, linear, self.executor) @@ -229,7 +227,12 @@ fn deinterleave<'a, T: Copy + 'a>( pixels: impl Iterator<Item = &'a [T]>, convert: impl Fn(T) -> f32, ) -> [Vec<f32>; 3] { - let mut planes: [Vec<f32>; 3] = [Vec::new(), Vec::new(), Vec::new()]; + let hint = pixels.size_hint().0; + let mut planes: [Vec<f32>; 3] = [ + Vec::with_capacity(hint), + Vec::with_capacity(hint), + Vec::with_capacity(hint), + ]; for px in pixels { for (plane, &v) in planes.iter_mut().zip(px) { plane.push(convert(v)); diff --git a/JPXL/crates/jpxl-perceptual/src/pool.rs b/JPXL/crates/jpxl-perceptual/src/pool.rs index f3c60a7e..9fa3d0bc 100644 --- a/JPXL/crates/jpxl-perceptual/src/pool.rs +++ b/JPXL/crates/jpxl-perceptual/src/pool.rs @@ -88,7 +88,51 @@ pub struct MomentBands<'a> { } /// Accumulates the three maps over one band of the seven planes. +/// +/// Dispatched to an AVX2 build where the host supports it. The lane grouping +/// cannot change a value: each pixel's two error terms depend on that pixel's +/// seven inputs alone, and the running `f64` sums are folded in the original +/// pixel order either way, so the vector build is bit-identical to the scalar +/// one (Rust performs no floating-point contraction). pub fn accumulate_band(bands: &MomentBands<'_>, sums: &mut MapSums) { + #[cfg(target_arch = "x86_64")] + if jpxl_core::cpu::has_avx2() { + // SAFETY: `accumulate_band_avx2` only requires that the host support + // AVX2, which `has_avx2` has just confirmed. + #[allow(unsafe_code)] + unsafe { + accumulate_band_avx2(bands, sums); + } + return; + } + accumulate_band_impl(bands, sums); +} + +/// [`accumulate_band`] compiled for AVX2. +/// +/// Calling it is `unsafe` unless the host supports AVX2 (see +/// [`jpxl_core::cpu::has_avx2`]); that is the whole contract. +#[cfg(target_arch = "x86_64")] +#[target_feature(enable = "avx2,fma")] +fn accumulate_band_avx2(bands: &MomentBands<'_>, sums: &mut MapSums) { + accumulate_band_impl(bands, sums); +} + +/// Pixels whose error terms are computed together as one vector lane block. +/// +/// Eight covers a full `f32` AVX2 register for the SSIM' arithmetic and two +/// `f64` registers for the asymmetry ratio, whose divide is the loop's +/// dominant cost. +const POOL_LANES: usize = 8; + +#[inline(always)] +#[allow( + clippy::indexing_slicing, + reason = "every range below ends at `whole`, which is at most the length \ + of the shortest plane, and every lane index is bounded by the \ + chunk arrays' own size" +)] +fn accumulate_band_impl(bands: &MomentBands<'_>, sums: &mut MapSums) { let MomentBands { img1, mu1, @@ -98,6 +142,94 @@ pub fn accumulate_band(bands: &MomentBands<'_>, sums: &mut MapSums) { s22, s12, } = *bands; + + // The scalar walk zips, stopping at the shortest plane; the lane walk + // must cover exactly the same pixels. + let n = img1 + .len() + .min(mu1.len()) + .min(s11.len()) + .min(img2.len()) + .min(mu2.len()) + .min(s22.len()) + .min(s12.len()); + let whole = n - n % POOL_LANES; + + let (i1c, _) = img1[..whole].as_chunks::<POOL_LANES>(); + let (m1c, _) = mu1[..whole].as_chunks::<POOL_LANES>(); + let (v11c, _) = s11[..whole].as_chunks::<POOL_LANES>(); + let (i2c, _) = img2[..whole].as_chunks::<POOL_LANES>(); + let (m2c, _) = mu2[..whole].as_chunks::<POOL_LANES>(); + let (v22c, _) = s22[..whole].as_chunks::<POOL_LANES>(); + let (v12c, _) = s12[..whole].as_chunks::<POOL_LANES>(); + + for ((((((i1, m1), v11), i2), m2), v22), v12) in i1c + .iter() + .zip(m1c) + .zip(v11c) + .zip(i2c) + .zip(m2c) + .zip(v22c) + .zip(v12c) + { + // Stage 1, lane-parallel: both per-pixel error terms. Nothing here + // reads the running sums, so the lanes are independent and the + // compiler is free to keep the two divides eight and four wide. + let mut ssim_d = [0.0f64; POOL_LANES]; + let mut asym = [0.0f64; POOL_LANES]; + for lane in 0..POOL_LANES { + let (i1, m1, v11) = (i1[lane], m1[lane], v11[lane]); + let (i2, m2, v22) = (i2[lane], m2[lane], v22[lane]); + let v12 = v12[lane]; + // SSIM' error. The luminance term keeps only its numerator + // (1 - (μ1-μ2)²); see the module docs for why the denominator is + // dropped. + let mu11 = m1 * m1; + let mu22 = m2 * m2; + let mu12 = m1 * m2; + let mu_diff = m1 - m2; + let num_m = 1.0 - mu_diff * mu_diff; + let num_s = 2.0 * (v12 - mu12) + SSIM_C2; + let denom_s = (v11 - mu11) + (v22 - mu22) + SSIM_C2; + ssim_d[lane] = (1.0 - f64::from((num_m * num_s) / denom_s)).max(0.0); + // Edge asymmetry: ratio of local high-pass magnitudes, minus one. + asym[lane] = + (1.0 + f64::from((i2 - m2).abs())) / (1.0 + f64::from((i1 - m1).abs())) - 1.0; + } + // Stage 2, ordered: fold into the running sums in pixel order, which + // keeps the result bit-identical to the fully scalar walk. + for lane in 0..POOL_LANES { + let d = ssim_d[lane]; + sums.ssim[0] += d; + sums.ssim[1] += d * d * d * d; + let d1 = asym[lane]; + let artifact = d1.max(0.0); + sums.artifact[0] += artifact; + sums.artifact[1] += artifact * artifact * artifact * artifact; + let lost = (-d1).max(0.0); + sums.detail_lost[0] += lost; + sums.detail_lost[1] += lost * lost * lost * lost; + } + } + + accumulate_scalar( + &[ + &img1[whole..], + &mu1[whole..], + &s11[whole..], + &img2[whole..], + &mu2[whole..], + &s22[whole..], + &s12[whole..], + ], + sums, + ); +} + +/// The original scalar walk, kept for the sub-lane remainder. +#[inline(always)] +fn accumulate_scalar(planes: &[&[f32]; 7], sums: &mut MapSums) { + let [img1, mu1, s11, img2, mu2, s22, s12] = *planes; for ((((((&i1, &m1), &v11), &i2), &m2), &v22), &v12) in img1 .iter() .zip(mu1) @@ -107,9 +239,6 @@ pub fn accumulate_band(bands: &MomentBands<'_>, sums: &mut MapSums) { .zip(s22) .zip(s12) { - // SSIM' error. The luminance term keeps only its numerator - // (1 - (μ1-μ2)²); see the module docs for why the denominator is - // dropped. let mu11 = m1 * m1; let mu22 = m2 * m2; let mu12 = m1 * m2; @@ -121,7 +250,6 @@ pub fn accumulate_band(bands: &MomentBands<'_>, sums: &mut MapSums) { sums.ssim[0] += d; sums.ssim[1] += d * d * d * d; - // Edge asymmetry: ratio of local high-pass magnitudes, minus one. let d1 = (1.0 + f64::from((i2 - m2).abs())) / (1.0 + f64::from((i1 - m1).abs())) - 1.0; let artifact = d1.max(0.0); sums.artifact[0] += artifact; @@ -158,6 +286,53 @@ mod tests { assert_eq!(ChannelTerms::from_sums(&sums, 4), ChannelTerms::default()); } + /// The dispatched walk (lane-grouped, AVX2 where the host has it) must be + /// bit-identical to the plain scalar walk — including on lengths that + /// leave a sub-lane remainder — because pooled sums feed the canonical + /// score, whose Contract A forbids host-dependent output. + #[test] + #[allow( + clippy::indexing_slicing, + clippy::cast_possible_truncation, + reason = "fixture generation and seven fixed-index plane borrows in a \ + test; an out-of-range index here is a test bug the panic \ + reports directly" + )] + fn the_lane_walk_is_bit_identical_to_the_scalar_walk() { + // A deterministic, sign-varied, non-smooth fill. + let plane = |seed: u32, len: usize| -> Vec<f32> { + (0..len) + .map(|i| { + let x = (i as u32).wrapping_mul(2_654_435_761).wrapping_add(seed); + (f64::from(x % 2003) / 1001.5 - 1.0) as f32 + }) + .collect() + }; + for len in [0usize, 1, 7, 8, 9, 64, 250] { + let planes: Vec<Vec<f32>> = (0..7u32).map(|s| plane(s * 97 + 13, len)).collect(); + let bands = MomentBands { + img1: &planes[0], + mu1: &planes[1], + s11: &planes[2], + img2: &planes[3], + mu2: &planes[4], + s22: &planes[5], + s12: &planes[6], + }; + let mut dispatched = MapSums::default(); + accumulate_band(&bands, &mut dispatched); + let mut scalar = MapSums::default(); + accumulate_scalar( + &[ + &planes[0], &planes[1], &planes[2], &planes[3], &planes[4], &planes[5], + &planes[6], + ], + &mut scalar, + ); + assert_eq!(dispatched, scalar, "length {len}"); + } + } + #[test] fn the_four_norm_of_a_constant_map_is_the_constant() { let sums = MapSums { diff --git a/JPXL/crates/jpxl-perceptual/src/ssimulacra2.rs b/JPXL/crates/jpxl-perceptual/src/ssimulacra2.rs index 1e6dbd9d..d589b2db 100644 --- a/JPXL/crates/jpxl-perceptual/src/ssimulacra2.rs +++ b/JPXL/crates/jpxl-perceptual/src/ssimulacra2.rs @@ -223,14 +223,15 @@ impl Ssimulacra2 { low_memory: bool, ) -> [ChannelTerms; 3] { let pixels = width * height; + // The blurs below write every sample of these planes before the pool + // reads them, so a reused plane needs no re-zeroing; resize only grows + // (zeroed) or truncates. for buf in [&mut self.mu2, &mut self.s22, &mut self.s12] { - buf.clear(); buf.resize(pixels, 0.0); } let recompute_reference = retention == ReferenceRetention::PlanesOnly; if recompute_reference { for buf in [&mut self.ref_mu, &mut self.ref_s11] { - buf.clear(); buf.resize(pixels, 0.0); } } diff --git a/JPXL/crates/jpxl-plan-render/Cargo.toml b/JPXL/crates/jpxl-plan-render/Cargo.toml index 56e2fcd7..9bed57df 100644 --- a/JPXL/crates/jpxl-plan-render/Cargo.toml +++ b/JPXL/crates/jpxl-plan-render/Cargo.toml @@ -11,7 +11,13 @@ license.workspace = true # over `jpxl-encode`'s plan types, and its parity against `jpxl-decode`, # `djxl` and `jxl-oxide` is proved by tests, not by sharing code. [dependencies] -jpxl-core = { path = "../jpxl-core", version = "0.3.0", default-features = false } +# `simd` is named explicitly: plan rendering sits on the quality path's +# hottest loops, and without it a consumer that builds this crate outside the +# full workspace (where `jpxl-encode` happens to unify the feature back on) +# would silently get the scalar DCT kernels. +jpxl-core = { path = "../jpxl-core", version = "0.3.0", default-features = false, features = [ + "simd", +] } jpxl-encode = { path = "../jpxl-encode", version = "0.3.0", default-features = false } [dev-dependencies] diff --git a/JPXL/crates/jpxl-plan-render/src/lib.rs b/JPXL/crates/jpxl-plan-render/src/lib.rs index b20c3116..04b9a5fc 100644 --- a/JPXL/crates/jpxl-plan-render/src/lib.rs +++ b/JPXL/crates/jpxl-plan-render/src/lib.rs @@ -193,6 +193,28 @@ impl RenderedFrame { .unwrap_or_else(|| srgb_to_linear(q as f32 / max)) } + /// The composed depth quantizer: the integer level that + /// [`Self::linear_sample`] forms from `linear_to_srgb(x)` — the sRGB + /// encode, the scale by `max`, the round, the finiteness guard and the + /// clamp, in that order. Kept callable on its own so threshold + /// construction and its verification bisect the actual function. + fn composed_level(x: f32, max: f32) -> usize { + let scaled = (linear_to_srgb(x) * max).round(); + if scaled.is_finite() { + // Clamped into [0, max] with max < 2^32 before the cast, so the + // narrowing is exact. + #[allow( + clippy::cast_possible_truncation, + clippy::cast_sign_loss, + reason = "the value is clamped to [0, max] first" + )] + let q = scaled.clamp(0.0, max) as u32; + usize::try_from(q).unwrap_or(0) + } else { + 0 + } + } + /// The planes quantized to `bits` per sample exactly as the decoder's /// integer output is: scaled, rounded and clamped. #[must_use] @@ -267,6 +289,160 @@ impl RenderedFrame { } self.planes } + + /// [`Self::into_linear_rgb_at_depth`] with the per-sample transfer-curve + /// lookup banded over `executor`'s workers. + /// + /// The lookup is an independent per-sample map, so each fixed row band is + /// converted on its own and the output is identical to the serial form at + /// any worker count. This is the large-frame scoring path, where the + /// 36-million-sample round trip is otherwise a serial tail on every probe. + #[must_use] + pub fn into_linear_rgb_at_depth_with( + mut self, + bits: u32, + executor: &EncodeExecutor, + ) -> [Vec<f32>; NUM_CHANNELS] { + let max = Self::full_scale(bits); + let lut = Self::linear_lut(bits, max); + let width = usize::try_from(self.width).unwrap_or(usize::MAX).max(1); + let bands = row_bands(&mut self.planes, width); + run_items(Some(executor), bands.len(), &|index| { + let Some((_, mut slices)) = bands.take(index) else { + return; + }; + for plane in slices.iter_mut() { + for value in plane.iter_mut() { + *value = Self::linear_sample(*value, max, &lut); + } + } + }); + self.planes + } +} + +/// Linear-domain thresholds of the composed depth quantizer for `levels` +/// integer levels at full scale `max`: entry `q - 1` is the smallest `f32` +/// whose [`RenderedFrame::composed_level`] is at least `q`. +/// +/// Each boundary is found by bisection over the non-negative `f32` bit +/// lattice against `composed_level` itself, so whatever the transfer curve's +/// branches and roundings do, the boundary is the actual function's; the +/// level function is monotone because every stage of the composition is. +/// Both sides of every boundary are asserted after the search. +fn quantize_thresholds(max: f32, levels: usize) -> Vec<f32> { + let mut thresholds = Vec::with_capacity(levels.saturating_sub(1)); + for q in 1..levels { + // 2.0 encodes above full scale, so every boundary sits below it, and + // non-negative `f32` bit patterns order exactly as their values. + let (mut lo, mut hi) = (0u32, 2.0f32.to_bits()); + while lo < hi { + let mid = lo + (hi - lo) / 2; + if RenderedFrame::composed_level(f32::from_bits(mid), max) >= q { + hi = mid; + } else { + lo = mid + 1; + } + } + let t = f32::from_bits(lo); + debug_assert!(RenderedFrame::composed_level(t, max) >= q); + debug_assert!(lo == 0 || RenderedFrame::composed_level(f32::from_bits(lo - 1), max) < q); + thresholds.push(t); + } + thresholds +} + +/// The composed level of one linear sample: how many thresholds it reaches. +/// Non-finite samples take level 0, exactly as the round trip's finiteness +/// guard sends them there; negative samples sit below every threshold. The +/// reference form — [`DepthClassifier`] reproduces it with a guide table. +#[inline] +fn level_of(x: f32, thresholds: &[f32]) -> usize { + if !x.is_finite() { + return 0; + } + thresholds.partition_point(|&t| t <= x) +} + +/// The composed depth quantizer as a lookup structure: the level thresholds, +/// the levels' linear values, and a guide over the non-negative `f32` bit +/// lattice pinning each bucket's samples to a narrow level range, so one +/// sample's classification is two loads and at most a short ordered scan. +#[derive(Debug)] +struct DepthClassifier { + bits: u32, + thresholds: Vec<f32>, + lut: Vec<f32>, + /// Per bucket, the levels of the bucket's smallest and largest values. + guide: Vec<(u32, u32)>, + /// `bit pattern >> shift` is the bucket of a value in `[0, 2.0)`. + shift: u32, +} + +/// The bit pattern of `2.0f32`, one past the last guided bucket: every +/// threshold lies strictly below 2.0 (full scale encodes below it), so any +/// larger sample takes the top level directly. +const DEPTH_GUIDE_END: u32 = 0x4000_0000; + +impl DepthClassifier { + fn new(bits: u32) -> Self { + let max = RenderedFrame::full_scale(bits); + let lut = RenderedFrame::linear_lut(bits, max); + let thresholds = quantize_thresholds(max, lut.len()); + let buckets = (lut.len() * 8).next_power_of_two().clamp(4096, 65_536); + let shift = DEPTH_GUIDE_END.trailing_zeros() - buckets.trailing_zeros(); + let guide = (0..buckets) + .map(|b| { + let lo_bits = u32::try_from(b).unwrap_or(0) << shift; + let hi_bits = lo_bits + ((1u32 << shift) - 1); + let lo = level_of(f32::from_bits(lo_bits), &thresholds); + let hi = level_of(f32::from_bits(hi_bits), &thresholds); + ( + u32::try_from(lo).unwrap_or(u32::MAX), + u32::try_from(hi).unwrap_or(u32::MAX), + ) + }) + .collect(); + Self { + bits, + thresholds, + lut, + guide, + shift, + } + } + + /// The level's linear value for one sample: `lut[level_of(x)]` for every + /// `f32`, non-finite and negative included, by way of the guide. + #[inline] + fn linear_at_depth(&self, x: f32) -> f32 { + if !x.is_finite() { + return self.lut.first().copied().unwrap_or(0.0); + } + let pattern = x.to_bits(); + if pattern >= 0x8000_0000 { + // Negative, -0.0 included: below every (positive) threshold. + return self.lut.first().copied().unwrap_or(0.0); + } + if pattern >= DEPTH_GUIDE_END { + // At least 2.0: above every threshold. + return self.lut.last().copied().unwrap_or(0.0); + } + let (lo, hi) = self + .guide + .get((pattern >> self.shift) as usize) + .copied() + .unwrap_or((0, 0)); + let mut level = lo as usize; + for &t in self.thresholds.get(lo as usize..hi as usize).unwrap_or(&[]) { + if t <= x { + level += 1; + } else { + break; + } + } + self.lut.get(level).copied().unwrap_or(0.0) + } } /// Wall time of one render's stages, in milliseconds. @@ -291,6 +467,10 @@ pub struct PlanRenderer { opsin: OpsinInverse, epf_params: EpfParams, gabor: GaborKernel, + /// The depth quantizer of the last [`Self::render_linear_at_depth_with`] + /// call, kept so a probe search bisects its thresholds once, not per + /// probe. + depth: Option<DepthClassifier>, } impl PlanRenderer { @@ -311,6 +491,7 @@ impl PlanRenderer { ), epf_params: EpfParams::default(), gabor: GaborKernel::defaults(), + depth: None, }) } @@ -360,6 +541,39 @@ impl PlanRenderer { .map(|(frame, _)| frame) } + /// [`Self::render_with`] fused with + /// [`RenderedFrame::into_linear_rgb_at_depth_with`]: the linear-sRGB + /// planes after the round trip through `bits`-deep integer samples, + /// without ever materialising the signalled sRGB encoding. Each sample's + /// quantized level is found in linear light through + /// [`quantize_thresholds`], so every output is bit-identical to rendering + /// and round-tripping in two steps — the equivalence across the whole + /// curve, both branch seams and non-finite samples is pinned by + /// `fused_depth_levels_match_the_srgb_round_trip`. + /// + /// This is the scoring path: a perceptual probe wants only these planes, + /// and the two-step path paid a full-frame `powf` encode per sample just + /// to quantize away its result. + /// + /// # Errors + /// + /// As [`Self::render`]. + pub fn render_linear_at_depth_with( + &mut self, + pixels: &ValidatedPixelPlan, + bits: u32, + executor: &EncodeExecutor, + ) -> Result<(u32, u32, [Vec<f32>; NUM_CHANNELS])> { + let classifier = match self.depth.take() { + Some(classifier) if classifier.bits == bits => classifier, + _ => DepthClassifier::new(bits), + }; + let rendered = self.render_inner(pixels, Some(executor), Some(&classifier)); + self.depth = Some(classifier); + let (frame, _) = rendered?; + Ok((frame.width, frame.height, frame.planes)) + } + /// [`Self::render_with`], also reporting where the time went. /// /// # Errors @@ -369,6 +583,20 @@ impl PlanRenderer { &mut self, pixels: &ValidatedPixelPlan, executor: Option<&EncodeExecutor>, + ) -> Result<(RenderedFrame, RenderTimings)> { + self.render_inner(pixels, executor, None) + } + + /// The full render pipeline. The colour stage ends in the signalled sRGB + /// encoding by default; with `linear_at_depth` it instead classifies each + /// linear sample against the thresholds and takes the level's linear + /// value from the table, and the returned frame's planes hold those + /// depth-quantized linear samples rather than the signalled encoding. + fn render_inner( + &mut self, + pixels: &ValidatedPixelPlan, + executor: Option<&EncodeExecutor>, + linear_at_depth: Option<&DepthClassifier>, ) -> Result<(RenderedFrame, RenderTimings)> { let mut timings = RenderTimings::default(); let millis = |start: std::time::Instant| { @@ -428,23 +656,51 @@ impl PlanRenderer { // I.5.2 + I.6 (LF): dequantize the three planes, then borrow // chroma from luma with the frame-wide LF factors. let lf = lf_planes(ir, &lf_mul, spatial.lf.extra_precision, (k_x_lf, k_b_lf))?; - let lf_at = |c: usize, bx: u32, by: u32| -> f32 { - if bx >= blocks.width || by >= blocks.height { - return 0.0; - } - let idx = usize::try_from(u64::from(by) * u64::from(blocks.width) + u64::from(bx)) - .unwrap_or(usize::MAX); - lf.get(c).and_then(|p| p.get(idx)).copied().unwrap_or(0.0) - }; - for (vb, coeffs) in group.blocks.iter().zip(ir.coefficients.iter()) { + // Warm the dequantization-matrix cache for every transform this + // group uses, so the per-varblock render below reads it through a + // shared immutable borrow and needs no `&mut self`. + for vb in group.blocks.iter() { + self.matrices_for(vb.transform)?; + } + let matrices_cache = &self.cache; + let epf_params = &self.epf_params; + + // I.5.3, I.6, I.8, I.9 and J.4.3 for one varblock, producing its + // placed samples and sigma writes without touching shared frame + // state. A varblock is a pure function of its own inputs, so this + // is byte-for-byte the serial render regardless of worker count. + let render_one = |i: usize| -> Result<RenderedVarblock> { + let vb = group + .blocks + .get(i) + .ok_or(RenderError::Unsupported("a varblock index past the group"))?; + let coeffs = ir + .coefficients + .get(i) + .ok_or(RenderError::Unsupported("a varblock with no coefficients"))?; + let lf_at = |c: usize, bx: u32, by: u32| -> f32 { + if bx >= blocks.width || by >= blocks.height { + return 0.0; + } + let idx = + usize::try_from(u64::from(by) * u64::from(blocks.width) + u64::from(bx)) + .unwrap_or(usize::MAX); + lf.get(c).and_then(|p| p.get(idx)).copied().unwrap_or(0.0) + }; + let transform = vb.transform; let (bx, by) = (vb.origin.bx(), vb.origin.by()); let hf_mul = vb.hf_mul.get(); let mul = hf_multiplier(global_scale, hf_mul); // I.5.3: dequantize all three channels. - let matrices = self.matrices_for(transform)?; + let matrices = matrices_cache + .get(transform.dequant_matrix_index()) + .and_then(Option::as_ref) + .ok_or(RenderError::Unsupported( + "a dequantization matrix that was not warmed", + ))?; let (rows, cols) = (transform.coeff_rows(), transform.coeff_cols()); let mut coeff: [CoeffMatrix; NUM_CHANNELS] = core::array::from_fn(|_| CoeffMatrix::zeros(rows, cols)); @@ -503,37 +759,21 @@ impl PlanRenderer { matrix.write_llf(&llf_from_lf(transform, &lf_rect)); } - // I.9: samples, placed at the varblock's frame position. + // I.9: samples, at the varblock's frame position (placed later). let x0 = rect.x0 + bx * 8; let y0 = rect.y0 + by * 8; - for (c, matrix) in coeff.iter().enumerate() { - let block = transform.samples_from_coefficients(matrix); - let Some(plane) = planes.get_mut(c) else { - continue; - }; - for row in 0..block.rows() { - let fy = y0.saturating_add(narrow(row)); - if fy >= height { - continue; - } - for col in 0..block.cols() { - let fx = x0.saturating_add(narrow(col)); - if fx >= width { - continue; - } - let idx = - usize::try_from(u64::from(fy) * u64::from(width) + u64::from(fx)) - .unwrap_or(usize::MAX); - if let Some(slot) = plane.get_mut(idx) { - *slot = block.at(col, row); - } - } - } - } + let samples: [SampleBlock; NUM_CHANNELS] = core::array::from_fn(|c| { + coeff.get(c).map_or_else( + || SampleBlock::zeros(transform.sample_rows(), transform.sample_cols()), + |matrix| transform.samples_from_coefficients(matrix), + ) + }); // J.4.3: sigma per 8x8 block of the varblock, from `mul` and - // the block's own `Sharpness`. + // the block's own `Sharpness`, as (frame-block index, value). let sharpness = group.sharpness.values(); + let mut sigma_writes: Vec<(usize, f32)> = + Vec::with_capacity(block_rows * block_cols); for dy in 0..block_rows { for dx in 0..block_cols { let (sbx, sby) = @@ -551,11 +791,100 @@ impl PlanRenderer { if fbx >= blocks_x || fby >= blocks_y { continue; } - if let Some(slot) = sigma.get_mut(fby * blocks_x + fbx) { - *slot = vardct_sigma(mul, s, &self.epf_params); + sigma_writes.push((fby * blocks_x + fbx, vardct_sigma(mul, s, epf_params))); + } + } + + Ok(RenderedVarblock { + x0, + y0, + samples, + sigma: sigma_writes, + }) + }; + + // Render varblocks in bounded, order-preserving chunks: each chunk's + // per-varblock compute (dequant, CfL, LLF, inverse transform) runs + // across the executor, then its samples are scattered band-parallel: + // every worker owns a disjoint [`BAND_ROWS`]-row band and writes + // only the varblock rows that fall inside it. Each sample belongs + // to exactly one varblock and exactly one band, so the placed + // pixels are identical to a serial render at any worker count; + // chunking keeps the transient per-chunk buffers small rather than + // holding one buffer per varblock at once. Sigma writes are a few + // hundredths of the sample volume and stay serial. + let n = group.blocks.len().min(ir.coefficients.len()); + let mut start = 0usize; + while start < n { + let end = (start + VARBLOCK_CHUNK).min(n); + let rendered = + render_varblock_chunk(executor, end - start, &|k| render_one(start + k))?; + // Bucket each varblock into the (at most two, since a varblock + // is at most 32 rows tall) bands its rows intersect, preserving + // chunk order within a bucket. + let bands = row_bands(&mut planes, dims.width); + let mut buckets: Vec<Vec<usize>> = vec![Vec::new(); bands.len()]; + for (k, rv) in rendered.iter().enumerate() { + let rows = rv.samples.iter().map(|b| b.rows()).max().unwrap_or(0); + if rows == 0 { + continue; + } + let top = usize::try_from(rv.y0).unwrap_or(usize::MAX); + let bottom = top.saturating_add(rows - 1); + let last_band = buckets.len().saturating_sub(1); + for band in + (top / BAND_ROWS).min(last_band)..=(bottom / BAND_ROWS).min(last_band) + { + if let Some(bucket) = buckets.get_mut(band) { + bucket.push(k); } } } + run_items(executor, bands.len(), &|index| { + let Some((row0, mut slices)) = bands.take(index) else { + return; + }; + let Some(bucket) = buckets.get(index) else { + return; + }; + let band_rows = slices.first().map_or(0, |s| s.len()) / dims.width.max(1); + for &k in bucket { + let Some(rv) = rendered.get(k) else { + continue; + }; + let x0 = usize::try_from(rv.x0).unwrap_or(usize::MAX); + let y0 = usize::try_from(rv.y0).unwrap_or(usize::MAX); + for (c, block) in rv.samples.iter().enumerate() { + let Some(plane) = slices.get_mut(c) else { + continue; + }; + for row in 0..block.rows() { + let fy = y0.saturating_add(row); + if fy < row0 || fy >= row0.saturating_add(band_rows) { + continue; + } + let base = (fy - row0).saturating_mul(dims.width); + for col in 0..block.cols() { + let fx = x0.saturating_add(col); + if fx >= dims.width { + continue; + } + if let Some(slot) = plane.get_mut(base.saturating_add(fx)) { + *slot = block.at(col, row); + } + } + } + } + } + }); + for rv in &rendered { + for &(idx, value) in &rv.sigma { + if let Some(slot) = sigma.get_mut(idx) { + *slot = value; + } + } + } + start = end; } } @@ -615,7 +944,8 @@ impl PlanRenderer { timings.epf_ms = millis(stage_start); - // Annex L: XYB -> linear sRGB -> the signalled sRGB encoding. + // Annex L: XYB -> linear sRGB -> the signalled sRGB encoding, or, + // for the scoring path, straight to the depth-quantized linear value. let stage_start = std::time::Instant::now(); { let bands = row_bands(&mut planes, dims.width); @@ -626,9 +956,20 @@ impl PlanRenderer { }; let [x, y, b] = &mut slices; opsin.convert_planes(x, y, b); - for plane in slices.iter_mut() { - for v in plane.iter_mut() { - *v = linear_to_srgb(*v); + match linear_at_depth { + None => { + for plane in slices.iter_mut() { + for v in plane.iter_mut() { + *v = linear_to_srgb(*v); + } + } + } + Some(classifier) => { + for plane in slices.iter_mut() { + for v in plane.iter_mut() { + *v = classifier.linear_at_depth(*v); + } + } } } }); @@ -731,6 +1072,47 @@ fn row_bands(planes: &mut [Vec<f32>; NUM_CHANNELS], width: usize) -> RowBands<'_ RowBands { items } } +/// One varblock's reconstruction result, produced off to the side so the +/// per-varblock compute can run across the executor and be scattered into the +/// frame afterwards. +struct RenderedVarblock { + /// Frame x of the varblock's top-left sample. + x0: u32, + /// Frame y of the varblock's top-left sample. + y0: u32, + /// The placed samples, one block per channel. + samples: [SampleBlock; NUM_CHANNELS], + /// J.4.3 sigma writes as `(frame-block index, value)`. + sigma: Vec<(usize, f32)>, +} + +/// Varblocks rendered per parallel chunk before their results are scattered. +/// +/// A chunk holds at most this many [`RenderedVarblock`] results at once, so the +/// transient sample storage stays a few megabytes rather than a whole frame's +/// worth, while still giving the executor enough work per chunk to keep every +/// worker busy on a large frame. +const VARBLOCK_CHUNK: usize = 2048; + +/// Renders `n` varblocks through `f`, across `executor` when present (results +/// in index order, so worker count cannot change them), serially otherwise. +fn render_varblock_chunk( + executor: Option<&EncodeExecutor>, + n: usize, + f: &(dyn Fn(usize) -> Result<RenderedVarblock> + Sync), +) -> Result<Vec<RenderedVarblock>> { + match executor { + Some(executor) => executor.map_ordered(n, f), + None => { + let mut out = Vec::with_capacity(n); + for i in 0..n { + out.push(f(i)?); + } + Ok(out) + } + } +} + /// Runs `items` independent closures on `executor` (serially without one). fn run_items(executor: Option<&EncodeExecutor>, items: usize, f: &(dyn Fn(usize) + Sync)) { match executor { @@ -749,3 +1131,63 @@ fn run_items(executor: Option<&EncodeExecutor>, items: usize, f: &(dyn Fn(usize) } } } + +#[cfg(test)] +mod tests { + use super::*; + + /// The fused threshold classifier must agree bit-for-bit with encoding to + /// sRGB and round-tripping through the depth quantizer, over a dense + /// sweep of the working range, the immediate bit-lattice neighbourhood of + /// every threshold (where any boundary error would live), both transfer + /// branch seams, and the non-finite specials. + #[test] + fn fused_depth_levels_match_the_srgb_round_trip() { + for bits in [1u32, 8, 12, 16] { + let max = RenderedFrame::full_scale(bits); + let classifier = DepthClassifier::new(bits); + let (lut, thresholds) = (&classifier.lut, &classifier.thresholds); + let check = |x: f32| { + let direct = RenderedFrame::linear_sample(linear_to_srgb(x), max, lut); + let searched = lut.get(level_of(x, thresholds)).copied().unwrap_or(0.0); + let guided = classifier.linear_at_depth(x); + assert_eq!( + direct.to_bits(), + searched.to_bits(), + "search: bits {bits}, x {x} ({:#010x})", + x.to_bits() + ); + assert_eq!( + direct.to_bits(), + guided.to_bits(), + "guide: bits {bits}, x {x} ({:#010x})", + x.to_bits() + ); + }; + for i in 0..120_000 { + check(-0.2 + i as f32 * 1.25e-5); + } + for &t in thresholds { + let b = t.to_bits(); + for d in 0..4u32 { + check(f32::from_bits(b.saturating_sub(d))); + check(f32::from_bits(b.saturating_add(d))); + } + } + for x in [ + f32::NAN, + f32::INFINITY, + f32::NEG_INFINITY, + -0.0, + 0.0, + 0.003_130_8, + 0.040_449_936, + 1.0, + 1.5, + f32::MIN_POSITIVE, + ] { + check(x); + } + } + } +} diff --git a/JPXL/crates/jpxl/src/lib.rs b/JPXL/crates/jpxl/src/lib.rs index 194259ce..2b0cf389 100644 --- a/JPXL/crates/jpxl/src/lib.rs +++ b/JPXL/crates/jpxl/src/lib.rs @@ -9,10 +9,15 @@ //! //! [`Encoder::with_ssimulacra2_score`] is the normal way to ask for lossy //! output: name the minimum perceptual quality and let the encoder find the -//! bytes. [`Encoder::with_target_bpp`] / [`Encoder::with_target_bytes`] (an -//! exact size) and [`Encoder::with_global_scale`] (a pinned quantizer) are -//! expert modes. Exactly one lossy target may be set; call -//! [`Encoder::lossless`] to reset before choosing another. +//! bytes. The score is a hard floor: a successful perceptual encode has had +//! its actual reconstruction canonically scored at or above the request. +//! When the bounded controller cannot verify such a stream, the encode fails +//! with [`Error::TargetNotMet`] unless [`Encoder::with_quality_fallback`] +//! opted into an explicit fallback. [`Encoder::with_target_bpp`] / +//! [`Encoder::with_target_bytes`] (an exact size) and +//! [`Encoder::with_global_scale`] (a pinned quantizer) are expert modes. +//! Exactly one lossy target may be set; call [`Encoder::lossless`] to reset +//! before choosing another. //! //! # Lossless RGB //! @@ -40,6 +45,7 @@ use std::fmt; pub use jpxl_core::limits::Limits; pub use jpxl_decode::decode::FloatPlane; pub use jpxl_decode::{DecodedImage, Plane}; +pub use jpxl_encode::ColourSpace; pub use jpxl_encode_policy::RateStatus; pub use jpxl_encode_policy::request::MetricVersion; @@ -68,6 +74,14 @@ pub enum Error { /// honoured in a later release, so callers can special-case it rather than /// treat it as their own bug. Unsupported(&'static str), + /// A perceptual encode whose bounded controller stopped without any + /// stream canonically verified at the requested minimum score, while the + /// encoder was left at [`QualityFallback::Refuse`] (the default). + /// + /// Nothing was emitted. The carried [`QualityMiss`] says how close the + /// search got and why it stopped; [`Encoder::with_quality_fallback`] + /// selects what to emit instead of failing. + TargetNotMet(QualityMiss), } impl fmt::Display for Error { @@ -77,6 +91,19 @@ impl fmt::Display for Error { Self::Encode(error) => write!(f, "encode failed: {error}"), Self::Policy(error) => write!(f, "encode policy failed: {error}"), Self::InvalidOption(message) | Self::Unsupported(message) => f.write_str(message), + Self::TargetNotMet(miss) => { + let why = match miss.kind { + QualityMissKind::LadderSaturated => "even the finest quantizer", + QualityMissKind::WorkBudgetExhausted => "the probe budget's best candidate", + }; + write!( + f, + "quality target not met: {why} verified {:.4}, below the requested \ + minimum {:.4} ({}); nothing was written — choose a quality fallback \ + to emit anyway", + miss.best_score, miss.requested_score, miss.metric_version, + ) + } } } } @@ -87,7 +114,7 @@ impl std::error::Error for Error { Self::Decode(error) => Some(error), Self::Encode(error) => Some(error), Self::Policy(error) => Some(error), - Self::InvalidOption(_) | Self::Unsupported(_) => None, + Self::InvalidOption(_) | Self::Unsupported(_) | Self::TargetNotMet(_) => None, } } } @@ -203,6 +230,64 @@ pub struct RateSummary { pub full_prices: u32, } +/// What a perceptual encode emits when the bounded controller stops without +/// any stream canonically verified at the requested minimum score. +/// +/// The score is a hard floor: an under-target stream is never an ordinary +/// success. [`Self::Refuse`] (the default) turns such an encode into +/// [`Error::TargetNotMet`]; the two alternatives are explicit contracts a +/// caller opts into with [`Encoder::with_quality_fallback`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum QualityFallback { + /// Fail with [`Error::TargetNotMet`] and emit nothing. The default. + #[default] + Refuse, + /// Emit a mathematically lossless stream instead, at the encoder's + /// lossless effort, reported as [`PerceptualStatus::FallbackLossless`]. + /// The floor holds (lossless trivially meets any score) at whatever byte + /// cost lossless carries — often far more than the lossy request implied. + Lossless, + /// Emit the finest canonically verified under-target stream, reported by + /// [`PerceptualStatus::SaturatedTop`] or + /// [`PerceptualStatus::UnderTargetWorkCap`] with its true + /// `achieved_score`. This knowingly weakens the floor for this encode; + /// the report says so explicitly. + BestEffort, +} + +/// Why a refused perceptual encode could not verify the requested score. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum QualityMissKind { + /// Even the quantizer ladder's finest rung scored below the request: the + /// lossy path cannot reach this score on this image. + LadderSaturated, + /// The effort's bounded probe budget ran out before any candidate met + /// the request; a finer candidate may exist but was never verified. + WorkBudgetExhausted, +} + +/// What a refused perceptual encode verified before it stopped, carried by +/// [`Error::TargetNotMet`]. +#[derive(Debug, Clone, PartialEq)] +#[non_exhaustive] +pub struct QualityMiss { + /// Why the search stopped short. + pub kind: QualityMissKind, + /// The minimum score the caller asked for. + pub requested_score: f64, + /// The canonical score of the finest verified candidate — the closest + /// the bounded search got to the request. + pub best_score: f64, + /// The metric definition the scores are on. + pub metric_version: MetricVersion, + /// How many candidate streams were scored. + pub probes: u32, + /// How many exact writer prices were paid. + pub prices: u32, + /// The controller's `jpxl.quality-trace/2` record. + pub trace_json: Option<String>, +} + /// Why a perceptual encode stopped where it did: the quality controller's /// terminal state, or one of the two routes that bypass it. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -215,15 +300,23 @@ pub enum PerceptualStatus { MetWorkCap, /// The coarsest quantizer already exceeds the score (a floor). SaturatedFloor, - /// The finest quantizer still misses the score (a ceiling). + /// The finest quantizer still misses the score (a ceiling). Reported only + /// under [`QualityFallback::BestEffort`]; the default refuses instead + /// with [`Error::TargetNotMet`]. SaturatedTop, /// The bounded controller ran out of probes before any candidate met the /// score; the emitted stream's `achieved_score` is below the request. + /// Reported only under [`QualityFallback::BestEffort`]; the default + /// refuses instead with [`Error::TargetNotMet`]. UnderTargetWorkCap, /// A bounded fresh-structure rescue supplied the selected stream. RescuedFreshStructure, /// A score of 100 was satisfied by the mathematically lossless path. RoutedToLossless, + /// The bounded lossy controller could not verify the requested score and + /// [`QualityFallback::Lossless`] emitted a mathematically lossless stream + /// in its place. `probes`/`prices` count the failed lossy search. + FallbackLossless, /// The frame is too small for the perceptual path to apply. UnsupportedTooSmall, } @@ -247,7 +340,7 @@ pub struct PerceptualOutcome { pub prices: u32, /// Whether the quantizer ladder ran out of rungs. pub saturated: bool, - /// The controller's `jpxl.quality-trace/1` record, when a search ran. + /// The controller's `jpxl.quality-trace/2` record, when a search ran. pub trace_json: Option<String>, } @@ -285,7 +378,7 @@ enum Mode { /// contract), [`with_target_bpp`](Self::with_target_bpp) / /// [`with_target_bytes`](Self::with_target_bytes), or /// [`with_global_scale`](Self::with_global_scale). -#[derive(Debug, Clone, Copy, PartialEq)] +#[derive(Debug, Clone, PartialEq)] pub struct Encoder { mode: Mode, lossless_effort: jpxl_encode::Effort, @@ -293,6 +386,9 @@ pub struct Encoder { resources: jpxl_encode::EncodeResources, container: bool, jxlp_fragment_size: Option<usize>, + quality_fallback: QualityFallback, + colour_space: ColourSpace, + exif: Option<Vec<u8>>, } impl Default for Encoder { @@ -304,6 +400,9 @@ impl Default for Encoder { resources: jpxl_encode::EncodeResources::default(), container: false, jxlp_fragment_size: None, + quality_fallback: QualityFallback::Refuse, + colour_space: ColourSpace::Srgb, + exif: None, } } } @@ -377,7 +476,11 @@ impl Encoder { /// contract). /// /// A score of 100 means mathematically lossless. The score must be finite - /// and in `0.0..=100.0`. + /// and in `0.0..=100.0`. The score is a hard floor: when the bounded + /// controller cannot canonically verify a stream at or above it, the + /// encode fails with [`Error::TargetNotMet`] unless + /// [`with_quality_fallback`](Self::with_quality_fallback) chose an + /// explicit fallback. pub fn with_ssimulacra2_score(self, score: f64) -> Result<Self> { let target = PerceptualTarget::new(PerceptualMetric::Ssimulacra2, score) .map_err(|_| Error::InvalidOption("ssimulacra2 score must be finite and in 0..=100"))?; @@ -429,6 +532,18 @@ impl Encoder { self } + /// Choose what a perceptual encode emits when its bounded controller + /// cannot verify a stream at the requested minimum score. + /// + /// The default is [`QualityFallback::Refuse`]: such an encode fails with + /// [`Error::TargetNotMet`] rather than returning under-target bytes as a + /// success. + #[must_use] + pub const fn with_quality_fallback(mut self, fallback: QualityFallback) -> Self { + self.quality_fallback = fallback; + self + } + /// Choose the lossless Modular search effort, from 1 (fastest) to 9 /// (densest). pub fn with_lossless_effort(mut self, effort: u8) -> Result<Self> { @@ -469,6 +584,44 @@ impl Encoder { self } + /// Declare the colour space of the samples handed to the encoder + /// (default: sRGB), signalled declaratively in the image header. + /// + /// The lossless Modular path stores samples untouched, so this changes + /// only how a colour-managed viewer interprets them. The lossy VarDCT + /// path is defined on sRGB input — its XYB transform and its perceptual + /// metric assume it — so a lossy encode of a non-sRGB colour space fails + /// with [`Error::Unsupported`] rather than mis-tagging the pixels. + #[must_use] + pub fn with_colour_space(mut self, colour_space: ColourSpace) -> Self { + self.colour_space = colour_space; + self + } + + /// Attach an Exif metadata block, carried in a Part 2 `Exif` box. + /// + /// `exif` must be the raw Exif/TIFF payload as JEITA CP-3451E defines it, + /// beginning with the TIFF byte-order header (`II*\0` or `MM\0*`) — the + /// same bytes a camera stores after the `Exif\0\0` marker of a JPEG + /// `APP1` segment, without that marker. Implies + /// [`with_container`](Self::with_container): metadata boxes have nowhere + /// to live outside a container. + /// + /// Per 18181-2 9.5 the codestream's own fields (dimensions, orientation) + /// take precedence over Exif equivalents at decode time. + pub fn with_exif(mut self, exif: Vec<u8>) -> Result<Self> { + if !matches!( + exif.first_chunk(), + Some([0x49, 0x49, 0x2A, 0x00] | [0x4D, 0x4D, 0x00, 0x2A]) + ) { + return Err(Error::InvalidOption( + "the Exif payload must begin with a TIFF byte-order header (II*\\0 or MM\\0*)", + )); + } + self.exif = Some(exif); + Ok(self) + } + /// Encode interleaved 8-bit sRGB samples. pub fn encode_rgb8(&self, width: u32, height: u32, rgb: &[u8]) -> Result<Vec<u8>> { Ok(self.encode_rgb8_reported(width, height, rgb)?.0) @@ -481,10 +634,10 @@ impl Encoder { height: u32, rgb: &[u8], ) -> Result<(Vec<u8>, EncodeReport)> { + self.require_srgb_for_lossy()?; match self.mode { Mode::Lossless => { - let samples: Vec<u16> = rgb.iter().map(|&sample| u16::from(sample)).collect(); - let bytes = self.encode_lossless(width, height, 3, 8, &samples)?; + let bytes = self.encode_lossless(width, height, 3, 8, rgb)?; Ok((bytes, EncodeReport::Lossless)) } Mode::Lossy(LossyTarget::Rate(target)) => { @@ -494,23 +647,20 @@ impl Encoder { width, height, rgb, &request, target, )?; let report = EncodeReport::Rate(rate_summary(&outcome)); - let bytes = self.wrap_lossy(outcome.codestream, 8); + let bytes = self.finish(outcome.codestream, 8); Ok((bytes, report)) } Mode::Lossy(LossyTarget::Perceptual(target)) => self.perceptual_encode( target, PerceptualSource::Rgb8 { width, height, rgb }, - || { - let samples: Vec<u16> = rgb.iter().map(|&sample| u16::from(sample)).collect(); - self.encode_lossless(width, height, 3, 8, &samples) - }, + || self.encode_lossless(width, height, 3, 8, rgb), ), Mode::Lossy(LossyTarget::FixedQuantizer(fixed)) => { let mut request = jpxl_encode_policy::EncodeRequest::for_fixed_quantizer(fixed); request.resources = self.resources; request.bits_per_sample = 8; let bytes = jpxl_encode_policy::encode_srgb8_vardct(width, height, rgb, &request)?; - let wrapped = self.wrap_lossy(bytes, 8); + let wrapped = self.finish(bytes, 8); let count = u64::try_from(wrapped.len()).unwrap_or(u64::MAX); Ok((wrapped, EncodeReport::FixedQuantizer { bytes: count })) } @@ -538,6 +688,7 @@ impl Encoder { bits_per_sample: u32, rgb: &[u16], ) -> Result<(Vec<u8>, EncodeReport)> { + self.require_srgb_for_lossy()?; match self.mode { Mode::Lossless => { let bytes = self.encode_lossless(width, height, 3, bits_per_sample, rgb)?; @@ -554,7 +705,7 @@ impl Encoder { target, )?; let report = EncodeReport::Rate(rate_summary(&outcome)); - let bytes = self.wrap_lossy(outcome.codestream, bits_per_sample); + let bytes = self.finish(outcome.codestream, bits_per_sample); Ok((bytes, report)) } Mode::Lossy(LossyTarget::Perceptual(target)) => self.perceptual_encode( @@ -577,7 +728,7 @@ impl Encoder { bits_per_sample, &request, )?; - let wrapped = self.wrap_lossy(bytes, bits_per_sample); + let wrapped = self.finish(bytes, bits_per_sample); let count = u64::try_from(wrapped.len()).unwrap_or(u64::MAX); Ok((wrapped, EncodeReport::FixedQuantizer { bytes: count })) } @@ -589,8 +740,12 @@ impl Encoder { /// The current VarDCT policy is RGB-only, so a lossy encoder returns a /// clear error instead of silently expanding greyscale to RGB. pub fn encode_gray8(&self, width: u32, height: u32, gray: &[u8]) -> Result<Vec<u8>> { - let samples: Vec<u16> = gray.iter().map(|&sample| u16::from(sample)).collect(); - self.encode_gray16(width, height, 8, &samples) + if matches!(self.mode, Mode::Lossy(_)) { + return Err(Error::InvalidOption( + "lossy VarDCT currently requires RGB input; use lossless mode for greyscale", + )); + } + self.encode_lossless(width, height, 1, 8, gray) } /// Encode high-precision greyscale samples losslessly. @@ -621,8 +776,15 @@ impl Encoder { where F: FnOnce() -> Result<Vec<u8>>, { - let routed = |status: PerceptualStatus| -> Result<(Vec<u8>, EncodeReport)> { - let bytes = lossless()?; + // A lossless stream trivially meets any score, so every lossless + // route reports `achieved_score` 100; `probes`/`prices` are those of + // whatever lossy search ran first (zero on the two early routes). + let routed = |bytes: Vec<u8>, + status: PerceptualStatus, + probes: u32, + prices: u32, + trace_json: Option<String>| + -> (Vec<u8>, EncodeReport) { let exact = u64::try_from(bytes.len()).unwrap_or(u64::MAX); let outcome = PerceptualOutcome { requested_score: target.minimum_score, @@ -630,21 +792,33 @@ impl Encoder { exact_bytes: exact, metric_version: target.metric.version(), status, - probes: 0, - prices: 0, + probes, + prices, saturated: false, - trace_json: None, + trace_json, }; - Ok((bytes, EncodeReport::Perceptual(outcome))) + (bytes, EncodeReport::Perceptual(outcome)) }; // `minimum_score` is validated into `0.0..=100.0`, so `>= 100.0` is the // exact-lossless request. if target.minimum_score >= 100.0 { - return routed(PerceptualStatus::RoutedToLossless); + return Ok(routed( + lossless()?, + PerceptualStatus::RoutedToLossless, + 0, + 0, + None, + )); } let (width, height, bits) = source.dimensions(); if width < jpxl_perceptual::MIN_DIMENSION || height < jpxl_perceptual::MIN_DIMENSION { - return routed(PerceptualStatus::UnsupportedTooSmall); + return Ok(routed( + lossless()?, + PerceptualStatus::UnsupportedTooSmall, + 0, + 0, + None, + )); } let mut request = jpxl_encode_policy::EncodeRequest::for_quality(self.effort.into()); @@ -705,7 +879,41 @@ impl Encoder { Effort::Quality => "quality", }; let trace_json = Some(outcome.trace_json(effort_name)); - let bytes = self.wrap_lossy(outcome.codestream, bits); + + // The hard floor: a stream the controller verified below the request + // is never an ordinary success. What happens instead is the encoder's + // [`QualityFallback`]; only `BestEffort` proceeds to emit it. + if matches!( + outcome.status, + jpxl_encode_policy::QualityStatus::SaturatedTop + | jpxl_encode_policy::QualityStatus::UnderTargetWorkCap + ) { + match self.quality_fallback { + QualityFallback::Refuse => { + return Err(Error::TargetNotMet(QualityMiss { + kind: quality_miss_kind(outcome.status), + requested_score: target.minimum_score, + best_score: outcome.achieved_score, + metric_version: target.metric.version(), + probes: outcome.stats.pixel_probes, + prices: outcome.stats.exact_prices, + trace_json, + })); + } + QualityFallback::Lossless => { + return Ok(routed( + lossless()?, + PerceptualStatus::FallbackLossless, + outcome.stats.pixel_probes, + outcome.stats.exact_prices, + trace_json, + )); + } + QualityFallback::BestEffort => {} + } + } + + let bytes = self.finish(outcome.codestream, bits); let report = PerceptualOutcome { requested_score: target.minimum_score, achieved_score: Some(outcome.achieved_score), @@ -736,13 +944,13 @@ impl Encoder { Ok((bytes, EncodeReport::Perceptual(report))) } - fn encode_lossless( + fn encode_lossless<S: Copy + Into<i32>>( &self, width: u32, height: u32, channels: usize, bits_per_sample: u32, - samples: &[u16], + samples: &[S], ) -> Result<Vec<u8>> { let image = jpxl_encode::Image::from_interleaved( width, @@ -751,14 +959,18 @@ impl Encoder { bits_per_sample, samples, )?; + // Ask for the naked codestream and wrap in `finish`, so the + // container logic (including the Exif box) lives in one place for + // the lossless and lossy paths alike. let options = jpxl_encode::EncodeOptions { - container: self.container, - jxlp_fragment_size: self.jxlp_fragment_size, + container: false, + jxlp_fragment_size: None, resources: self.resources, effort: self.lossless_effort, + colour_space: self.colour_space, ..jpxl_encode::EncodeOptions::default() }; - Ok(jpxl_encode::encode(&image, &options)?) + Ok(self.finish(jpxl_encode::encode(&image, &options)?, bits_per_sample)) } fn lossy_request( @@ -771,8 +983,11 @@ impl Encoder { request } - fn wrap_lossy(&self, codestream: Vec<u8>, bits_per_sample: u32) -> Vec<u8> { - if !self.container && self.jxlp_fragment_size.is_none() { + /// Wraps a finished codestream per the encoder's container options: a + /// container when asked for (or implied by a `jxlp` fragment size or an + /// Exif payload), with the Exif box appended after the codestream boxes. + fn finish(&self, codestream: Vec<u8>, bits_per_sample: u32) -> Vec<u8> { + if !self.container && self.jxlp_fragment_size.is_none() && self.exif.is_none() { return codestream; } let level = if bits_per_sample > 8 { @@ -780,10 +995,39 @@ impl Encoder { } else { jpxl_encode::container::DEFAULT_LEVEL }; - match self.jxlp_fragment_size { + let mut file = match self.jxlp_fragment_size { Some(size) => jpxl_encode::container::wrap_fragmented(&codestream, level, size), None => jpxl_encode::container::wrap(&codestream, level), + }; + if let Some(ref exif) = self.exif { + jpxl_encode::container::append_exif(&mut file, exif); } + file + } + + /// The lossy VarDCT pipeline is defined on sRGB input: its XYB transform + /// and its perceptual metric assume it, so anything else must be encoded + /// losslessly rather than mis-tagged. + const fn require_srgb_for_lossy(&self) -> Result<()> { + if matches!(self.mode, Mode::Lossy(_)) && !matches!(self.colour_space, ColourSpace::Srgb) { + return Err(Error::Unsupported( + "lossy VarDCT is defined on sRGB input; encode other colour spaces losslessly", + )); + } + Ok(()) + } +} + +/// Which [`QualityMissKind`] an under-target terminal status names. +/// +/// Only [`QualityStatus::SaturatedTop`](jpxl_encode_policy::QualityStatus) and +/// [`QualityStatus::UnderTargetWorkCap`](jpxl_encode_policy::QualityStatus) +/// are misses; every other status carries a verified at-or-above-target +/// stream and never reaches this mapping. +const fn quality_miss_kind(status: jpxl_encode_policy::QualityStatus) -> QualityMissKind { + match status { + jpxl_encode_policy::QualityStatus::SaturatedTop => QualityMissKind::LadderSaturated, + _ => QualityMissKind::WorkBudgetExhausted, } } @@ -826,6 +1070,78 @@ mod tests { assert_eq!(decoded.interleaved_colour(), gray); } + /// The archive-pipeline claim: an Exif payload attached at the facade + /// comes back byte-identical from the decoder's box walk, on both the + /// lossless and the lossy path, and implies the container. + #[test] + fn an_exif_payload_survives_encode_to_the_decoders_box_walk() { + let mut exif = vec![0x4D, 0x4D, 0x00, 0x2A]; + exif.extend((0..32u32).map(|i| (i % 5) as u8)); + + let (width, height) = (64u32, 64u32); + let rgb = gradient_rgb8(width, height); + for encoded in [ + Encoder::new() + .with_exif(exif.clone()) + .expect("valid payload") + .encode_rgb8(width, height, &rgb) + .expect("lossless encode"), + Encoder::new() + .with_target_bytes(2_048) + .expect("target") + .with_effort(Effort::Fast) + .with_threads(1) + .expect("threads") + .with_exif(exif.clone()) + .expect("valid payload") + .encode_rgb8(width, height, &rgb) + .expect("lossy encode"), + ] { + assert!(jpxl_decode::container::is_container(&encoded)); + let mut guard = + jpxl_core::limits::AllocGuard::new(&jpxl_core::limits::Limits::relaxed()); + let tree = jpxl_decode::container::BoxTree::parse(&encoded, &mut guard).expect("boxes"); + tree.validate().expect("conforming"); + let boxes = tree.exif().expect("well-formed Exif"); + assert_eq!(boxes.len(), 1); + let first = boxes.first().expect("length asserted above"); + assert_eq!(first.payload, &exif[..]); + decode(&encoded).expect("still decodes"); + } + } + + #[test] + fn a_payload_without_a_tiff_header_is_rejected() { + assert!(Encoder::new().with_exif(vec![]).is_err()); + assert!( + Encoder::new() + .with_exif(vec![0xFF, 0xD8, 0xFF, 0xE1]) + .is_err() + ); + } + + /// A non-sRGB colour space is signalled on the lossless path and refused + /// (not mis-tagged) on the lossy path. + #[test] + fn colour_space_signalling_and_the_lossy_guard() { + let rgb: Vec<u16> = (0..4 * 4 * 3u16).map(|i| i * 512).collect(); + let encoded = Encoder::new() + .with_colour_space(ColourSpace::Rec2020) + .encode_rgb16(4, 4, 16, &rgb) + .expect("lossless rec2020"); + let decoded = decode(&encoded).expect("decode"); + assert_eq!(decoded.interleaved_colour(), rgb); + + let lossy = Encoder::new() + .with_target_bytes(2_048) + .expect("target") + .with_colour_space(ColourSpace::Rec2020); + assert!(matches!( + lossy.encode_rgb16(4, 4, 16, &rgb), + Err(Error::Unsupported(_)) + )); + } + #[test] fn builder_rejects_nonsensical_targets() { assert!(Encoder::new().with_target_bpp(0.0).is_err()); @@ -886,6 +1202,7 @@ mod tests { for wrapped in [ encoder + .clone() .with_container(true) .encode_rgb8(width, height, &rgb) .expect("container encode"), diff --git a/JPXL/crates/jpxl/tests/quality_encode.rs b/JPXL/crates/jpxl/tests/quality_encode.rs index bc68a219..e2481cee 100644 --- a/JPXL/crates/jpxl/tests/quality_encode.rs +++ b/JPXL/crates/jpxl/tests/quality_encode.rs @@ -13,7 +13,10 @@ clippy::cast_sign_loss )] -use jpxl::{Decoder, Effort, EncodeReport, Encoder, PerceptualStatus}; +use jpxl::{ + Decoder, Effort, EncodeReport, Encoder, Error, PerceptualStatus, QualityFallback, + QualityMissKind, +}; use jpxl_perceptual::{LinearRgbView, score_pair}; struct Rng(u32); @@ -147,7 +150,7 @@ fn every_effort_meets_every_target_and_reports_the_score_the_file_has() { outcome .trace_json .as_deref() - .is_some_and(|t| t.starts_with("{\"schema\":\"jpxl.quality-trace/1\"")) + .is_some_and(|t| t.starts_with("{\"schema\":\"jpxl.quality-trace/2\"")) ); let independent = rescore(w, h, &rgb, &bytes); assert!( @@ -231,6 +234,128 @@ fn tiny_frames_and_a_perfect_score_route_to_lossless() { ); } +/// The hard floor at the facade: a target the bounded Fast search cannot +/// verify is refused by default with [`Error::TargetNotMet`] and emits +/// nothing; the two fallbacks are explicit and say what they did. The 64x64 +/// gradient tops out near 99.13 on this metric, so 99.5 is a deterministic +/// miss. +#[test] +fn an_unmet_target_refuses_by_default_and_falls_back_only_on_request() { + let (w, h) = (64u32, 64u32); + let rgb = synthetic(w, h, 11); + let encoder = || { + Encoder::new() + .with_ssimulacra2_score(99.5) + .unwrap() + .with_effort(Effort::Fast) + .with_threads(1) + .unwrap() + }; + + // Default: refuse, with the miss fully described. + let refused = encoder().encode_rgb8_reported(w, h, &rgb); + let miss = match refused { + Err(Error::TargetNotMet(miss)) => miss, + other => panic!("expected TargetNotMet, got {other:?}"), + }; + assert_eq!(miss.kind, QualityMissKind::LadderSaturated); + assert!((miss.requested_score - 99.5).abs() < f64::EPSILON); + assert!(miss.best_score < 99.5, "{}", miss.best_score); + assert!(miss.probes >= 1 && miss.probes <= 4, "{}", miss.probes); + assert!( + miss.trace_json + .as_deref() + .is_some_and(|t| t.starts_with("{\"schema\":\"jpxl.quality-trace/2\"")) + ); + + // Lossless fallback: the floor holds (achieved 100), the status says how, + // and the bytes are exactly the lossless encoder's. + let (bytes, report) = encoder() + .with_quality_fallback(QualityFallback::Lossless) + .encode_rgb8_reported(w, h, &rgb) + .unwrap(); + let outcome = perceptual(&report); + assert_eq!(outcome.status, PerceptualStatus::FallbackLossless); + assert_eq!(outcome.achieved_score, Some(100.0)); + assert_eq!(outcome.probes, miss.probes); + assert_eq!( + bytes, + Encoder::new().lossless().encode_rgb8(w, h, &rgb).unwrap() + ); + + // Best effort: the under-target stream is emitted, but only with its + // true score and an explicit saturation status. + let (bytes, report) = encoder() + .with_quality_fallback(QualityFallback::BestEffort) + .encode_rgb8_reported(w, h, &rgb) + .unwrap(); + let outcome = perceptual(&report); + assert_eq!(outcome.status, PerceptualStatus::SaturatedTop); + assert!(outcome.saturated); + let achieved = outcome.achieved_score.expect("a measured score"); + assert!((achieved - miss.best_score).abs() < f64::EPSILON); + let independent = rescore(w, h, &rgb, &bytes); + assert!((independent - achieved).abs() < 1e-6); +} + +/// PR 4 wall measurement: the *in-search* cost of the transform summary, +/// which shares the cover's forward cache (unlike the standalone +/// `jpxl features --transform-summary` path, which pays for a throwaway +/// DCT8 pass). Ignored by default — run explicitly on a quiet machine: +/// `cargo test -p jpxl --profile fast-debug --test quality_encode +/// transform_shadow_wall -- --ignored --nocapture`. +#[test] +#[ignore = "timing measurement; run on a quiet machine"] +fn transform_shadow_wall_delta() { + use jpxl_encode_policy::request::{PerceptualMetric, PerceptualTarget}; + use jpxl_encode_policy::{ + AnalysisAtlas, EncodeRequest, PreparedFrame, QualityBudget, RateSearchPreset, + search_frame_perceptual_with_budget, + }; + use jpxl_perceptual::PlanRenderEvaluator; + + let (w, h) = (2400u32, 1800u32); + let rgb = synthetic(w, h, 21); + let request = EncodeRequest::for_quality(RateSearchPreset::Balanced); + let executor = request.resources.executor(); + let target = PerceptualTarget::new(PerceptualMetric::Ssimulacra2, 85.0).unwrap(); + let base = QualityBudget::for_preset(RateSearchPreset::Balanced); + + let measure = |transform_shadow: bool| { + let mut best = f64::INFINITY; + for _ in 0..3 { + let frame = PreparedFrame::from_srgb8_with(w, h, &rgb, Some(&executor)).unwrap(); + let atlas = AnalysisAtlas::analyze(&frame); + let mut ev = PlanRenderEvaluator::from_srgb8(w, h, &rgb, &executor).unwrap(); + let start = std::time::Instant::now(); + let outcome = search_frame_perceptual_with_budget( + &frame, + &atlas, + &request, + target, + &mut ev, + &executor, + QualityBudget { + transform_shadow, + ..base + }, + ) + .unwrap(); + assert!(outcome.achieved_score >= 85.0); + best = best.min(start.elapsed().as_secs_f64()); + } + best + }; + + let off = measure(false); + let on = measure(true); + eprintln!( + "transform_shadow off: {off:.3}s, on: {on:.3}s, delta {:+.3}s ({:+.1}% of the search)", + on - off, + (on - off) / off * 100.0 + ); +} + /// PR 5: the Balanced perceptual policy bank meets the target and is never /// larger than the baseline-only (fixed-policy) result at the same score. /// diff --git a/JPXL/docs/CHANGELOG.md b/JPXL/docs/CHANGELOG.md index 70eebea5..5e5ef6f9 100644 --- a/JPXL/docs/CHANGELOG.md +++ b/JPXL/docs/CHANGELOG.md @@ -13,6 +13,51 @@ Format: [Keep a Changelog](https://keepachangelog.com/), semantic versioning. ## Unreleased +- 2026-08-25 — Large-frame `--quality` probes are faster: the varblock + reconstruction and the transfer-curve linearization — previously serial + on every probe — now run banded across the worker pool with byte-identical + output at any thread count. On the locked 12 MP anchor at 8 threads the + quality-mode wall drops 21 % (release), with peak memory and every emitted + stream unchanged. +- 2026-08-25 — The one-shot crossing predictor is now the default `--quality` + controller seed: a generated transparent model (`qpv2-st-1`, source + + DCT8-summary features, trained on 130 image families) picks the first + fresh plan — its risk-adjusted candidate when confident, its median when + uncertain — and the bounded navigator continues under the same probe/price + caps with canonical verification before every emission, so the hard score + floor is unchanged. Promotion A/B on 441 never-tuned holdout cells: zero + floor violations on either arm, byte geomean 0.997 (locked 13-image + holdout) and 0.990 (50 painting families), reconstructions −18% and wall + −16% on the paintings, 12 MP anchor wall −21%. Known bounded regressions: + sub-kilobyte saturated fixtures (worst +131 bytes) and target-95 cells + (≤1.12× with higher achieved scores). Build `jpxl-encode-policy` without + default features for the legacy table-seeded controller. +- 2026-08-24 — Quality trace schema is now `jpxl.quality-trace/2`: adds a + whole-search `work` block (pixel plans, reconstructions, metric + evaluations, entropy trainings, emissions — policy trials and the reducer + included), a shadow `prediction` block (null until a generated crossing + model is present), and a `transform_features` block (null unless the + research budget asked for it). `/1` traces stay readable by the harness. + New calibration tooling: `jpxl quality-ladder` (fresh production-policy + pixel plans at pinned effective scales, canonically scored, optionally + exact-priced, as JSONL), `jpxl features --transform-summary` (DCT8-derived + transform features), and `tools/quality_oracle_labels.py` / + `tools/quality_predictor_v2.py` (oracle-label sweeps over the full + effective ladder and the trained crossing predictor). The quality-corpus + manifest gains `family_id`/`variant_id`/`generator_family`/ + `source_capture_id`, and the fixture generator fails if any image family + crosses a split. +- 2026-08-24 — The `--quality` score is now a hard floor at the public + surface: an encode whose bounded search cannot verify the requested score + fails (`quality target not met` on stderr, exit 1, no output file) instead + of writing an under-target stream with exit 0. `--quality-fallback + lossless` emits a mathematically lossless stream instead + (`status=fallback_lossless`); `--quality-fallback best-effort` emits the + finest verified under-target stream with its true `saturated_top` / + `under_target_work_cap` status (the previous behavior, now explicit). The + `jpxl` facade gained `Error::TargetNotMet(QualityMiss)`, `QualityFallback`, + and `Encoder::with_quality_fallback`; a refused encode still appends its + `JPXL_QUALITY_TRACE` record, and every met-path stream is byte-identical. - 2026-08-22 — Lossy encoding now leads with quality: `jpxl encode --quality [N]` (alias `--ssimulacra2`, `--lossy`) sets a minimum SSIMULACRA2 score (0..100, 100 = lossless); `--bpp`, `--target-bytes`, and the new diff --git a/JPXL/docs/LICENSING-AUDIT.md b/JPXL/docs/LICENSING-AUDIT.md new file mode 100644 index 00000000..e728898c --- /dev/null +++ b/JPXL/docs/LICENSING-AUDIT.md @@ -0,0 +1,203 @@ +# JPXL licensing audit (MIT-only) + +Scope: the JPXL Rust workspace under `JPXL/` and the repository's licensing +surface. This audit answers one question: **is JPXL cleanly MIT-licensed, with +no copyleft and nothing that blocks redistribution under MIT?** + +Answer: **yes.** Every JPXL crate declares MIT; every third-party dependency is +permissive and MIT-compatible; no dependency is copyleft (no GPL / LGPL / AGPL / +MPL / CDDL / EUPL / SSPL / OSL / CeCILL anywhere in the graph). + +- Method: `cargo tree --workspace [--all-features] -e normal,build,dev + --format "{p}|{l}"`, plus manual inspection of `LICENSE`, `JPXL/LICENSE-MIT`, + and every crate's `Cargo.toml`. +- Host / date: recorded per run; the dependency set below reflects + `JPXL/Cargo.lock` as committed. + +--- + +## 1. JPXL's own packages — MIT + +- Workspace default: `JPXL/Cargo.toml` sets `[workspace.package] license = "MIT"`. +- Every member crate (`jpxl`, `jpxl-bitstream`, `jpxl-core`, `jpxl-entropy`, + `jpxl-decode`, `jpxl-encode`, `jpxl-encode-policy`, `jpxl-cli`, + `jpxl-conformance`, `jpxl-perceptual`, `jpxl-plan-render`, and the WIP + `jpxl-jpeg`) declares `license.workspace = true`, i.e. inherits MIT. No crate + overrides it. No crate uses a dual or `OR`-ed expression of its own. +- Repository license files: + - `LICENSE` (repo root) — MIT, `SPDX-License-Identifier: MIT`, + "Copyright (c) 2026 JPXL contributors", plus a trailing note that the + *gitignored* working material (a BSD-3-Clause libjxl checkout used only as a + black-box oracle, copyrighted ISO/IEC standards documents, and local test + images) is **not** covered and **not** distributed. + - `JPXL/LICENSE-MIT` — the identical MIT grant, same copyright line. + +**Consistency check:** the two files carry the same MIT terms and the same +copyright holder and year. They are consistent; the only difference is the root +`LICENSE`'s explanatory note about gitignored non-distributed material, which is +appropriate to keep at the repo root. **No edit required.** + +--- + +## 2. Copyleft scan — none + +``` +cargo tree --workspace --all-features -e normal,build,dev --format "{p}|{l}" \ + | grep -iE 'GPL|LGPL|AGPL|MPL|CDDL|EUPL|CPAL|SSPL|OSL|CeCILL' +``` + +Result: **no matches.** The old AGPL `jxl-encoder` project (AGPL-3.0-only OR a +commercial license) is **not present** in this repository and is not a +dependency; per AGENTS.md §2 its only permitted channel of inheritance is the +`JPEG_XL_CLEAN_IMPLEMENTATION_LESSONS.md` lessons file, which carries no code. + +--- + +## 3. Third-party dependencies + +Every dependency below is permissive and compatible with distributing the +aggregate under MIT. Licenses shown are the crates' own SPDX expressions; where +an expression is an `OR`, the project may elect the MIT arm when one is offered. + +### 3.1 Default build (`cargo build --workspace`, no optional features) + +These are the crates a normal build and the shipped CLI actually pull. All but +two offer MIT directly; the two that do not (`moxcms`, `pxfm`) offer +BSD-3-Clause, which is permissive and MIT-compatible. + +| Crate | License (SPDX) | MIT arm? | +|---|---|---| +| adler2 | 0BSD OR MIT OR Apache-2.0 | yes | +| autocfg | Apache-2.0 OR MIT | yes | +| bitflags | MIT OR Apache-2.0 | yes | +| bytemuck | Zlib OR Apache-2.0 OR MIT | yes | +| byteorder-lite | Unlicense OR MIT | yes | +| cfg-if | MIT OR Apache-2.0 | yes | +| color_quant | MIT | yes | +| crc32fast | MIT OR Apache-2.0 | yes | +| crossbeam-deque | MIT OR Apache-2.0 | yes | +| crossbeam-epoch | MIT OR Apache-2.0 | yes | +| crossbeam-utils | MIT OR Apache-2.0 | yes | +| either | MIT OR Apache-2.0 | yes | +| fax | MIT | yes | +| fdeflate | MIT OR Apache-2.0 | yes | +| flate2 | MIT OR Apache-2.0 | yes | +| gif | MIT OR Apache-2.0 | yes | +| half | MIT OR Apache-2.0 | yes | +| image | MIT OR Apache-2.0 | yes | +| image-webp | MIT OR Apache-2.0 | yes | +| miniz_oxide | MIT OR Zlib OR Apache-2.0 | yes | +| **moxcms** | **BSD-3-Clause OR Apache-2.0** | **no (BSD-3-Clause)** | +| num-traits | MIT OR Apache-2.0 | yes | +| png | MIT OR Apache-2.0 | yes | +| proc-macro2 | MIT OR Apache-2.0 | yes | +| **pxfm** | **BSD-3-Clause OR Apache-2.0** | **no (BSD-3-Clause)** | +| qoi | MIT OR Apache-2.0 | yes | +| quick-error | MIT OR Apache-2.0 | yes | +| quote | MIT OR Apache-2.0 | yes | +| rayon | MIT OR Apache-2.0 | yes | +| rayon-core | MIT OR Apache-2.0 | yes | +| safe_arch | Zlib OR Apache-2.0 OR MIT | yes | +| simd-adler32 | MIT | yes | +| syn | MIT OR Apache-2.0 | yes | +| tiff | MIT | yes | +| unicode-ident | (MIT OR Apache-2.0) AND Unicode-3.0 | yes (code); Unicode-3.0 covers data tables | +| weezl | MIT OR Apache-2.0 | yes | +| wide | Zlib OR Apache-2.0 OR MIT | yes | +| zerocopy | BSD-2-Clause OR Apache-2.0 OR MIT | yes | +| zerocopy-derive | BSD-2-Clause OR Apache-2.0 OR MIT | yes | +| zune-core | MIT OR Apache-2.0 OR Zlib | yes | +| zune-jpeg | MIT OR Apache-2.0 OR Zlib | yes | + +`moxcms`, `pxfm`, `image`, `image-webp`, `png`, `gif`, `tiff`, `qoi`, +`zune-*`, `weezl`, `color_quant`, `fax`, `byteorder-lite` reach the graph only +through the `image` raster adapter used by `jpxl-cli`; the public `jpxl` facade +and all normative codec crates are pixel-buffer based and do not pull the file +adapters (see `JPXL/Cargo.toml` notes). + +### 3.2 Optional, measurement-only (feature-gated, off by default) + +These enter **only** with the `jpxl-conformance` metric features +(`ssimulacra2` / `butteraugli`), used to *grade* lossy encodes against the +oracle. A default `cargo build --workspace` does not fetch them, and no +normative crate depends on them (see the extended note in `JPXL/Cargo.toml`). + +| Crate | License (SPDX) | MIT arm? | +|---|---|---| +| aligned-vec | MIT | yes | +| archmage | MIT OR Apache-2.0 | yes | +| archmage-macros | MIT OR Apache-2.0 | yes | +| av-data | MIT | yes | +| byte-slice-cast | MIT | yes | +| **butteraugli** | **BSD-3-Clause** | **no (BSD-3-Clause)** | +| bytes | MIT | yes | +| equator | MIT | yes | +| equator-macro | MIT | yes | +| **imgref** | **CC0-1.0 OR Apache-2.0** | **no (CC0 / Apache-2.0)** | +| log | MIT OR Apache-2.0 | yes | +| magetypes | MIT OR Apache-2.0 | yes | +| num-bigint | MIT OR Apache-2.0 | yes | +| num-derive | MIT OR Apache-2.0 | yes | +| num-integer | MIT OR Apache-2.0 | yes | +| num-rational | MIT OR Apache-2.0 | yes | +| rgb | MIT | yes | +| rustversion | MIT OR Apache-2.0 | yes | +| safe_unaligned_simd | MIT OR Apache-2.0 | yes | +| **ssimulacra2** | **BSD-2-Clause** | **no (BSD-2-Clause)** | +| syn (3.x) | MIT OR Apache-2.0 | yes | +| thiserror | MIT OR Apache-2.0 | yes | +| thiserror-impl | MIT OR Apache-2.0 | yes | +| **v_frame** | **BSD-2-Clause** | **no (BSD-2-Clause)** | +| yuvxyb | MIT | yes | +| yuvxyb-math | MIT | yes | + +--- + +## 4. Items worth a human's eye (none blocking) + +1. **`moxcms` and `pxfm` on the default path** are `BSD-3-Clause OR Apache-2.0` + — MIT is not one of their options. This is fully compatible with shipping the + aggregate under MIT (permissive dependencies inside an MIT project are + normal), but the BSD-3-Clause copyright/attribution notices for these two + crates should be retained in any distributed `NOTICE`/third-party-licenses + bundle. They enter only through the `image` raster adapter in `jpxl-cli`. + +2. **`butteraugli` (measurement-only) is a Rust port of libjxl's butteraugli.** + BSD-3-Clause, permissive. It is off by default and never linked into a normal + build. AGENTS.md §2 and the `JPXL/Cargo.toml` clean-room note already record + that using it as a *metric* is the same category as running `cjxl`/`djxl` as + oracles, whereas wiring it into the encoder's own rate control would make the + perceptual model a derivative of libjxl and needs a recorded decision first. + No such wiring exists today. + +3. **`unicode-ident`** carries `Unicode-3.0` for its data tables (in addition to + MIT/Apache for code). Unicode-3.0 is permissive and only asks that its notice + be retained. `imgref` (measurement path) offers CC0-1.0 (public-domain + equivalent) or Apache-2.0. Neither is copyleft. + +4. **Attribution hygiene for release:** because the graph mixes MIT with + Apache-2.0, BSD-2/3-Clause, Zlib, 0BSD, Unlicense, CC0 and Unicode-3.0 + (all permissive), a shipped binary should carry a generated third-party + license notice (e.g. `cargo about` / `cargo-bundle-licenses`). This is an + attribution convenience, not a licensing conflict. + +## 5. ISO / standards text + +No ISO/IEC 18181 text is referenced or embedded in any shipped crate. The +standards markdown/PDF sources live in gitignored directories +(`markdowns/`, `latex/`, `original-pdfs-.../`) and never enter the build or git +history, per AGENTS.md §3 and §9. Clause-number citations in comments are fine; +quoted normative passages are not, and none were found in the workspace crates. + +## Reproduce + +```sh +cd JPXL +# Full dependency + license graph (default features): +cargo tree --workspace -e normal,build,dev --format "{p} {l}" +# Superset including the optional measurement metrics: +cargo tree --workspace --all-features -e normal,build,dev --format "{p} {l}" +# Copyleft check (expect no output): +cargo tree --workspace --all-features -e normal,build,dev --format "{l}" \ + | grep -iE 'GPL|LGPL|AGPL|MPL|CDDL|EUPL|SSPL' +``` diff --git a/JPXL/docs/jpeg-recompression-plan.md b/JPXL/docs/jpeg-recompression-plan.md new file mode 100644 index 00000000..980c5ef8 --- /dev/null +++ b/JPXL/docs/jpeg-recompression-plan.md @@ -0,0 +1,193 @@ +# JPEG bitstream recompression — implementation plan + +Status: **plan** (no code yet). AKR work record: +`jpegxl-rs.work.jpeg-bitstream-recompression`. Written 2026-08-25, from +ISO/IEC 18181-2:2024 §9.11 + Annex A (`sources/markdowns/standard-markdowns/part2.md`) +and ISO/IEC 18181-1:2024 (`sources/latex/part1.tex`). Clean-room: everything +below derives from the standards and this workspace; libjxl remains a +black-box oracle (`cjxl --lossless_jpeg=1`, `djxl --pixels_to_jpeg` / +default JPEG reconstruction) for interop testing only. + +## 1. What the feature is + +An existing JPEG (ISO/IEC 10918-1) is represented as a JPEG XL file +**without decoding to pixels**: its quantized DCT coefficients are carried +in a kVarDCT frame using only 8×8 blocks (Part 1 §C, note in the frame +header clause: "existing JPEG images can be represented using the kVarDCT +encoding using only 8×8 DCTs"), and everything the coefficients do not +determine — marker order, Huffman tables, scan script, padding bits, +restart markers, APPn/COM payloads, trailing garbage — goes into the +**JPEG Bitstream Reconstruction Data box** (`jbrd`, Part 2 §9.11). Decoding +the frame yields the image; running Part 2 Annex A over the coefficients + +`jbrd` yields the **byte-exact original JPEG**. Typical size win on +Huffman-coded JPEGs is ~20 %; the transcode is exact by construction, so +there is no quality question and none of the perceptual controller is +involved. + +Why we want it: the user's archives are JPEG-heavy (thousands of scanned +paintings); pixel-decoding them into a fresh lossy encode either loses +generation quality or (lossless) inflates the file. Recompression is the +only path that is simultaneously smaller and exactly reversible. + +## 2. What the standard requires (digest) + +### 2.1 `jbrd` box (Part 2 §9.11, Tables 11–18) + +A bundle-coded header (same `Bits()`/`U32()`/`Bool()` conventions as +Part 1) followed by one Brotli stream: + +- `is_grey`; `marker[]` — one byte per segment, `0xC0 + Bits(6)` each, + terminated by `0xD9` (EOI). Counts of specific values determine array + sizes downstream: `0xE0..=0xEF` → `num_app_markers`, `0xFE` → + `num_com_markers`, `0xDA` → `num_scans`, `0xFF` → `num_intermarker`, + `0xDD` present → `has_dri`. +- `AppMarker { type, length }` per APPn; `com_length[]`; + `num_quant_tables` (1 + Bits(2)) and `QuantTable { precision, index, + is_last }`; `comp_type` (2 bits; `== 3` carries explicit `num_comp` and + 8-bit `component_id[]`; `comp_type == 2` gates a conditional row); + `component_q_idx[]`. +- `num_huff` (U32) and `HuffmanCode { is_ac, id, is_last, counts[16], + values[sum(counts)] }`. +- `ScanInfo { num_comps, Ss, Se, Al, Ah, ScanComponentInfo{comp_idx, + ac_tbl_idx, dc_tbl_idx}[], last_needed_pass }` per scan; + `restart_interval` if `has_dri`; `ScanMoreInfo { reset_point[], + ExtraZeroRun{num_runs, run_length}[] }` per scan (progressive/EOBRUN + quirks and non-optimal encoders). +- `intermarker_length[]`, `tail_data_length`, `has_padding` + padding + bit-bank (`nbit`, `bbit[]`). +- Brotli payload: APPn payloads of type 0, COM payloads, unrecognized + inter-marker data, tail data — in that order. + +**OCR caveat**: the Table 11–18 transcription in `part2.md` is visibly +scrambled in places (mis-merged rows, mangled U32 distributions). Before +implementing the bundle reader/writer, spot-check every table against the +original scan pages 12–14 (`sources/original-pdfs-do-not-read-first-if-markdown-exists/original/`, +Part 2 = "2024 jun"), per the STANDARDS_INDEX caveat. Treat the exact U32 +distributions as **unverified** until then. + +### 2.2 Reconstruction procedure (Part 2 Annex A) + +Deterministic serializer over the marker array with iterator semantics +(every stored element consumed exactly once, in order): SOF (types +0xC0/C1/C2/C9/CA; component order note — `jpeg_upsampling` is Cb,Y,Cr +order while the JPEG SOF is Y,Cb,Cr), DHT (10918-1 B.2.4.2 with the +last-nonzero `L_i` decremented-by-1 convention), RSTn, EOI (+ tail data), +SOS (10918-1 B.2.3; entropy coding per 10918-1 F.1.2/G.1.2 with the +`extra_zero_run`/`reset_point`/padding-bit amendments for bit-exactness), +DQT (factors come **from the codestream's dequant matrices**, Part 1 +I.2.4; an unused table repeats the previous one), DRI, APPn (type 0 raw; +type 1 ICC re-chunked as `ICC_PROFILE` APP2 parts from the decoded ICC +profile; types 2/3 Exif/XMP re-wrapped from the Exif/XML boxes), COM, +unrecognized (0xFF) raw copy. + +### 2.3 Part 1 carriage requirements + +The recompressed frame must decode to exactly the JPEG's dequantized-able +coefficients, which constrains the encoder to a fixed shape: + +- `metadata.xyb_encoded = false`, frame `do_YCbCr = true` (or grey); + `jpeg_upsampling[3]` for 4:2:0/4:2:2/4:4:0 (Part 1: subsampled channels + are coded at reduced resolution and upsampled per J.2; group division + unaffected). +- All varblocks 8×8 DCT8; no Gaborish, no EPF, no patches/splines/noise; + CfL zeroed (or the exact signalled defaults that make the coefficient + passthrough exact — to be pinned from Part 1 during implementation). +- Dequantization matrices in **RAW encoding mode** (Part 1 I.2.4: "for + encoding mode RAW, the dequantization matrices are equal to the params + matrices multiplied by params…") so each JPEG quant table maps to an + exact JXL dequant matrix and Annex A's DQT can read the factors back. +- LF/DC: JPEG DC coefficients land in the LF image; the LF quantization + must be chosen so quantized-LF ↔ JPEG-DC is bijective (this is the one + place where "exact" needs a worked proof against Part 1 §H/I rather + than an assumption — flagged as an open question below). +- HF: JPEG's zigzag AC coefficients map into JXL's coefficient order with + the standard's natural-order permutation; values pass through the + entropy coder unchanged (quantized integers). + +## 3. Architecture + +New crate **`jpxl-jpeg`** (parallel to `jpxl-entropy`): a clean-room +ISO/IEC 10918-1 *coefficient-level* codec. No pixel path, no IDCT: + +- `parse`: markers → segments; DQT/DHT/SOF/SOS/DRI/APPn/COM structures; + baseline + progressive Huffman entropy decode to per-component + coefficient planes; capture everything Annex A needs (padding bits, + extra zero runs, reset points, inter-marker garbage, tail data). +- `serialize`: the exact inverse — Annex A is its specification. +- Self-gate: `serialize(parse(jpeg)) == jpeg` byte-for-byte, with no JXL + involvement. This isolates all 10918-1 subtleties before any codestream + work. + +Boundary rules (mirrors `jpxl-entropy`): `jpxl-encode` may depend on +`jpxl-jpeg`; `jpxl-jpeg` depends on nothing in the workspace except +possibly `jpxl-core`. The policy crate is **not involved** — recompression +is a deterministic transcode, not a search. Container work (reading and +writing the `jbrd` box, Brotli) lands in `jpxl-bitstream`; a Brotli codec +is a new dependency decision (the workspace is near-zero-dependency — +either a vetted `brotli` crate behind the CLI/container feature, or +scope-limited vendoring; decision to be made at Phase C, recorded in AKR). + +Prerequisites in existing crates (verified 2026-08-25): + +- `jpxl-encode` writes only `xyb_encoded` frames today + (`vardct/headers.rs` hard-codes the assumption; `frame.rs` writes + `do_YCbCr = false`): needs the YCbCr + `jpeg_upsampling` header path + and RAW dequant-matrix signalling. +- `jpxl-decode` already decodes YCbCr + `jpeg_upsampling` + (`frame/upsampling.rs`, `e2e_ycbcr.rs`), so our own decoder can oracle + the carriage phase. Its dequant RAW path exists + (`vardct/dequant_matrix.rs`). +- The decoder needs a coefficient-export surface (decode to quantized + coefficients, not pixels) for reconstruction — today it renders pixels. + +## 4. Phases and gates + +Each phase is a separately committable, gated unit; the phase gates are +the completion metrics for the work record. + +- **Phase A — `jpxl-jpeg` codec.** Gate: byte-exact + `serialize(parse(x)) == x` over ≥500 JPEGs sampled deterministically + from the Pol Art archive (read-only; decoded copies only, originals + never touched) covering baseline/progressive, 4:4:4/4:2:0/4:2:2, + grey, restart intervals, Exif/ICC/XMP, and trailing garbage; plus + graceful *refusal* (typed error) of arithmetic-coded, hierarchical, + 12-bit, and lossless JPEG modes. +- **Phase B — coefficient carriage.** Encode coefficient planes into a + YCbCr VarDCT frame; decode with `jpxl-decode`'s new coefficient + export. Gate: coefficients round-trip exactly (all components, all + subsamplings) on the Phase A corpus; existing decoder conformance and + encoder tests untouched. +- **Phase C — `jbrd` + Annex A.** Writer populates the box during Phase B + encode; reconstructor = `jpxl-jpeg::serialize` fed from the decoded + coefficients + box. Gate: **byte-exact original JPEG from the .jxl + file alone**, same corpus; total .jxl ≤ original JPEG bytes on ≥95 % + of the corpus with the geomean reduction reported (expect ≈ −15–20 %). +- **Phase D — surface.** CLI: `jpxl encode in.jpg out.jxl` defaults to + recompression when the input is a compatible JPEG (explicit + `--jpeg=pixels|recompress|auto` to override; incompatible modes fall + back to the pixel path with a notice); `jpxl decode out.jxl out.jpg` + reconstructs the original when `jbrd` is present. Facade: + `Encoder::recompress_jpeg(&[u8])`, `Decoder`-side reconstruction. + Gate: CLI round-trip tests, CHANGELOG, README. +- **Phase E — interop (black-box only).** Our .jxl reconstructed by + `djxl` equals the original; `cjxl --lossless_jpeg=1` output is + reconstructed by us. Gate: both directions byte-exact on a 50-image + sample. Any mismatch is debugged against the *standard*, never by + reading libjxl source. + +## 5. Open questions (to resolve during Phase B, recorded in AKR as they close) + +1. **DC bijectivity**: the exact Part 1 LF quantization arithmetic that + makes JPEG DC carriage lossless (worked derivation from §H/I needed; + this is the highest technical risk). +2. **CfL/filter signalling**: the precise header settings that make the + HF passthrough exact (zero CfL vs. signalled defaults). +3. **Table 11–18 field distributions**: verify against the original scan + (OCR caveat above) before freezing the bundle reader. +4. **10918-1 source**: the workspace has no copy of the JPEG spec; Annex + A references B.2.3/B.2.4/F.1.2/G.1.2 normatively. Acquire and + register it under `sources/` before Phase A entropy work (the + marker-structure layer can proceed from Part 2 alone). +5. **Brotli dependency** policy (see §3). +6. **Levels/profile**: whether recompressed streams stay within level 5 + for the archive's dimensions (Part 1 §M check). diff --git a/JPXL/tools/README-codec-compare.md b/JPXL/tools/README-codec-compare.md index 5732ee09..ccdc0350 100644 --- a/JPXL/tools/README-codec-compare.md +++ b/JPXL/tools/README-codec-compare.md @@ -28,23 +28,30 @@ A jpxl **quality** curve row adds: `jpxl encode --quality Q [--effort fast|balanced] --threads N in.ppm out.jxl` selects the perceptual VarDCT path with a minimum SSIMULACRA2 score `Q` -(0..100; 100 = lossless) and prints exactly one line: +(0..100; 100 = lossless) and, on success, prints exactly one line: ``` quality_target=85.0000 achieved=85.1372 bytes=412883 metric=ssimulacra2-jpxl-1 effort=balanced probes=3 prices=2 status=met ``` `status` is one of: `met`, `met_adjacent_rungs`, `met_work_cap`, -`saturated_floor`, `saturated_top`, `rescued_fresh_structure`, -`routed_to_lossless`, `unsupported_too_small`. - -When `JPXL_QUALITY_TRACE=<path>` is set, a `jpxl.quality-trace/1` JSONL file is -written. The harness sets this per curve point (under the work dir) unless +`saturated_floor`, `saturated_top`, `under_target_work_cap`, +`rescued_fresh_structure`, `routed_to_lossless`, `fallback_lossless`, +`unsupported_too_small`. + +The score is a hard floor by default: an encode whose bounded search cannot +verify `Q` exits 1 with `quality target not met` on stderr and writes no +output file. The harness therefore passes `--quality-fallback best-effort`, +which emits the finest verified under-target stream with its true +`saturated_top` / `under_target_work_cap` status so a curve point is always +measurable (`--quality-fallback lossless` instead emits a lossless stream as +`fallback_lossless`). + +When `JPXL_QUALITY_TRACE=<path>` is set, a `jpxl.quality-trace/2` JSONL file is +written (also for a refused encode — the failed search is still calibration +input). The harness sets this per curve point (under the work dir) unless `--no-quality-trace` is given, and merges `wall_by_phase` into the record. -> Note: targets below 100 currently error from the binary ("perceptual quality -> targets land in PR 4"). `--quality 100` runs live (status `routed_to_lossless`). - ## Subcommands ### `curve` — quality axis diff --git a/JPXL/tools/bench_vs_libjxl.sh b/JPXL/tools/bench_vs_libjxl.sh new file mode 100644 index 00000000..3f4c8a71 --- /dev/null +++ b/JPXL/tools/bench_vs_libjxl.sh @@ -0,0 +1,315 @@ +#!/usr/bin/env bash +# bench_vs_libjxl.sh — reproducible size / SSIMULACRA2 / wall comparison of the +# JPXL encoder against the libjxl oracle (cjxl / djxl), with full provenance. +# +# WHAT IT DOES +# For every input image and every requested setting it emits one table row: +# encoder setting bytes bpp ssimulacra2 wall_s status +# JPXL rows come from `jpxl encode --quality Q`; cjxl rows from `cjxl -d D`. +# Every stream is decoded back to pixels and scored with the SAME metric — +# the in-tree production SSIMULACRA2 exposed by `jpxl compare` — so the two +# encoders are graded on one identical yardstick. Sizes are the real encoded +# byte counts; wall is the best of N encode runs. +# +# This is deliberately an HONEST, not a matched, comparison: JPXL targets an +# SSIMULACRA2 floor while cjxl targets a Butteraugli distance, so the reader +# compares the (bytes, ssimulacra2) points, not same-named settings. See the +# project README section "Quality, density, and speed versus libjxl". +# +# PROVENANCE +# A header block records: UTC date, host, every binary's path + version + +# sha256, the exact flags, the SSIMULACRA2 metric version, and each input's +# sha256 and dimensions. With --jsonl the same data is written as one JSON +# object per row (schema "jpxl.bench-vs-libjxl/1") for machine consumption. +# +# ORACLES +# cjxl / djxl are looked up in tools/oracle-bin/ first (where +# setup-oracles.sh installs them), then on $PATH, then via --cjxl/--djxl. +# If they cannot run on this host (e.g. only the Windows .exe oracle is +# present, or none was set up) the script still runs and emits JPXL-only +# rows, clearly marked — the table stays reproducible and simply fills the +# cjxl columns when a working oracle is available. +# +# CLEAN ROOM (AGENTS.md §2) +# cjxl/djxl are used strictly as black boxes: this script only *runs* them. +# No libjxl source is read. +# +# INPUTS +# Any raster JPXL can read (PNG, PPM, JPEG, ...). Each input is first +# normalised to a canonical P6 PPM via a *lossless* JPXL round-trip +# (encode → decode), so both encoders and the scorer see identical source +# pixels with no external image tool required. RGB (3-channel) inputs only; +# images with alpha are skipped with a note (the lossy encoder has no alpha). +# +# USAGE +# tools/bench_vs_libjxl.sh [options] <input|dir> [<input|dir> ...] +# +# OPTIONS +# --quality "Q ..." SSIMULACRA2 target(s) for JPXL (default: "70 85 90") +# --distance "D ..." Butteraugli distance(s) for cjxl (default: "3.0 1.5 1.0") +# --effort E JPXL lossy effort: fast|balanced (default: balanced) +# --cjxl-effort N cjxl -e effort 1..9 (default: 7) +# --threads N thread cap for both encoders (default: 4) +# --runs N encode timing runs, best is kept (default: 1) +# --jpxl PATH jpxl binary (default: target/fast-debug or release) +# --cjxl PATH cjxl binary override +# --djxl PATH djxl binary override +# --work-dir DIR scratch dir (default: a mktemp under $TMPDIR) +# --jsonl PATH also write one JSON object per row to PATH +# -h, --help this help +# +# EXIT: 0 if at least one row was produced; non-zero on setup error. +set -uo pipefail + +# ---------------------------------------------------------------- locate self +script_dir="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" +jpxl_root="$(cd -- "${script_dir}/.." && pwd)" +oracle_bin="${script_dir}/oracle-bin" + +# ------------------------------------------------------------------- defaults +qualities="70 85 90" +distances="3.0 1.5 1.0" +effort="balanced" +cjxl_effort="7" +threads="4" +runs="1" +jpxl_bin="" +cjxl_bin="" +djxl_bin="" +work_dir="" +jsonl_path="" +inputs=() + +die() { printf 'error: %s\n' "$*" >&2; exit 2; } +note() { printf '%s\n' "# $*"; } + +# --------------------------------------------------------------- parse args +while [[ $# -gt 0 ]]; do + case "$1" in + --quality) qualities="$2"; shift 2 ;; + --distance) distances="$2"; shift 2 ;; + --effort) effort="$2"; shift 2 ;; + --cjxl-effort) cjxl_effort="$2"; shift 2 ;; + --threads) threads="$2"; shift 2 ;; + --runs) runs="$2"; shift 2 ;; + --jpxl) jpxl_bin="$2"; shift 2 ;; + --cjxl) cjxl_bin="$2"; shift 2 ;; + --djxl) djxl_bin="$2"; shift 2 ;; + --work-dir) work_dir="$2"; shift 2 ;; + --jsonl) jsonl_path="$2"; shift 2 ;; + -h|--help) sed -n '2,72p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'; exit 0 ;; + --*) die "unknown option: $1" ;; + *) inputs+=("$1"); shift ;; + esac +done + +[[ ${#inputs[@]} -gt 0 ]] || die "no input files or directories given (see --help)" + +# --------------------------------------------------------------- find jpxl +if [[ -z "${jpxl_bin}" ]]; then + for cand in "${jpxl_root}/target/release/jpxl" "${jpxl_root}/target/fast-debug/jpxl"; do + [[ -x "${cand}" ]] && { jpxl_bin="${cand}"; break; } + done +fi +[[ -n "${jpxl_bin}" && -x "${jpxl_bin}" ]] \ + || die "jpxl binary not found; build it (cargo build -p jpxl-cli --release) or pass --jpxl" + +# ---------------------------------------------- find + validate the oracle +# Returns 0 and echoes a usable path, or non-zero if the candidate cannot run +# on this host (wrong architecture, missing, non-executable). +runnable() { "$1" --version >/dev/null 2>&1; } + +find_oracle() { + local name="$1" override="$2" cand + if [[ -n "${override}" ]]; then + runnable "${override}" && { echo "${override}"; return 0; } + return 1 + fi + for cand in "${oracle_bin}/${name}" "$(command -v "${name}" 2>/dev/null || true)"; do + [[ -n "${cand}" && -x "${cand}" ]] || continue + runnable "${cand}" && { echo "${cand}"; return 0; } + done + return 1 +} + +cjxl_path="$(find_oracle cjxl "${cjxl_bin}" || true)" +djxl_path="$(find_oracle djxl "${djxl_bin}" || true)" +have_oracle=0 +[[ -n "${cjxl_path}" && -n "${djxl_path}" ]] && have_oracle=1 + +# --------------------------------------------------------------- work dir +if [[ -z "${work_dir}" ]]; then + work_dir="$(mktemp -d "${TMPDIR:-/tmp}/jpxl-bench.XXXXXX")" || die "mktemp failed" + trap 'rm -rf "${work_dir}"' EXIT +else + mkdir -p "${work_dir}" || die "cannot create work dir ${work_dir}" +fi + +# --------------------------------------------------------------- helpers +sha() { sha256sum "$1" 2>/dev/null | cut -d' ' -f1; } +size() { stat -c '%s' "$1" 2>/dev/null || wc -c < "$1"; } +now() { date +%s.%N; } +elapsed() { awk "BEGIN{printf \"%.3f\", $2 - $1}"; } + +# Parse a P6 PPM header for "W H" (skips the maxval line). Whitespace-robust. +ppm_dims() { + head -c 64 "$1" | tr '\n\t' ' ' | awk '{print $2, $3}' +} + +# in-tree SSIMULACRA2 of two PPMs, or "n/a". +score_ssimulacra2() { + local ref="$1" cand="$2" out + out="$("${jpxl_bin}" compare "${ref}" "${cand}" 2>/dev/null)" || { echo "n/a"; return; } + echo "${out}" | grep -oE 'ssimulacra2_jpxl=[0-9.]+' | head -1 | cut -d= -f2 +} + +metric_version() { + "${jpxl_bin}" compare "$1" "$1" 2>/dev/null \ + | grep -oE 'ssimulacra2_jpxl_version=[^ ]+' | head -1 | cut -d= -f2 +} + +# Best-of-N wall time for a command (args after the first). Prints seconds. +best_wall() { + local n="$1"; shift + local best="" t0 t1 d i + for ((i=0; i<n; i++)); do + t0="$(now)"; "$@" >/dev/null 2>&1; t1="$(now)" + d="$(elapsed "${t0}" "${t1}")" + if [[ -z "${best}" ]] || awk "BEGIN{exit !(${d} < ${best})}"; then best="${d}"; fi + done + echo "${best}" +} + +json_escape() { printf '%s' "$1" | sed 's/\\/\\\\/g; s/"/\\"/g'; } + +emit_jsonl() { + [[ -n "${jsonl_path}" ]] || return 0 + printf '{"schema":"jpxl.bench-vs-libjxl/1","image":"%s","encoder":"%s","setting":"%s","bytes":%s,"bpp":%s,"ssimulacra2":"%s","wall_s":%s,"status":"%s"}\n' \ + "$(json_escape "$1")" "$2" "$3" "$4" "$5" "$6" "$7" "$8" >> "${jsonl_path}" +} + +# --------------------------------------------------------------- gather inputs +files=() +for arg in "${inputs[@]}"; do + if [[ -d "${arg}" ]]; then + while IFS= read -r -d '' f; do files+=("${f}"); done \ + < <(find "${arg}" -type f \( -iname '*.ppm' -o -iname '*.png' -o -iname '*.jpg' \ + -o -iname '*.jpeg' -o -iname '*.bmp' -o -iname '*.tif' -o -iname '*.tiff' \) -print0 | sort -z) + elif [[ -f "${arg}" ]]; then + files+=("${arg}") + else + printf 'warning: skipping %s (not a file or directory)\n' "${arg}" >&2 + fi +done +[[ ${#files[@]} -gt 0 ]] || die "no readable input images found" + +# --------------------------------------------------------------- provenance header +[[ -n "${jsonl_path}" ]] && : > "${jsonl_path}" +metric_ver="?" + +note "jpxl.bench-vs-libjxl provenance" +note "date_utc: $(date -u +%Y-%m-%dT%H:%M:%SZ)" +note "host: $(uname -srm)" +note "jpxl: ${jpxl_bin}" +note "jpxl_version: $("${jpxl_bin}" --version 2>&1 | head -1)" +note "jpxl_sha256: $(sha "${jpxl_bin}")" +if [[ ${have_oracle} -eq 1 ]]; then + note "cjxl: ${cjxl_path}" + note "cjxl_version: $("${cjxl_path}" --version 2>&1 | head -1)" + note "cjxl_sha256: $(sha "${cjxl_path}")" + note "djxl: ${djxl_path}" + note "djxl_version: $("${djxl_path}" --version 2>&1 | head -1)" + note "djxl_sha256: $(sha "${djxl_path}")" +else + note "cjxl/djxl: NOT AVAILABLE on this host — emitting JPXL-only rows." + note " (Run tools/setup-oracles.sh on this platform to enable them.)" +fi +note "jpxl_effort: ${effort} cjxl_effort: -e ${cjxl_effort} threads: ${threads} runs: ${runs}" +note "jpxl_qualities: ${qualities}" +note "cjxl_distances: ${distances}" +note "" + +# --------------------------------------------------------------- table header +printf '%-28s %-8s %-10s %10s %7s %12s %8s %s\n' \ + image encoder setting bytes bpp ssimulacra2 wall_s status + +produced=0 + +for src in "${files[@]}"; do + base="$(basename "${src}")" + stem="${base%.*}" + ref_ppm="${work_dir}/${stem}.src.ppm" + + # Normalise to canonical P6 via a lossless JPXL round-trip. + if [[ "${src}" == *.ppm || "${src}" == *.PPM ]]; then + cp -f "${src}" "${ref_ppm}" + else + if ! "${jpxl_bin}" encode "${src}" "${work_dir}/${stem}.src.jxl" >/dev/null 2>&1 \ + || ! "${jpxl_bin}" decode "${work_dir}/${stem}.src.jxl" "${ref_ppm}" >/dev/null 2>&1; then + printf 'warning: cannot normalise %s to PPM (alpha or unsupported input?) — skipping\n' "${src}" >&2 + continue + fi + fi + read -r w h < <(ppm_dims "${ref_ppm}") + [[ -n "${w}" && -n "${h}" && "${w}" -gt 0 && "${h}" -gt 0 ]] \ + || { printf 'warning: bad PPM dims for %s — skipping\n' "${src}" >&2; continue; } + pixels=$(( w * h )) + [[ "${metric_ver}" == "?" ]] && metric_ver="$(metric_version "${ref_ppm}")" + + # ------------------------------------------------------------- JPXL rows + for q in ${qualities}; do + out_jxl="${work_dir}/${stem}.q${q}.jxl" + dec_ppm="${work_dir}/${stem}.q${q}.ppm" + enc_out="$("${jpxl_bin}" encode --quality "${q}" --effort "${effort}" \ + --quality-fallback best-effort --threads "${threads}" \ + "${ref_ppm}" "${out_jxl}" 2>&1)" + if [[ ! -s "${out_jxl}" ]]; then + printf '%-28s %-8s %-10s %10s %7s %12s %8s %s\n' \ + "${stem}" jpxl "q${q}" - - - - "encode-failed" + continue + fi + status="$(printf '%s\n' "${enc_out}" | grep -oE 'status=[a-z_]+' | head -1 | cut -d= -f2)" + [[ -n "${status}" ]] || status="ok" + wall="$(best_wall "${runs}" "${jpxl_bin}" encode --quality "${q}" --effort "${effort}" \ + --quality-fallback best-effort --threads "${threads}" "${ref_ppm}" "${out_jxl}")" + "${jpxl_bin}" decode "${out_jxl}" "${dec_ppm}" >/dev/null 2>&1 + s2="$(score_ssimulacra2 "${ref_ppm}" "${dec_ppm}")" + b="$(size "${out_jxl}")" + bpp="$(awk "BEGIN{printf \"%.4f\", ${b}*8/${pixels}}")" + printf '%-28s %-8s %-10s %10s %7s %12s %8s %s\n' \ + "${stem}" jpxl "q${q}" "${b}" "${bpp}" "${s2}" "${wall}" "${status}" + emit_jsonl "${base}" jpxl "q${q}" "${b}" "${bpp}" "${s2}" "${wall}" "${status}" + produced=1 + done + + # ------------------------------------------------------------- cjxl rows + if [[ ${have_oracle} -eq 1 ]]; then + for d in ${distances}; do + out_jxl="${work_dir}/${stem}.d${d}.jxl" + dec_ppm="${work_dir}/${stem}.d${d}.ppm" + if ! "${cjxl_path}" -d "${d}" -e "${cjxl_effort}" --num_threads="${threads}" \ + "${ref_ppm}" "${out_jxl}" >/dev/null 2>&1 || [[ ! -s "${out_jxl}" ]]; then + printf '%-28s %-8s %-10s %10s %7s %12s %8s %s\n' \ + "${stem}" cjxl "d${d}" - - - - "encode-failed" + continue + fi + wall="$(best_wall "${runs}" "${cjxl_path}" -d "${d}" -e "${cjxl_effort}" \ + --num_threads="${threads}" "${ref_ppm}" "${out_jxl}")" + "${djxl_path}" "${out_jxl}" "${dec_ppm}" >/dev/null 2>&1 + s2="$(score_ssimulacra2 "${ref_ppm}" "${dec_ppm}")" + b="$(size "${out_jxl}")" + bpp="$(awk "BEGIN{printf \"%.4f\", ${b}*8/${pixels}}")" + printf '%-28s %-8s %-10s %10s %7s %12s %8s %s\n' \ + "${stem}" cjxl "d${d}" "${b}" "${bpp}" "${s2}" "${wall}" "ok" + emit_jsonl "${base}" cjxl "d${d}" "${b}" "${bpp}" "${s2}" "${wall}" "ok" + produced=1 + done + fi +done + +note "" +note "ssimulacra2_metric: ${metric_ver} (in-tree production metric; higher is better, 100 = identical)" +[[ -n "${jsonl_path}" ]] && note "jsonl: ${jsonl_path}" + +[[ ${produced} -eq 1 ]] || die "no comparison rows were produced" +exit 0 diff --git a/JPXL/tools/build-release-final.sh b/JPXL/tools/build-release-final.sh new file mode 100644 index 00000000..97550fbf --- /dev/null +++ b/JPXL/tools/build-release-final.sh @@ -0,0 +1,85 @@ +#!/usr/bin/env bash +# Build the shipping `jpxl` binary: the `release-final` profile (full LTO) +# wrapped in the measured PGO cycle. +# +# 1. build instrumented (-Cprofile-generate) with release-final +# 2. train on the canonical test-set images across the three encode paths +# (perceptual --quality, target-rate --bpp, lossless) plus a decode +# 3. merge the profiles with the toolchain's llvm-profdata +# 4. rebuild with -Cprofile-use +# +# The training set is the one the recorded PGO evidence used: the seven +# canonical 1024x768 photographs at test-set/ plus the 12 MP mid photo, with +# larger images left unseen so the recorded generalization claim +# (@jpegxl-rs.evidence.phase8-5-pgo-generalizes-2026-08-14) keeps meaning. +# +# Usage: tools/build-release-final.sh [--out <dir>] +# Run from anywhere; paths are derived from the script location. + +set -euo pipefail + +JPXL_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)" +REPO_DIR="$(cd -- "$JPXL_DIR/.." && pwd)" +TEST_SET="$REPO_DIR/test-set" +OUT_DIR="$JPXL_DIR/target/release-final" + +if [[ "${1:-}" == "--out" && -n "${2:-}" ]]; then + OUT_DIR="$2" +fi + +HOST_TRIPLE="$(rustc -vV | sed -n 's/^host: //p')" +LLVM_PROFDATA="$(rustc --print sysroot)/lib/rustlib/$HOST_TRIPLE/bin/llvm-profdata" +if [[ ! -x "$LLVM_PROFDATA" ]]; then + echo "llvm-profdata not found; install with: rustup component add llvm-tools" >&2 + exit 1 +fi + +PGO_DIR="$(mktemp -d "${TMPDIR:-/tmp}/jpxl-pgo.XXXXXX")" +TRAIN_OUT="$PGO_DIR/out" +mkdir -p "$TRAIN_OUT" +trap 'rm -rf "$PGO_DIR"' EXIT + +TRAIN_IMAGES=("$TEST_SET"/*.png) +MID_IMAGE="$(find "$TEST_SET/one-12mp" -name '*.png' | head -n 1 || true)" +if [[ ${#TRAIN_IMAGES[@]} -eq 0 || ! -f "${TRAIN_IMAGES[0]}" ]]; then + echo "no training images found at $TEST_SET/*.png" >&2 + exit 1 +fi + +echo "== 1/4: instrumented release-final build" +(cd "$JPXL_DIR" && RUSTFLAGS="-Cprofile-generate=$PGO_DIR" \ + cargo build --profile release-final -p jpxl-cli) +JPXL_BIN="$JPXL_DIR/target/release-final/jpxl" + +echo "== 2/4: training encodes" +train_one() { + local img="$1" stem + stem="$(basename "${img%.png}")" + "$JPXL_BIN" encode --quality --effort balanced --threads 4 \ + "$img" "$TRAIN_OUT/$stem-q.jxl" + "$JPXL_BIN" encode --bpp 1.0 --effort balanced --threads 4 \ + "$img" "$TRAIN_OUT/$stem-r.jxl" + "$JPXL_BIN" encode --effort 1 --threads 4 \ + "$img" "$TRAIN_OUT/$stem-l.jxl" + "$JPXL_BIN" decode "$TRAIN_OUT/$stem-q.jxl" "$TRAIN_OUT/$stem-q.ppm" +} +for img in "${TRAIN_IMAGES[@]}"; do + train_one "$img" +done +if [[ -n "$MID_IMAGE" ]]; then + train_one "$MID_IMAGE" +fi + +echo "== 3/4: merging profiles" +"$LLVM_PROFDATA" merge -o "$PGO_DIR/merged.profdata" "$PGO_DIR"/*.profraw + +echo "== 4/4: PGO-optimized release-final build" +(cd "$JPXL_DIR" && RUSTFLAGS="-Cprofile-use=$PGO_DIR/merged.profdata" \ + cargo build --profile release-final -p jpxl-cli) + +if [[ "$OUT_DIR" != "$JPXL_DIR/target/release-final" ]]; then + mkdir -p "$OUT_DIR" + cp "$JPXL_BIN" "$OUT_DIR/jpxl" +fi +echo "shipping binary: $OUT_DIR/jpxl" +"$OUT_DIR/jpxl" --version diff --git a/JPXL/tools/codec_compare.py b/JPXL/tools/codec_compare.py index 6fc98ac5..b79b8c90 100644 --- a/JPXL/tools/codec_compare.py +++ b/JPXL/tools/codec_compare.py @@ -43,7 +43,10 @@ # /2 adds a same-effort JPXL rate baseline and keeps controller-achieved scores # separate from the common decoded in-tree score used for matched-rate work. QUALITY_SUMMARY_SCHEMA = "jpxl.codec-quality-summary/2" -QUALITY_TRACE_SCHEMA = "jpxl.quality-trace/1" +# Accepted quality-trace schemas, preferred first: /2 adds the whole-search +# `work` block and the shadow `prediction` block; /1 traces predate them and +# stay readable. +QUALITY_TRACE_SCHEMAS = ("jpxl.quality-trace/2", "jpxl.quality-trace/1") METRIC_VARIATION_SCHEMA = "jpxl.metric-variation/1" METRIC_VARIATION_INPUT_SCHEMA = "jpxl.metric-variation-input/1" RISK_INPUT_SCHEMA = "jpxl.edge-risk-input/1" @@ -662,12 +665,21 @@ def quality_encode_command( threads: int, effort: str, ) -> list[str]: - """The perceptual VarDCT encode: a minimum SSIMULACRA2 target of ``score``.""" + """The perceptual VarDCT encode: a minimum SSIMULACRA2 target of ``score``. + + The harness opts into ``--quality-fallback best-effort``: by default an + unmet target now exits 1 and writes nothing, but a curve point wants the + under-target stream with its true ``saturated_top`` / + ``under_target_work_cap`` status and measured score, which is what this + mode emits. + """ return [ str(binary), "encode", "--quality", f"{score:.4f}", + "--quality-fallback", + "best-effort", "--effort", effort, "--threads", @@ -706,16 +718,16 @@ def find(name: str, cast: Any) -> Any: def read_quality_trace(path: Path) -> dict[str, Any]: - """Return the ``jpxl.quality-trace/1`` object written to a trace file.""" + """Return the ``jpxl.quality-trace/*`` object written to a trace file.""" records = [ json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip() ] for record in records: - if record.get("schema") == QUALITY_TRACE_SCHEMA: + if record.get("schema") in QUALITY_TRACE_SCHEMAS: return record - raise HarnessError(f"no {QUALITY_TRACE_SCHEMA} record in trace file {path}") + raise HarnessError(f"no {QUALITY_TRACE_SCHEMAS[0]} record in trace file {path}") def curve_point( diff --git a/JPXL/tools/make-quality-guard-fixtures.py b/JPXL/tools/make-quality-guard-fixtures.py index a9690702..e7cea413 100644 --- a/JPXL/tools/make-quality-guard-fixtures.py +++ b/JPXL/tools/make-quality-guard-fixtures.py @@ -40,6 +40,8 @@ import hashlib import io import json +import os +import re import struct import sys import zlib @@ -173,6 +175,29 @@ def to_u8(x: np.ndarray) -> np.ndarray: return np.clip(np.rint(x), 0, 255).astype(np.uint8) +def hsv_to_rgb(h: np.ndarray, s: np.ndarray, v: np.ndarray) -> np.ndarray: + """Vectorised HSV->RGB (all inputs/outputs in [0, 1]); deterministic. + + Broadcasts scalar ``s``/``v`` against an array ``h``; returns an ``...x3`` + float array. Hand-rolled (no colorsys loop, no third-party colour dep) so + the raster stays bit-identical across environments. + """ + h = np.mod(np.asarray(h, dtype=np.float64), 1.0) + s = np.asarray(s, dtype=np.float64) * np.ones_like(h) + v = np.asarray(v, dtype=np.float64) * np.ones_like(h) + i = np.floor(h * 6.0).astype(np.int64) + f = h * 6.0 - i + p = v * (1.0 - s) + q = v * (1.0 - f * s) + t = v * (1.0 - (1.0 - f) * s) + i = np.mod(i, 6) + cond = [i == k for k in range(6)] + r = np.select(cond, [v, q, p, p, t, v]) + g = np.select(cond, [t, v, v, q, p, p]) + b = np.select(cond, [p, p, t, v, v, q]) + return np.stack([r, g, b], axis=-1) + + def load_rgb(rel_path: str) -> np.ndarray: with Image.open(REPO_ROOT / rel_path) as im: return np.asarray(im.convert("RGB"), dtype=np.uint8).copy() @@ -422,6 +447,289 @@ def build_saturated(width: int, height: int, seed: int, soft: bool) -> np.ndarra return arr +# --------------------------------------------------------------------------- # +# Saturated content builders (banding/clipping-stress coverage; memo §10). Each +# function is a distinct content *process* — not one generator re-seeded — so a +# fresh fixture id is a genuinely independent source family. +# --------------------------------------------------------------------------- # + + +def build_sat_hue_wheel(width: int, height: int, seed: int, sectors: int) -> np.ndarray: + """Fully-saturated angular pie: hue quantised into ``sectors`` hard slices.""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + cx, cy = (width - 1) / 2.0, (height - 1) / 2.0 + ang = (np.arctan2(yy - cy, xx - cx) / (2 * np.pi)) % 1.0 + hue = np.floor(ang * sectors) / sectors + rgb = hsv_to_rgb(hue, np.ones_like(hue), np.ones_like(hue)) + _ = seed + return to_u8(rgb * 255.0) + + +def build_sat_hue_bands(width: int, height: int, seed: int, angle_deg: float, bands: int) -> np.ndarray: + """Angled stripes cycling through fully-saturated hues (sharp band edges).""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + theta = np.deg2rad(angle_deg) + proj = xx * np.cos(theta) + yy * np.sin(theta) + t = (proj - proj.min()) / max(1e-9, proj.max() - proj.min()) + hue = np.floor(t * bands) / bands + rgb = hsv_to_rgb(hue, np.ones_like(hue), np.ones_like(hue)) + _ = seed + return to_u8(rgb * 255.0) + + +def build_sat_chroma_ramp(width: int, height: int, seed: int, hue: float, axis: str) -> np.ndarray: + """Neutral-to-fully-saturated chroma ramp along one axis (near-clipping end).""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + if axis == "h": + t = xx / max(1, width - 1) + elif axis == "v": + t = yy / max(1, height - 1) + else: # diagonal + t = (xx + yy) / max(1, (width - 1) + (height - 1)) + rgb = hsv_to_rgb(np.full_like(t, hue), t, np.ones_like(t)) + _ = seed + return to_u8(rgb * 255.0) + + +def build_sat_texture_edges(width: int, height: int, seed: int, cell: int) -> np.ndarray: + """High-frequency saturated checkerboard of complementary hues + hard blocks.""" + rng = np.random.default_rng(seed) + yy, xx = np.mgrid[0:height, 0:width] + checker = ((xx // cell) + (yy // cell)) % 2 + h0 = float(rng.random()) + h1 = (h0 + 0.5) % 1.0 + ones = np.ones((height, width)) + a = hsv_to_rgb(np.full((height, width), h0), ones, ones) + b = hsv_to_rgb(np.full((height, width), h1), ones, ones) + rgb = np.where(checker[..., None] == 0, a, b) + for _ in range(6): + x0 = int(rng.integers(0, max(1, width - cell))) + y0 = int(rng.integers(0, max(1, height - cell))) + w = int(rng.integers(cell, cell * 3)) + hh = int(rng.integers(cell, cell * 3)) + col = hsv_to_rgb(np.array(float(rng.random())), np.array(1.0), np.array(1.0)) + rgb[y0 : y0 + hh, x0 : x0 + w] = col + return to_u8(rgb * 255.0) + + +def build_sat_noise(width: int, height: int, seed: int) -> np.ndarray: + """Per-pixel random fully-saturated hue (saturated high-entropy texture).""" + rng = np.random.default_rng(seed) + hue = rng.random((height, width)) + ones = np.ones((height, width)) + rgb = hsv_to_rgb(hue, ones, ones) + return to_u8(rgb * 255.0) + + +def build_sat_solid_tile(width: int, height: int, color: tuple[int, int, int]) -> np.ndarray: + """A single saturated colour — trivially compresses to a sub-kilobyte stream.""" + arr = np.empty((height, width, 3), dtype=np.uint8) + arr[:, :] = color + return arr + + +def build_sat_block_tile(width: int, height: int, colors: list[tuple[int, int, int]]) -> np.ndarray: + """A few saturated colour blocks (still near-trivially / sub-kilobyte codeable).""" + arr = np.zeros((height, width, 3), dtype=np.uint8) + n = len(colors) + cols = 2 + rows = (n + 1) // 2 + cw, ch = width // cols, height // rows + for idx, color in enumerate(colors): + r, c = divmod(idx, cols) + y1 = (r + 1) * ch if r < rows - 1 else height + x1 = (c + 1) * cw if c < cols - 1 else width + arr[r * ch : y1, c * cw : x1] = color + return arr + + +def build_sat_rings(width: int, height: int, seed: int, ring_count: int) -> np.ndarray: + """Concentric fully-saturated rings, hue stepped by radius (hard ring edges).""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + cx, cy = (width - 1) / 2.0, (height - 1) / 2.0 + d = np.sqrt((xx - cx) ** 2 + (yy - cy) ** 2) + ring = np.floor((d / max(1e-9, d.max())) * ring_count).astype(np.int64) + hue = (ring / max(1, ring_count)) % 1.0 + rgb = hsv_to_rgb(hue, np.ones_like(d), np.ones_like(d)) + _ = seed + return to_u8(rgb * 255.0) + + +def build_sat_out_of_gamut(width: int, height: int, seed: int) -> np.ndarray: + """Channels swung well beyond [0,1] then hard-clipped: broad clipped plateaus.""" + rng = np.random.default_rng(seed) + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + tx = xx / max(1, width - 1) + ty = yy / max(1, height - 1) + phase = float(rng.random()) + r = 2.2 * np.sin(2 * np.pi * (tx + 0.05 * phase)) + g = 2.2 * np.cos(2 * np.pi * ty) + b = 2.2 * np.sin(2 * np.pi * (tx + ty)) + rgb = np.clip(np.stack([r, g, b], axis=-1), 0.0, 1.0) + return to_u8(rgb * 255.0) + + +def build_sat_voronoi(width: int, height: int, seed: int, cells: int) -> np.ndarray: + """Voronoi partition, each cell a random fully-saturated hue with hard seams.""" + rng = np.random.default_rng(seed) + pts = rng.random((cells, 2)) * np.array([width, height]) + hues = rng.random(cells) + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + best = np.zeros((height, width), dtype=np.int64) + bestd = np.full((height, width), np.inf) + for i in range(cells): + d = (xx - pts[i, 0]) ** 2 + (yy - pts[i, 1]) ** 2 + m = d < bestd + best[m] = i + bestd[m] = d[m] + hue = hues[best] + rgb = hsv_to_rgb(hue, np.ones_like(hue), np.ones_like(hue)) + return to_u8(rgb * 255.0) + + +# --------------------------------------------------------------------------- # +# Gradient / banding-stress builders (memo §10). Distinct processes: angled +# linear ramps, shallow luma/chroma sweeps, dithered ramps, radial sky glows, +# multi-stop, bilinear, conic and diamond fields. +# --------------------------------------------------------------------------- # + + +def build_grad_angled( + width: int, height: int, seed: int, angle_deg: float, + c0: tuple[float, float, float], c1: tuple[float, float, float], +) -> np.ndarray: + """Two-colour linear ramp at an arbitrary angle.""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + theta = np.deg2rad(angle_deg) + proj = xx * np.cos(theta) + yy * np.sin(theta) + t = (proj - proj.min()) / max(1e-9, proj.max() - proj.min()) + a = np.asarray(c0, dtype=np.float64) + b = np.asarray(c1, dtype=np.float64) + rgb = (1 - t)[..., None] * a + t[..., None] * b + _ = seed + return to_u8(rgb * 255.0) + + +def build_grad_luma_shallow( + width: int, height: int, seed: int, base: float, span: float, angle_deg: float +) -> np.ndarray: + """Grayscale ramp over a narrow code-value span (severe banding stress).""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + theta = np.deg2rad(angle_deg) + proj = xx * np.cos(theta) + yy * np.sin(theta) + t = (proj - proj.min()) / max(1e-9, proj.max() - proj.min()) + g = to_u8(base + span * t) + _ = seed + return np.repeat(g[..., None], 3, axis=2) + + +def build_grad_chroma_shallow( + width: int, height: int, seed: int, hue: float, angle_deg: float +) -> np.ndarray: + """Constant-luma, shallow saturation sweep — chroma banding stress.""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + theta = np.deg2rad(angle_deg) + proj = xx * np.cos(theta) + yy * np.sin(theta) + t = (proj - proj.min()) / max(1e-9, proj.max() - proj.min()) + sat = 0.06 + 0.10 * t + rgb = hsv_to_rgb(np.full_like(t, hue), sat, np.full_like(t, 0.65)) + _ = seed + return to_u8(rgb * 255.0) + + +def build_grad_dither(width: int, height: int, seed: int, angle_deg: float, amp: float) -> np.ndarray: + """Smooth ramp plus seeded additive dither noise.""" + rng = np.random.default_rng(seed) + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + theta = np.deg2rad(angle_deg) + proj = xx * np.cos(theta) + yy * np.sin(theta) + t = (proj - proj.min()) / max(1e-9, proj.max() - proj.min()) + base = 0.30 + 0.30 * t + noise = rng.normal(0.0, amp / 255.0, size=(height, width)) + v = base + noise + rgb = np.clip(np.stack([v, v, v * 0.98], axis=-1), 0.0, 1.0) + return to_u8(rgb * 255.0) + + +def build_grad_radial_sky( + width: int, height: int, seed: int, cx_frac: float, cy_frac: float +) -> np.ndarray: + """Off-centre warm-to-cool radial sky glow with faint seeded noise.""" + rng = np.random.default_rng(seed) + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + cx, cy = cx_frac * width, cy_frac * height + d = np.sqrt((xx - cx) ** 2 + (yy - cy) ** 2) + t = d / max(1e-9, d.max()) + r = 0.95 - 0.55 * t + g = 0.85 - 0.30 * t + b = 0.60 + 0.35 * t + noise = rng.normal(0.0, 1.0 / 255.0, size=(height, width)) + rgb = np.clip(np.stack([r + noise, g + noise, b + noise], axis=-1), 0.0, 1.0) + return to_u8(rgb * 255.0) + + +def build_grad_sunset_multistop(width: int, height: int, seed: int, angle_deg: float) -> np.ndarray: + """Five-stop sunset gradient along an angle (piecewise-linear colour stops).""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + theta = np.deg2rad(angle_deg) + proj = xx * np.cos(theta) + yy * np.sin(theta) + t = (proj - proj.min()) / max(1e-9, proj.max() - proj.min()) + stops_t = np.array([0.0, 0.3, 0.55, 0.75, 1.0]) + stops_c = np.array([ + [0.05, 0.05, 0.20], + [0.35, 0.10, 0.30], + [0.85, 0.35, 0.25], + [0.98, 0.70, 0.35], + [1.00, 0.92, 0.70], + ]) + r = np.interp(t, stops_t, stops_c[:, 0]) + g = np.interp(t, stops_t, stops_c[:, 1]) + b = np.interp(t, stops_t, stops_c[:, 2]) + rgb = np.stack([r, g, b], axis=-1) + _ = seed + return to_u8(rgb * 255.0) + + +def build_grad_bilinear( + width: int, height: int, seed: int, + corners: list[tuple[float, float, float]], +) -> np.ndarray: + """Four-corner bilinear colour interpolation.""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + u = xx / max(1, width - 1) + v = yy / max(1, height - 1) + tl, tr, bl, br = (np.asarray(c, dtype=np.float64) for c in corners) + top = (1 - u)[..., None] * tl + u[..., None] * tr + bot = (1 - u)[..., None] * bl + u[..., None] * br + rgb = (1 - v)[..., None] * top + v[..., None] * bot + _ = seed + return to_u8(rgb * 255.0) + + +def build_grad_conic(width: int, height: int, seed: int) -> np.ndarray: + """Low-saturation conic (angular) hue sweep — a gentle pastel chroma gradient.""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + cx, cy = (width - 1) / 2.0, (height - 1) / 2.0 + ang = (np.arctan2(yy - cy, xx - cx) / (2 * np.pi)) % 1.0 + rgb = hsv_to_rgb(ang, np.full_like(ang, 0.25), np.full_like(ang, 0.85)) + _ = seed + return to_u8(rgb * 255.0) + + +def build_grad_diamond(width: int, height: int, seed: int) -> np.ndarray: + """L1 (diamond) distance gradient from centre.""" + yy, xx = np.mgrid[0:height, 0:width].astype(np.float64) + cx, cy = (width - 1) / 2.0, (height - 1) / 2.0 + d = np.abs(xx - cx) + np.abs(yy - cy) + t = d / max(1e-9, d.max()) + r = 0.20 + 0.70 * t + g = 0.60 - 0.20 * t + b = 0.80 - 0.55 * t + rgb = np.clip(np.stack([r, g, b], axis=-1), 0.0, 1.0) + _ = seed + return to_u8(rgb * 255.0) + + def crop(arr: np.ndarray, w: int, h: int, ox: int, oy: int) -> np.ndarray: H, W, _ = arr.shape ox = max(0, min(ox, W - w)) @@ -500,6 +808,12 @@ def registry() -> list[Fixture]: fx.append(_synth("text-screenshot-holdout-1600x900", "text-screenshot", "holdout", "synthetic/text-screenshot", lambda: build_text(1600, 900, 19001, dark=False), 19001, "Holdout paragraphs on white, fresh seed")) + fx.append(_synth("text-screenshot-white2-1280x800", "text-screenshot", "calibration", + "synthetic/text-screenshot", lambda: build_text(1280, 800, 1005, dark=False), + 1005, "Second white-background paragraph layout, fresh seed")) + fx.append(_synth("text-screenshot-ui2-1600x1000", "text-screenshot", "development", + "synthetic/text-screenshot", lambda: build_text_ui(1600, 1000, 1006), + 1006, "Second UI mock layout, fresh seed (UI class in both splits)")) # ---- line-art (synthetic) ---------------------------------------------- # fx.append(_synth("line-art-aa-shapes-1024x1024", "line-art", "calibration", @@ -517,6 +831,9 @@ def registry() -> list[Fixture]: fx.append(_synth("line-art-holdout-900x900", "line-art", "holdout", "synthetic/line-art", lambda: build_line_art(900, 900, 29001, anti_alias=True), 29001, "Holdout anti-aliased vector shapes, fresh seed")) + fx.append(_synth("line-art-hatch2-640x640", "line-art", "development", + "synthetic/line-art", lambda: build_hatch(640, 640, 2005), + 2005, "Second hatch-pattern layout, fresh seed (hatch in both splits)")) # ---- gradient (synthetic) ---------------------------------------------- # fx.append(_synth("gradient-horizontal-1024x512", "gradient", "calibration", @@ -537,6 +854,95 @@ def registry() -> list[Fixture]: fx.append(_synth("gradient-holdout-radial-700x700", "gradient", "holdout", "synthetic/gradient", lambda: build_gradient(700, 700, "radial", 39001), 39001, "Holdout radial ramp, fresh seed")) + fx.append(_synth("gradient-sky-noise2-1024x512", "gradient", "calibration", + "synthetic/gradient", lambda: build_gradient(1024, 512, "sky", 3006), + 3006, "Second sky-like noisy gradient, fresh seed (banding stress in both splits)")) + fx.append(_synth("gradient-sky-noise3-800x600", "gradient", "development", + "synthetic/gradient", lambda: build_gradient(800, 600, "sky", 3007), + 3007, "Third sky-like noisy gradient, fresh seed")) + + # ---- gradient/banding-stress coverage expansion (memo §10): distinct + # generating processes, each a fresh family. ------------------------------ # + # Angled two-colour linear ramps. + fx.append(_synth("gradient-angled-30-1024x512", "gradient", "calibration", + "synthetic/gradient", + lambda: build_grad_angled(1024, 512, 3100, 30.0, (0.05, 0.10, 0.35), (0.95, 0.80, 0.30)), + 3100, "Two-colour linear ramp at 30 degrees")) + fx.append(_synth("gradient-angled-75-800x600", "gradient", "development", + "synthetic/gradient", + lambda: build_grad_angled(800, 600, 3101, 75.0, (0.10, 0.30, 0.15), (0.90, 0.40, 0.70)), + 3101, "Two-colour linear ramp at 75 degrees")) + fx.append(_synth("gradient-angled-150-holdout-640x640", "gradient", "holdout", + "synthetic/gradient", + lambda: build_grad_angled(640, 640, 39100, 150.0, (0.20, 0.20, 0.60), (0.85, 0.85, 0.20)), + 39100, "Holdout two-colour linear ramp at 150 degrees, fresh seed")) + # Shallow luma ramps over a narrow code-value span (banding stress). + fx.append(_synth("gradient-luma-shallow-h-1024x512", "gradient", "calibration", + "synthetic/gradient", + lambda: build_grad_luma_shallow(1024, 512, 3102, 96.0, 16.0, 0.0), + 3102, "Horizontal grayscale ramp over 16 code values (banding stress)")) + fx.append(_synth("gradient-luma-shallow-v-800x600", "gradient", "development", + "synthetic/gradient", + lambda: build_grad_luma_shallow(800, 600, 3103, 40.0, 24.0, 90.0), + 3103, "Vertical grayscale ramp over 24 code values (banding stress)")) + # Shallow chroma sweeps (constant luma). + fx.append(_synth("gradient-chroma-shallow-red-1024x512", "gradient", "calibration", + "synthetic/gradient", + lambda: build_grad_chroma_shallow(1024, 512, 3104, 0.02, 0.0), + 3104, "Shallow horizontal red-chroma sweep at constant luma")) + fx.append(_synth("gradient-chroma-shallow-blue-800x600", "gradient", "development", + "synthetic/gradient", + lambda: build_grad_chroma_shallow(800, 600, 3105, 0.62, 90.0), + 3105, "Shallow vertical blue-chroma sweep at constant luma")) + fx.append(_synth("gradient-chroma-shallow-green-holdout-640x640", "gradient", "holdout", + "synthetic/gradient", + lambda: build_grad_chroma_shallow(640, 640, 39101, 0.33, 45.0), + 39101, "Holdout shallow diagonal green-chroma sweep, fresh seed")) + # Dithered ramps. + fx.append(_synth("gradient-dither-lowamp-1024x512", "gradient", "calibration", + "synthetic/gradient", + lambda: build_grad_dither(1024, 512, 3106, 0.0, 1.0), + 3106, "Smooth horizontal ramp plus low-amplitude seeded dither")) + fx.append(_synth("gradient-dither-highamp-800x600", "gradient", "development", + "synthetic/gradient", + lambda: build_grad_dither(800, 600, 3107, 60.0, 4.0), + 3107, "Angled ramp plus higher-amplitude seeded dither")) + # Radial sky glows. + fx.append(_synth("gradient-radial-sky-glow-1024x768", "gradient", "calibration", + "synthetic/gradient", + lambda: build_grad_radial_sky(1024, 768, 3108, 0.5, 0.15), + 3108, "Centred warm-to-cool radial sky glow with faint noise")) + fx.append(_synth("gradient-radial-sky-offcenter-800x600", "gradient", "development", + "synthetic/gradient", + lambda: build_grad_radial_sky(800, 600, 3109, 0.2, 0.8), + 3109, "Off-centre radial sky glow with faint noise")) + # Multi-stop sunset. + fx.append(_synth("gradient-sunset-multistop-1024x512", "gradient", "calibration", + "synthetic/gradient", + lambda: build_grad_sunset_multistop(1024, 512, 3110, 90.0), + 3110, "Five-stop vertical sunset gradient")) + # Bilinear four-corner fields. + fx.append(_synth("gradient-bilinear-768x768", "gradient", "calibration", + "synthetic/gradient", + lambda: build_grad_bilinear(768, 768, 3111, [ + (0.10, 0.20, 0.60), (0.80, 0.30, 0.20), + (0.20, 0.70, 0.30), (0.90, 0.85, 0.40)]), + 3111, "Four-corner bilinear colour field")) + fx.append(_synth("gradient-bilinear-holdout-640x640", "gradient", "holdout", + "synthetic/gradient", + lambda: build_grad_bilinear(640, 640, 39102, [ + (0.60, 0.10, 0.40), (0.20, 0.60, 0.70), + (0.80, 0.80, 0.20), (0.10, 0.30, 0.55)]), + 39102, "Holdout four-corner bilinear colour field, fresh seed")) + # Conic and diamond fields. + fx.append(_synth("gradient-conic-720x720", "gradient", "calibration", + "synthetic/gradient", + lambda: build_grad_conic(720, 720, 3112), + 3112, "Low-saturation conic (angular) pastel hue sweep")) + fx.append(_synth("gradient-diamond-768x768", "gradient", "development", + "synthetic/gradient", + lambda: build_grad_diamond(768, 768, 3113), + 3113, "L1 diamond-distance colour gradient from centre")) # ---- saturated (synthetic + one photo crop) ---------------------------- # fx.append(_synth("saturated-primaries-hard-512x512", "saturated", "calibration", @@ -556,6 +962,84 @@ def registry() -> list[Fixture]: "test-set/test-set-4mp/20260606_203230_4mp.png", "centre-ish 512x512 crop at (x=900,y=700), sRGB 8-bit, no resampling")) + # ---- saturated coverage expansion (memo §10): distinct saturated/clipping + # content processes, each a fresh family. --------------------------------- # + # Quantised hue wheels. + fx.append(_synth("saturated-hue-wheel-512x512", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_hue_wheel(512, 512, 4100, 8), + 4100, "Fully-saturated 8-sector angular hue wheel (hard slice edges)")) + fx.append(_synth("saturated-hue-wheel-fine-384x384", "saturated", "development", + "synthetic/saturated", lambda: build_sat_hue_wheel(384, 384, 4101, 16), + 4101, "Fully-saturated 16-sector angular hue wheel")) + fx.append(_synth("saturated-hue-wheel-holdout-320x320", "saturated", "holdout", + "synthetic/saturated", lambda: build_sat_hue_wheel(320, 320, 49100, 12), + 49100, "Holdout 12-sector fully-saturated hue wheel, fresh seed")) + # Angled saturated hue bands. + fx.append(_synth("saturated-hue-bands-0deg-512x512", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_hue_bands(512, 512, 4102, 0.0, 12), + 4102, "Vertical fully-saturated hue bands (12 hues, hard edges)")) + fx.append(_synth("saturated-hue-bands-45deg-512x512", "saturated", "development", + "synthetic/saturated", lambda: build_sat_hue_bands(512, 512, 4103, 45.0, 12), + 4103, "Diagonal fully-saturated hue bands (12 hues)")) + fx.append(_synth("saturated-hue-bands-90deg-448x448", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_hue_bands(448, 448, 4104, 90.0, 16), + 4104, "Horizontal fully-saturated hue bands (16 hues)")) + # Neutral-to-saturated chroma ramps (near-clipping end). + fx.append(_synth("saturated-chroma-ramp-red-h-512x512", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_chroma_ramp(512, 512, 4105, 0.0, "h"), + 4105, "Horizontal gray-to-full-red chroma ramp (near-clipping)")) + fx.append(_synth("saturated-chroma-ramp-blue-v-512x512", "saturated", "development", + "synthetic/saturated", lambda: build_sat_chroma_ramp(512, 512, 4106, 0.62, "v"), + 4106, "Vertical gray-to-full-blue chroma ramp (near-clipping)")) + fx.append(_synth("saturated-chroma-ramp-green-diag-holdout-320x320", "saturated", "holdout", + "synthetic/saturated", lambda: build_sat_chroma_ramp(320, 320, 49101, 0.33, "d"), + 49101, "Holdout diagonal gray-to-full-green chroma ramp, fresh seed")) + # Saturated high-frequency texture and per-pixel noise. + fx.append(_synth("saturated-texture-checker-512x512", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_texture_edges(512, 512, 4107, 16), + 4107, "Complementary saturated checkerboard (16px cells) plus hard blocks")) + fx.append(_synth("saturated-texture-fine-384x384", "saturated", "development", + "synthetic/saturated", lambda: build_sat_texture_edges(384, 384, 4108, 6), + 4108, "Fine complementary saturated checkerboard (6px cells)")) + fx.append(_synth("saturated-noise-hue-384x384", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_noise(384, 384, 4109), + 4109, "Per-pixel random fully-saturated hue (high-entropy texture)")) + fx.append(_synth("saturated-noise-hue-holdout-320x320", "saturated", "holdout", + "synthetic/saturated", lambda: build_sat_noise(320, 320, 49102), + 49102, "Holdout per-pixel random saturated hue, fresh seed")) + # Sub-kilobyte-encodable saturated tiles (solid / few-block). + fx.append(_synth("saturated-solid-red-256x256", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_solid_tile(256, 256, (255, 0, 0)), + 4110, "Solid pure red tile (sub-kilobyte-encodable saturated fixture)")) + fx.append(_synth("saturated-solid-cyan-256x256", "saturated", "development", + "synthetic/saturated", lambda: build_sat_solid_tile(256, 256, (0, 255, 255)), + 4111, "Solid pure cyan tile (sub-kilobyte-encodable saturated fixture)")) + fx.append(_synth("saturated-solid-magenta-288x288", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_solid_tile(288, 288, (255, 0, 255)), + 4112, "Solid pure magenta tile (sub-kilobyte-encodable saturated fixture)")) + fx.append(_synth("saturated-block-holdout-256x256", "saturated", "holdout", + "synthetic/saturated", lambda: build_sat_block_tile(256, 256, [ + (255, 255, 0), (0, 0, 255), (0, 255, 0), (255, 0, 0)]), + 4113, "Four-block saturated tile (near-trivially codeable), holdout")) + # Concentric saturated rings. + fx.append(_synth("saturated-rings-512x512", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_rings(512, 512, 4114, 10), + 4114, "Concentric fully-saturated rings (hue stepped by radius)")) + fx.append(_synth("saturated-rings-400x400", "saturated", "development", + "synthetic/saturated", lambda: build_sat_rings(400, 400, 4115, 16), + 4115, "Concentric fully-saturated rings, finer ring spacing")) + # Out-of-gamut hard-clip plateaus. + fx.append(_synth("saturated-out-of-gamut-512x512", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_out_of_gamut(512, 512, 4116), + 4116, "Channels swung beyond [0,1] then hard-clipped (broad clipped plateaus)")) + fx.append(_synth("saturated-out-of-gamut-448x448", "saturated", "development", + "synthetic/saturated", lambda: build_sat_out_of_gamut(448, 448, 4117), + 4117, "Second out-of-gamut hard-clip field, fresh seed")) + # Saturated Voronoi partition. + fx.append(_synth("saturated-voronoi-512x512", "saturated", "calibration", + "synthetic/saturated", lambda: build_sat_voronoi(512, 512, 4118, 28), + 4118, "28-cell saturated Voronoi partition (hard cell seams)")) + # ---- gradient/line-art done; tiny crops (exempt from 256x256) ---------- # tiny_sizes = [(7, 7), (8, 8), (16, 16), (33, 20), (64, 64)] # Calibration tiny crops from calibration scene 20240501_110934. @@ -580,7 +1064,8 @@ def registry() -> list[Fixture]: "64x64 crop at (x=1200,y=900), sRGB 8-bit")) _ = tiny_sizes # documented full set (7x7,8x8,16x16,33x20,64x64) spread across splits - # ---- noise-lowlight (derived, development scene only) ------------------- # + # ---- noise-lowlight (derived; one family per capture, split follows the + # capture so the class exists in both calibration and development) ------- # fx.append(_derived( "noise-lowlight-184356-1024x768", "noise-lowlight", "development", "derived/noise-lowlight", @@ -588,6 +1073,27 @@ def registry() -> list[Fixture]: "test-set/20240502_184356.png", "linear-light x0.25 darken, then seeded Poisson(scale=500)+Gaussian(sigma=0.006) noise, re-encode sRGB 8-bit", seed=5001)) + fx.append(_derived( + "noise-lowlight-151356-1024x768", "noise-lowlight", "calibration", + "derived/noise-lowlight", + lambda: build_lowlight(load_rgb("test-set/20240502_151356.png"), 5002), + "test-set/20240502_151356.png", + "linear-light x0.25 darken, then seeded Poisson(scale=500)+Gaussian(sigma=0.006) noise, re-encode sRGB 8-bit", + seed=5002)) + fx.append(_derived( + "noise-lowlight-105759-1024x768", "noise-lowlight", "development", + "derived/noise-lowlight", + lambda: build_lowlight(load_rgb("test-set/20240503_105759.png"), 5003), + "test-set/20240503_105759.png", + "linear-light x0.25 darken, then seeded Poisson(scale=500)+Gaussian(sigma=0.006) noise, re-encode sRGB 8-bit", + seed=5003)) + fx.append(_derived( + "noise-lowlight-110934-1024x768", "noise-lowlight", "calibration", + "derived/noise-lowlight", + lambda: build_lowlight(load_rgb("test-set/20240501_110934.png"), 5004), + "test-set/20240501_110934.png", + "linear-light x0.25 darken, then seeded Poisson(scale=500)+Gaussian(sigma=0.006) noise, re-encode sRGB 8-bit", + seed=5004)) # ---- grayscale (derived) ----------------------------------------------- # fx.append(_derived( @@ -596,6 +1102,12 @@ def registry() -> list[Fixture]: lambda: build_grayscale(load_rgb("test-set/20240502_192515.png")), "test-set/20240502_192515.png", "Rec.709 linear-light luma, re-encode sRGB, replicated to R=G=B")) + fx.append(_derived( + "grayscale-scene-151800-1024x768", "grayscale", "calibration", + "derived/grayscale", + lambda: build_grayscale(load_rgb("test-set/20240502_151800.png")), + "test-set/20240502_151800.png", + "Rec.709 linear-light luma, re-encode sRGB, replicated to R=G=B")) fx.append(_derived( "grayscale-photo-203230-crop-1024x1024", "grayscale", "calibration", "derived/grayscale", @@ -683,9 +1195,43 @@ def sidecar_for(fx: Fixture, ppm_sha: str, png_sha: str | None) -> dict[str, Any return doc +CAPTURE_STEM_RE = re.compile(r"(20\d{6}_\d{6})") + + +def source_capture_id(fx: Fixture) -> str | None: + """The camera-capture stem a fixture ultimately comes from, or ``None``. + + Parsed from the parent path's basename (``20260606_203230_4mp.png`` and + ``20260606_203230_result.png`` are one capture), so every crop, resolution + and colour variant of one photograph shares the id. + """ + if fx.parent is None: + return None + m = CAPTURE_STEM_RE.search(os.path.basename(fx.parent["path"])) + return m.group(1) if m else None + + +def family_fields(fx: Fixture) -> dict[str, Any]: + """The split-hygiene fields of one fixture. + + ``family_id`` groups every derivative of one independent source: the + capture stem for photographic sources and their derivatives, the fixture's + own id for synthetic fixtures (fresh seeds are independent families by + design). All members of a family must share one split — ``cmd_build`` + enforces it — so training/holdout separation is mechanically auditable. + """ + capture = source_capture_id(fx) + return { + "family_id": capture if capture is not None else fx.fixture_id, + "variant_id": fx.fixture_id, + "generator_family": fx.klass if fx.kind in ("synthetic", "derived") else None, + "source_capture_id": capture, + } + + def manifest_entry(fx: Fixture, ppm_sha: str) -> dict[str, Any]: w, h, depth = fx.dims() - return { + entry = { "id": fx.fixture_id, "path": fx.ppm_rel(), "sha256": ppm_sha, @@ -698,6 +1244,8 @@ def manifest_entry(fx: Fixture, ppm_sha: str) -> dict[str, Any]: "provenance": fx.provenance, "sidecar": fx.sidecar_rel(), } + entry.update(family_fields(fx)) + return entry def realise(fx: Fixture) -> tuple[bytes, bytes | None]: @@ -721,6 +1269,16 @@ def cmd_build(check: bool) -> int: if dupes: print(f"ERROR: duplicate fixture ids: {sorted(dupes)}", file=sys.stderr) return 2 + # Family split-hygiene guard: every derivative of one source must stay in + # one split, or family leakage silently flatters any model trained on the + # calibration/development splits. + family_splits: dict[str, set[str]] = {} + for f in fixtures: + family_splits.setdefault(family_fields(f)["family_id"], set()).add(f.split) + leaking = {fam: sorted(s) for fam, s in family_splits.items() if len(s) > 1} + if leaking: + print(f"ERROR: image families cross splits: {leaking}", file=sys.stderr) + return 2 manifest_images: list[dict[str, Any]] = [] ppm_mismatch = 0 diff --git a/JPXL/tools/one_shot_promotion_ab.py b/JPXL/tools/one_shot_promotion_ab.py new file mode 100644 index 00000000..1a4bcd55 --- /dev/null +++ b/JPXL/tools/one_shot_promotion_ab.py @@ -0,0 +1,167 @@ +#!/usr/bin/env python3 +"""A/B the one-shot controller build against the default build (promotion +screen for the one-shot program, memo section 11). + +Runs `jpxl encode --quality T` for every image of the requested manifest +splits at every target, once per binary, capturing the report line and the +`jpxl.quality-trace/2` record, then prints the promotion comparison: + +* floor: any achieved < requested on either arm; +* bytes: per-cell and geometric-mean byte ratio one-shot/default; +* work: pixel plans / reconstructions / metric evaluations per arm; +* wall: per-cell encode wall and the ratio. + +Both arms run serially and interleaved (A, B, A, B ...) on the same images +so drift hits both equally. Standard library only. +""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import subprocess +import sys +import tempfile +import time + + +def parse_line(stdout: str) -> dict: + out = {} + for token in stdout.split(): + if "=" in token: + key, value = token.split("=", 1) + out[key] = value + return out + + +def run_cell(binary: str, image_path: str, target: float, threads: int, trace_path: str) -> dict: + env = dict(os.environ, JPXL_QUALITY_TRACE=trace_path) + out_path = os.path.join(tempfile.gettempdir(), "oneshot-ab.jxl") + start = time.perf_counter() + completed = subprocess.run( + [ + binary, + "encode", + "--quality", + f"{target}", + "--threads", + str(threads), + image_path, + out_path, + ], + capture_output=True, + text=True, + env=env, + ) + wall = time.perf_counter() - start + if completed.returncode != 0: + return {"error": completed.stderr.strip()[-160:], "wall": wall} + line = parse_line(completed.stdout) + trace = {} + try: + with open(trace_path, encoding="utf-8") as fh: + for raw in fh: + record = json.loads(raw) + if record.get("schema", "").startswith("jpxl.quality-trace/"): + trace = record + except OSError: + pass + finally: + try: + os.remove(trace_path) + except OSError: + pass + return { + "achieved": float(line.get("achieved", "nan")), + "bytes": int(line.get("bytes", "0")), + "status": line.get("status", "?"), + "probes": int(line.get("probes", "0")), + "prices": int(line.get("prices", "0")), + "work": trace.get("work"), + "predicted_rung": trace.get("predicted_rung"), + "wall": wall, + } + + +def main(argv: list[str]) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--manifest", required=True) + parser.add_argument("--splits", nargs="+", default=["holdout"]) + parser.add_argument("--default-binary", required=True) + parser.add_argument("--one-shot-binary", required=True) + parser.add_argument("--targets", nargs="+", type=float, + default=[30.0, 50.0, 70.0, 80.0, 85.0, 90.0, 95.0]) + parser.add_argument("--threads", type=int, default=4) + parser.add_argument("--max-pixels", type=int, default=None, + help="skip images above this pixel count") + parser.add_argument("--output", required=True) + args = parser.parse_args(argv) + + with open(args.manifest, encoding="utf-8") as fh: + manifest = json.load(fh) + base_dir = os.path.dirname(os.path.abspath(args.manifest)) + images = [im for im in manifest["images"] if im["split"] in set(args.splits)] + + rows = [] + for image in images: + path = os.path.join(base_dir, image["path"]) + with open(path, "rb") as fh: + magic = fh.readline().strip() + dims = fh.readline().split() + if magic != b"P6": + continue + width, height = int(dims[0]), int(dims[1]) + if min(width, height) < 64: + continue + if args.max_pixels and width * height > args.max_pixels: + print(f"skip (>{args.max_pixels}px): {image['id']}") + continue + for target in args.targets: + cell = {"image_id": image["id"], "class": image.get("class"), + "pixels": width * height, "target": target} + trace = os.path.join(tempfile.gettempdir(), "oneshot-ab-trace.jsonl") + cell["default"] = run_cell(args.default_binary, path, target, args.threads, trace) + cell["one_shot"] = run_cell(args.one_shot_binary, path, target, args.threads, trace) + rows.append(cell) + d, o = cell["default"], cell["one_shot"] + print( + f"{image['id']} t={target}: bytes {d.get('bytes')}->{o.get('bytes')} " + f"achieved {d.get('achieved'):.2f}->{o.get('achieved'):.2f} " + f"recon {d.get('work', {}).get('reconstructions')}->" + f"{o.get('work', {}).get('reconstructions')} " + f"wall {d.get('wall'):.2f}s->{o.get('wall'):.2f}s", + flush=True, + ) + + ok = [r for r in rows if "error" not in r["default"] and "error" not in r["one_shot"]] + floor_default = [r for r in ok if r["default"]["achieved"] < r["target"]] + floor_one_shot = [r for r in ok if r["one_shot"]["achieved"] < r["target"]] + ratios = [r["one_shot"]["bytes"] / r["default"]["bytes"] for r in ok if r["default"]["bytes"]] + geomean = math.exp(sum(math.log(x) for x in ratios) / len(ratios)) if ratios else None + recon = lambda arm: sum(r[arm]["work"]["reconstructions"] for r in ok if r[arm].get("work")) + wall_ratio = [r["one_shot"]["wall"] / r["default"]["wall"] for r in ok] + summary = { + "cells": len(rows), + "compared": len(ok), + "floor_violations_default": len(floor_default), + "floor_violations_one_shot": len(floor_one_shot), + "byte_geomean_one_shot_over_default": geomean, + "worst_cell_byte_ratio": max(ratios) if ratios else None, + "total_reconstructions_default": recon("default"), + "total_reconstructions_one_shot": recon("one_shot"), + "wall_geomean_ratio": ( + math.exp(sum(math.log(x) for x in wall_ratio) / len(wall_ratio)) + if wall_ratio + else None + ), + } + with open(args.output, "w", encoding="utf-8", newline="\n") as fh: + fh.write(json.dumps({"summary": summary, "rows": rows}, indent=1) + "\n") + print(json.dumps(summary, indent=2)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:])) diff --git a/JPXL/tools/oracle-bin/PINNED_REVISIONS.txt b/JPXL/tools/oracle-bin/PINNED_REVISIONS.txt index d283e848..2744808d 100644 --- a/JPXL/tools/oracle-bin/PINNED_REVISIONS.txt +++ b/JPXL/tools/oracle-bin/PINNED_REVISIONS.txt @@ -1,18 +1,18 @@ # JPXL oracle provenance -# Generated by Windows MinGW rebuild -- do not edit by hand. +# Generated by tools/setup-oracles.sh -- do not edit by hand. # Oracles are black boxes: their source is never read. -date: 2026-08-05T08:23:38Z -host: Windows Microsoft Windows NT 10.0.26200.0 +date: 2026-08-25T07:40:41Z +host: Linux 6.17.0-41-generic x86_64 [libjxl] -path: D:/Rust-projects/jpegXL-rs/libjxl +path: /mnt/Samsung980_1TB/Rust-projects/jpegXL-rs/libjxl revision: 196a43d996aa6ed33ebf98812a7c6d43b2b6d01b -describe: v0.12-snapshot-2-g196a43d9-dirty -cmake_flags: -DCMAKE_BUILD_TYPE=Release -DBUILD_TESTING=OFF -DJPEGXL_ENABLE_BENCHMARK=OFF -DJPEGXL_ENABLE_EXAMPLES=OFF -DJPEGXL_ENABLE_MANPAGES=OFF -DJPEGXL_ENABLE_PLUGINS=OFF -DJPEGXL_ENABLE_DOXYGEN=OFF -DJPEGXL_ENABLE_JNI=OFF -DJPEGXL_ENABLE_SJPEG=OFF -DJPEGXL_ENABLE_OPENEXR=OFF -DBUILD_SHARED_LIBS=OFF -G Ninja (MinGW build-win) -djxl_version: djxl v0.13.0 196a43d9 [_AVX2_,SSE4,SSE2] {GNU 16.1.0} -cjxl_version: cjxl v0.13.0 196a43d9 [_AVX2_,SSE4,SSE2] {GNU 16.1.0} +describe: v0.12-snapshot-2-g196a43d9 +cmake_flags: -DCMAKE_BUILD_TYPE=Release -DBUILD_TESTING=OFF -DJPEGXL_ENABLE_BENCHMARK=OFF -DJPEGXL_ENABLE_EXAMPLES=OFF -DJPEGXL_ENABLE_MANPAGES=OFF -DJPEGXL_ENABLE_PLUGINS=OFF -DJPEGXL_ENABLE_DOXYGEN=OFF -DJPEGXL_ENABLE_JNI=OFF -DJPEGXL_ENABLE_SJPEG=OFF -DJPEGXL_ENABLE_OPENEXR=OFF -DBUILD_SHARED_LIBS=OFF -G Ninja +djxl_version: djxl v0.13.0 196a43d9 [_AVX2_,SSE4,SSE2] {GNU 15.2.0} +cjxl_version: cjxl v0.13.0 196a43d9 [_AVX2_,SSE4,SSE2] {GNU 15.2.0} [jxl-oxide] -path: C:\Users\dk\.cargo\bin\jxl-oxide.exe -version: jxl-oxide-cli 0.12.6 +path: /home/dk/.cargo/bin/jxl-oxide +version: jxl-oxide-cli 0.12.6 diff --git a/JPXL/tools/quality_corpus_extend.py b/JPXL/tools/quality_corpus_extend.py new file mode 100644 index 00000000..7a56f859 --- /dev/null +++ b/JPXL/tools/quality_corpus_extend.py @@ -0,0 +1,227 @@ +#!/usr/bin/env python3 +"""Extend the quality corpus from an external image collection (one-shot +program corpus growth, memo section 10). + +Scans a directory tree of JPEG/PNG images, samples a diverse subset (capped +per subdirectory so one prolific source cannot dominate), converts each pick +to an 8-bit P6 PPM under ``test-set/<name>/``, and writes a manifest with +the same family fields the main corpus carries: + +* every distinct source file is one image **family** (same-directory files + whose normalised name stems match are folded into one family — duplicate + scans of one artwork must not straddle splits); +* families are assigned calibration / development / ext-holdout splits by a + deterministic hash of the family id, so the assignment is reproducible + and never depends on scan order; +* exact byte-duplicates are dropped. + +The originals are never touched; the manifest records the source path, +its sha256 and the conversion command. Standard library plus Pillow (the +same dependency the fixture generator uses). +""" + +from __future__ import annotations + +import argparse +import datetime +import hashlib +import json +import os +import re +import sys + +from PIL import Image + +TOOL_VERSION = "1.0.0" +MANIFEST_SCHEMA = "jpxl.codec-corpus-ext/1" +IMAGE_EXTS = (".jpg", ".jpeg", ".png") +# Reproductions of one artwork often differ only by a resolution or copy +# suffix; fold those into one family. +STEM_NOISE = re.compile(r"[\s_\-]*(\(\d+\)|copy|\d{3,4}x\d{3,4}|small|large|hd)$", re.I) + + +def normalised_stem(filename: str) -> str: + stem = os.path.splitext(filename)[0].lower() + previous = None + while previous != stem: + previous = stem + stem = STEM_NOISE.sub("", stem).strip() + return stem or filename.lower() + + +def family_key(rel_dir: str, filename: str) -> str: + return f"{rel_dir}/{normalised_stem(filename)}".replace("\\", "/") + + +def split_for_family(family: str, holdout_fraction: float, dev_fraction: float) -> str: + """Deterministic split assignment from the family id alone.""" + digest = hashlib.sha256(family.encode("utf-8")).digest() + value = int.from_bytes(digest[:8], "big") / float(1 << 64) + if value < holdout_fraction: + return "ext-holdout" + if value < holdout_fraction + dev_fraction: + return "development" + return "calibration" + + +def scan(root: str) -> list[tuple[str, str, int]]: + """Every image as (relative dir, filename, byte size), sorted.""" + out = [] + for dirpath, _, filenames in os.walk(root): + rel = os.path.relpath(dirpath, root) + for name in sorted(filenames): + if os.path.splitext(name)[1].lower() not in IMAGE_EXTS: + continue + try: + size = os.path.getsize(os.path.join(dirpath, name)) + except OSError: + continue + out.append((rel, name, size)) + out.sort() + return out + + +def sample( + images: list[tuple[str, str, int]], + max_images: int, + per_dir_cap: int, +) -> list[tuple[str, str]]: + """A diverse deterministic subset: round-robin over directories, each + directory's files ordered by a content-independent hash of their family + key, capped per directory.""" + by_dir: dict[str, list[tuple[str, str]]] = {} + seen_families: set[str] = set() + for rel, name, _ in images: + family = family_key(rel, name) + if family in seen_families: + continue + seen_families.add(family) + by_dir.setdefault(rel, []).append((rel, name)) + for entries in by_dir.values(): + # Salted into its own domain: ordering by the *split* hash would bias + # the picked families toward one split. + entries.sort( + key=lambda e: hashlib.sha256(b"order:" + family_key(*e).encode()).hexdigest() + ) + del entries[per_dir_cap:] + picked: list[tuple[str, str]] = [] + queues = sorted(by_dir.items()) + index = 0 + while len(picked) < max_images and any(q for _, q in queues): + rel, queue = queues[index % len(queues)] + if queue: + picked.append(queue.pop(0)) + index += 1 + return picked + + +def sha256_bytes(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +def to_ppm(image: Image.Image) -> bytes: + rgb = image.convert("RGB") + header = f"P6\n{rgb.width} {rgb.height}\n255\n".encode() + return header + rgb.tobytes() + + +def cmd_build(args: argparse.Namespace) -> int: + images = scan(args.source_dir) + if not images: + print(f"no images under {args.source_dir}", file=sys.stderr) + return 1 + picked = sample(images, args.max_images, args.per_dir_cap) + out_dir = os.path.join("test-set", args.name) + os.makedirs(out_dir, exist_ok=True) + + entries = [] + dropped = 0 + seen_hashes: set[str] = set() + for index, (rel, name) in enumerate(picked): + source = os.path.join(args.source_dir, rel, name) + try: + with open(source, "rb") as fh: + raw = fh.read() + source_sha = sha256_bytes(raw) + if source_sha in seen_hashes: + dropped += 1 + continue + seen_hashes.add(source_sha) + with Image.open(source) as im: + im.load() + if min(im.width, im.height) < args.min_side: + dropped += 1 + continue + if im.width * im.height > args.max_pixels: + dropped += 1 + continue + ppm = to_ppm(im) + except Exception as error: # noqa: BLE001 - a bad file is data, not a bug + print(f"skip (unreadable): {name} ({type(error).__name__})", file=sys.stderr) + continue + family = family_key(rel, name) + image_id = f"{args.name}-{index:04d}-{sha256_bytes(family.encode())[:8]}" + ppm_name = f"{image_id}.ppm" + with open(os.path.join(out_dir, ppm_name), "wb") as fh: + fh.write(ppm) + entries.append( + { + "id": image_id, + "path": f"{args.name}/{ppm_name}", + "sha256": sha256_bytes(ppm), + "split": split_for_family(family, args.holdout_fraction, args.dev_fraction), + "class": args.image_class, + "kind": "external", + "bit_depth": 8, + "license": "private source collection; corpus use only, not redistributed", + "provenance": f"decoded from {source} (sha256 {source_sha})", + "family_id": family, + "variant_id": image_id, + "generator_family": None, + "source_capture_id": None, + } + ) + + manifest = { + "schema": MANIFEST_SCHEMA, + "generated_by": f"quality_corpus_extend.py {TOOL_VERSION}", + "generated_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "source_dir": args.source_dir, + "splits": ["calibration", "development", "ext-holdout"], + "images": entries, + } + manifest_path = os.path.join("test-set", f"{args.name}-manifest.json") + with open(manifest_path, "w", encoding="utf-8", newline="\n") as fh: + fh.write(json.dumps(manifest, indent=1, ensure_ascii=False) + "\n") + by_split: dict[str, int] = {} + for e in entries: + by_split[e["split"]] = by_split.get(e["split"], 0) + 1 + print( + f"wrote {len(entries)} images ({dropped} dropped) to {out_dir}; " + f"splits {by_split}; manifest {manifest_path}" + ) + return 0 + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--source-dir", required=True) + parser.add_argument("--name", required=True, help="corpus name under test-set/") + parser.add_argument("--max-images", type=int, default=240) + parser.add_argument("--per-dir-cap", type=int, default=6) + parser.add_argument("--min-side", type=int, default=128) + parser.add_argument("--max-pixels", type=int, default=13_000_000) + parser.add_argument("--holdout-fraction", type=float, default=0.2) + parser.add_argument("--dev-fraction", type=float, default=0.3) + parser.add_argument("--image-class", default="painting") + parser.set_defaults(func=cmd_build) + return parser + + +def main(argv: list[str]) -> int: + args = build_parser().parse_args(argv) + return args.func(args) + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:])) diff --git a/JPXL/tools/quality_oracle_labels.py b/JPXL/tools/quality_oracle_labels.py new file mode 100644 index 00000000..1a6a4373 --- /dev/null +++ b/JPXL/tools/quality_oracle_labels.py @@ -0,0 +1,529 @@ +#!/usr/bin/env python3 +"""Production-endpoint oracle labels for the one-shot quality program (PR 2). + +For every requested image of the quality corpus this tool sweeps the full +effective-scale ladder with ``jpxl quality-ladder`` — a *fresh* production +Balanced (or Fast) pixel plan per rung, scored by the canonical in-tree +SSIMULACRA2 — then locates each target knot's crossing by adaptive geometric +densification, exact-prices the crossing neighbourhood, and reduces the raw +measurements to one labels row per image x target: + +* the **coarsest measured rung whose fresh plan meets the target** (the label + a crossing predictor must reproduce), plus the log-loss interpolated + crossing scale for regression smoothness; +* **censoring** instead of a fake crossing when even the ladder's top rung + misses the target (``crossing > top``), and a ``floor`` flag when the + ladder's coarsest rung already meets it; +* the **local loss slope** ``beta = -d ln(100 - score) / d ln(scale)`` fitted + around the crossing; +* **exact bytes** at the label rung and its measured neighbours; +* the frame's deterministic source features and the manifest's family/split + fields, so training can split and weight by image family. + +Unlike ``calibrate_initial_rung.py`` (fixed-quantizer ``--global-scale`` +sweeps with ``HfMul = 1``), every point here is the production quality +pixel policy over the complete effective ladder including its ``HfMul`` +segments — the same operating points the score controller emits. + +Subcommands: + + sweep Run the ladder sweeps; one raw JSONL file per image (resumable). + labels Reduce raw sweeps + manifest to a labels JSONL with provenance. + +Standard library only. The pure helpers are importable for the unit tests in +``tools/tests/test_quality_oracle_labels.py``. +""" + +from __future__ import annotations + +import argparse +import datetime +import hashlib +import json +import math +import os +import subprocess +import sys +import time + +TOOL_VERSION = "1.0.0" +RAW_SCHEMA = "jpxl.quality-oracle-raw/1" +LABELS_SCHEMA = "jpxl.quality-oracle-labels/1" +DEFAULT_TARGETS = [30.0, 50.0, 70.0, 80.0, 85.0, 90.0, 95.0] +LOSS_EPSILON = 1e-3 +# A sentinel far above any representable effective scale; the encoder clamps +# it to the ladder's top rung and reports the real value back. +TOP_SENTINEL = 2_000_000_000 +# Below this many pixels on a side SSIMULACRA2 is under its own floor and the +# encoder refuses the sweep; such images carry no oracle signal. +MIN_SIDE = 64 + + +# -------------------------------------------------------------------------- +# Pure helpers (unit-tested) +# -------------------------------------------------------------------------- + + +def loss(score: float) -> float: + """The metric loss of a score, floored so its logarithm is finite.""" + return max(100.0 - score, LOSS_EPSILON) + + +def geometric_grid(lo: int, hi: int, ratio: float) -> list[int]: + """Integer effective scales from ``lo`` to ``hi`` at ``ratio`` steps. + + Both endpoints are included; consecutive duplicates are dropped. + """ + if lo < 1 or hi < lo or ratio <= 1.0: + raise ValueError("need 1 <= lo <= hi and ratio > 1") + out = [] + value = float(lo) + while value < hi: + out.append(round(value)) + value *= ratio + out.append(hi) + deduped = [] + for v in out: + if not deduped or v > deduped[-1]: + deduped.append(v) + return deduped + + +def crossing_state(points: list[tuple[int, float]], target: float) -> dict: + """Where ``target`` crosses a measured (scale, score) ladder. + + ``points`` must be sorted ascending by scale with unique scales. The + label definition is the *coarsest measured point meeting the target*, so + a local score reversal never hides a coarser feasible point. Returns:: + + {"state": "censored"} # nothing meets it + {"state": "floor", "above": (s, score)} # the first point does + {"state": "crossed", "below": (..), "above": (..)} + + where ``above`` is the coarsest meeting point and ``below`` the next + coarser measured point. + """ + meeting = [i for i, (_, score) in enumerate(points) if score >= target] + if not meeting: + return {"state": "censored"} + first = meeting[0] + if first == 0: + return {"state": "floor", "above": points[0]} + return {"state": "crossed", "below": points[first - 1], "above": points[first]} + + +def refine_scales(below: int, above: int, count: int) -> list[int]: + """``count`` geometric scales strictly inside ``(below, above)``.""" + if above <= below + 1 or count <= 0: + return [] + out = [] + for i in range(1, count + 1): + t = i / (count + 1) + s = round(math.exp(math.log(below) + t * (math.log(above) - math.log(below)))) + if below < s < above and (not out or s > out[-1]): + out.append(s) + return out + + +def interp_crossing(below: tuple[int, float], above: tuple[int, float], target: float) -> float: + """The log-loss interpolated crossing scale between two bracket points. + + Mirrors the encoder's ``log_loss_crossing``: linear in + ``ln(100 - score)`` against ``ln(scale)``. Falls back to the geometric + midpoint when the bracket does not order in loss. + """ + (s_lo, score_lo), (s_hi, score_hi) = below, above + x_lo, x_hi = math.log(s_lo), math.log(s_hi) + y_lo, y_hi = math.log(loss(score_lo)), math.log(loss(score_hi)) + if y_hi >= y_lo or x_hi <= x_lo: + return math.exp((x_lo + x_hi) / 2.0) + slope = (y_hi - y_lo) / (x_hi - x_lo) + x = x_lo + (math.log(loss(target)) - y_lo) / slope + return math.exp(min(max(x, x_lo), x_hi)) + + +def local_beta(points: list[tuple[int, float]], crossing_scale: float, k: int = 4) -> float | None: + """Least-squares ``-d ln(loss) / d ln(scale)`` over the ``k`` nearest points. + + ``None`` when fewer than two distinct points exist or the fit is not a + positive slope (loss must fall as the scale rises). + """ + if len(points) < 2: + return None + nearest = sorted(points, key=lambda p: abs(math.log(p[0]) - math.log(crossing_scale)))[:k] + xs = [math.log(s) for s, _ in nearest] + ys = [math.log(loss(score)) for _, score in nearest] + n = len(xs) + mean_x = sum(xs) / n + mean_y = sum(ys) / n + var_x = sum((x - mean_x) ** 2 for x in xs) + if var_x <= 0.0: + return None + slope = sum((x - mean_x) * (y - mean_y) for x, y in zip(xs, ys)) / var_x + beta = -slope + return beta if beta > 0.0 else None + + +def merge_points(rounds: list[list[dict]]) -> list[dict]: + """Merge ladder records from several runs, keyed by rung. + + A later record replaces an earlier one only when it adds pricing; scores + are deterministic per rung, so duplicates otherwise carry no news. + """ + by_rung: dict[int, dict] = {} + for records in rounds: + for r in records: + rung = r["rung"] + held = by_rung.get(rung) + if held is None or (held.get("bytes") is None and r.get("bytes") is not None): + by_rung[rung] = r + return [by_rung[k] for k in sorted(by_rung)] + + +def sha256_file(path: str) -> str: + h = hashlib.sha256() + with open(path, "rb") as fh: + for chunk in iter(lambda: fh.read(1 << 20), b""): + h.update(chunk) + return h.hexdigest() + + +# -------------------------------------------------------------------------- +# Sweep driver +# -------------------------------------------------------------------------- + + +def run_ladder( + jpxl: str, + image_path: str, + scales: list[int], + threads: int, + effort: str, + price: bool, +) -> tuple[dict, list[dict]]: + """One ``jpxl quality-ladder`` invocation: (header, point records).""" + command = [ + jpxl, + "quality-ladder", + "--scales", + ",".join(str(s) for s in scales), + "--effort", + effort, + "--threads", + str(threads), + ] + if price: + command.append("--price") + command.append(image_path) + completed = subprocess.run(command, capture_output=True, text=True, check=False) + if completed.returncode != 0: + raise RuntimeError( + f"quality-ladder failed on {image_path}: {completed.stderr.strip()}" + ) + records = [json.loads(line) for line in completed.stdout.splitlines() if line.strip()] + if not records or records[0].get("schema") != "jpxl.quality-ladder/1": + raise RuntimeError(f"unexpected quality-ladder output on {image_path}") + return records[0], records[1:] + + +def sweep_image( + jpxl: str, + image_path: str, + targets: list[float], + threads: int, + effort: str, + coarse_ratio: float, + refine_points: int, + refine_rounds: int, +) -> dict: + """The full adaptive sweep of one image: coarse, refine, price.""" + coarse_scales = geometric_grid(1, 73728, coarse_ratio) + [TOP_SENTINEL] + header, coarse = run_ladder(jpxl, image_path, coarse_scales, threads, effort, price=False) + rounds = [coarse] + + for _ in range(refine_rounds): + merged = merge_points(rounds) + pts = [(r["effective_scale"], r["score"]) for r in merged] + wanted: list[int] = [] + for target in targets: + state = crossing_state(pts, target) + if state["state"] != "crossed": + continue + below_scale = state["below"][0] + above_scale = state["above"][0] + wanted.extend(refine_scales(below_scale, above_scale, refine_points)) + wanted = sorted(set(wanted)) + if not wanted: + break + _, extra = run_ladder(jpxl, image_path, wanted, threads, effort, price=False) + rounds.append(extra) + + # Price the crossing neighbourhood: the label rung and its measured + # neighbours on each side, per target, deduped. + merged = merge_points(rounds) + pts = [(r["effective_scale"], r["score"]) for r in merged] + price_scales: set[int] = set() + for target in targets: + state = crossing_state(pts, target) + if state["state"] == "censored": + continue + above_scale = state["above"][0] + index = next(i for i, (s, _) in enumerate(pts) if s == above_scale) + for j in (index - 1, index, index + 1): + if 0 <= j < len(pts): + price_scales.add(pts[j][0]) + if price_scales: + _, priced = run_ladder( + jpxl, image_path, sorted(price_scales), threads, effort, price=True + ) + rounds.append(priced) + + return { + "schema": RAW_SCHEMA, + "header": header, + "targets": targets, + "points": merge_points(rounds), + } + + +def cmd_sweep(args: argparse.Namespace) -> int: + started = time.monotonic() + with open(args.manifest, encoding="utf-8") as fh: + manifest = json.load(fh) + # Corpus image paths are relative to the manifest's own directory + # (test-set/quality-corpus.json sits beside quality-guard/). + base_dir = os.path.dirname(os.path.abspath(args.manifest)) + os.makedirs(args.out_dir, exist_ok=True) + + wanted_splits = set(args.splits) + selected = [ + image + for image in manifest["images"] + if image["split"] in wanted_splits and (not args.only or image["id"] in args.only) + ] + skipped = 0 + for image in selected: + if args.time_budget_minutes is not None: + elapsed = (time.monotonic() - started) / 60.0 + if elapsed >= args.time_budget_minutes: + print( + f"time budget of {args.time_budget_minutes} min reached after " + f"{elapsed:.1f} min; the sweep is resumable — run again to continue" + ) + break + out_path = os.path.join(args.out_dir, f"{image['id']}.jsonl") + if os.path.exists(out_path) and not args.force: + skipped += 1 + continue + image_path = os.path.join(base_dir, image["path"]) + # The metric floor: tiny fixtures carry no oracle signal. + with open(image_path, "rb") as fh: + magic = fh.readline() + dims = fh.readline().split() + if magic.strip() != b"P6" or len(dims) < 2: + print(f"skip (not P6): {image['id']}", file=sys.stderr) + continue + width, height = int(dims[0]), int(dims[1]) + if min(width, height) < MIN_SIDE: + print(f"skip (below {MIN_SIDE}px metric floor): {image['id']}") + continue + print(f"sweep {image['id']} ({width}x{height}, {image['split']})", flush=True) + raw = sweep_image( + args.jpxl, + image_path, + [float(t) for t in args.targets], + args.threads, + args.effort, + args.coarse_ratio, + args.refine_points, + args.refine_rounds, + ) + raw["image"] = { + "id": image["id"], + "path": image["path"], + "sha256": image["sha256"], + "split": image["split"], + "class": image["class"], + "family_id": image.get("family_id"), + "variant_id": image.get("variant_id"), + "source_capture_id": image.get("source_capture_id"), + } + with open(out_path, "w", encoding="utf-8", newline="\n") as fh: + fh.write(json.dumps(raw, sort_keys=False) + "\n") + if skipped: + print(f"{skipped} image(s) already swept (use --force to redo)") + return 0 + + +# -------------------------------------------------------------------------- +# Label reduction +# -------------------------------------------------------------------------- + + +def labels_for_raw(raw: dict) -> list[dict]: + """Reduce one raw sweep to per-target label rows.""" + points = raw["points"] + pts = [(r["effective_scale"], r["score"]) for r in points] + by_scale = {r["effective_scale"]: r for r in points} + top = points[-1] + rows = [] + for target in raw["targets"]: + state = crossing_state(pts, target) + row: dict = { + "image_id": raw["image"]["id"], + "family_id": raw["image"]["family_id"], + "split": raw["image"]["split"], + "class": raw["image"]["class"], + "target": target, + "state": state["state"], + "top_rung": top["rung"], + "top_effective_scale": top["effective_scale"], + "top_score": top["score"], + } + if state["state"] == "censored": + # The crossing exists beyond the ladder: a right-censored label. + row.update( + { + "label_rung": None, + "label_effective_scale": None, + "crossing_scale": None, + "beta": None, + "label_bytes": None, + "neighbor_bytes": [], + } + ) + else: + above = state["above"] + above_record = by_scale[above[0]] + crossing = ( + interp_crossing(state["below"], above, target) + if state["state"] == "crossed" + else float(above[0]) + ) + index = next(i for i, (s, _) in enumerate(pts) if s == above[0]) + neighbor_bytes = [] + for j in (index - 1, index, index + 1): + if 0 <= j < len(pts): + record = by_scale[pts[j][0]] + if record.get("bytes") is not None: + neighbor_bytes.append( + { + "rung": record["rung"], + "effective_scale": record["effective_scale"], + "score": record["score"], + "bytes": record["bytes"], + } + ) + row.update( + { + "label_rung": above_record["rung"], + "label_effective_scale": above_record["effective_scale"], + "label_score": above[1], + "crossing_scale": crossing, + "beta": local_beta(pts, crossing), + "label_bytes": above_record.get("bytes"), + "neighbor_bytes": neighbor_bytes, + } + ) + rows.append(row) + return rows + + +def git_provenance(repo_root: str) -> dict: + def run(*argv: str) -> str: + return subprocess.run( + ["git", *argv], cwd=repo_root, capture_output=True, text=True, check=True + ).stdout.strip() + + dirty = run("status", "--porcelain") != "" + return {"commit": run("rev-parse", "HEAD"), "dirty": dirty} + + +def cmd_labels(args: argparse.Namespace) -> int: + repo_root = os.path.dirname(os.path.dirname(os.path.abspath(args.manifest))) + raw_files = sorted( + os.path.join(sweep_dir, name) + for sweep_dir in args.sweep_dir + for name in os.listdir(sweep_dir) + if name.endswith(".jsonl") + ) + if not raw_files: + print(f"no raw sweeps in {args.sweep_dir}", file=sys.stderr) + return 1 + all_rows = [] + headers = [] + for path in raw_files: + with open(path, encoding="utf-8") as fh: + raw = json.loads(fh.read()) + if raw.get("schema") != RAW_SCHEMA: + print(f"skip (wrong schema): {path}", file=sys.stderr) + continue + headers.append(raw["header"]) + for row in labels_for_raw(raw): + row["source_features"] = raw["header"]["source_features"] + all_rows.append(row) + + provenance = { + "schema": LABELS_SCHEMA, + "generated_by": f"quality_oracle_labels.py {TOOL_VERSION}", + "generated_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "git": git_provenance(repo_root), + "manifest_sha256": sha256_file(args.manifest), + "metric_version": headers[0]["metric_version"] if headers else None, + "effort": headers[0]["effort"] if headers else None, + "images": len(raw_files), + "rows": len(all_rows), + "censored_rows": sum(1 for r in all_rows if r["state"] == "censored"), + } + with open(args.output, "w", encoding="utf-8", newline="\n") as fh: + fh.write(json.dumps(provenance, sort_keys=False) + "\n") + for row in all_rows: + fh.write(json.dumps(row, sort_keys=False) + "\n") + print( + f"wrote {len(all_rows)} label rows ({provenance['censored_rows']} censored) " + f"from {len(raw_files)} sweeps to {args.output}" + ) + return 0 + + +# -------------------------------------------------------------------------- +# Entry +# -------------------------------------------------------------------------- + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="command", required=True) + + sweep = sub.add_parser("sweep", help="run the adaptive ladder sweeps") + sweep.add_argument("--manifest", required=True) + sweep.add_argument("--jpxl", required=True, help="path to the jpxl binary") + sweep.add_argument("--out-dir", required=True) + sweep.add_argument("--splits", nargs="+", default=["calibration", "development"]) + sweep.add_argument("--targets", nargs="+", default=DEFAULT_TARGETS, type=float) + sweep.add_argument("--threads", type=int, default=4) + sweep.add_argument("--effort", choices=["fast", "balanced"], default="balanced") + sweep.add_argument("--coarse-ratio", type=float, default=1.6) + sweep.add_argument("--refine-points", type=int, default=5) + sweep.add_argument("--refine-rounds", type=int, default=2) + sweep.add_argument("--force", action="store_true") + sweep.add_argument("--only", nargs="*", default=None, help="restrict to these image ids") + sweep.add_argument("--time-budget-minutes", type=float, default=None, + help="stop cleanly (resumable) once this much wall time has passed") + sweep.set_defaults(func=cmd_sweep) + + labels = sub.add_parser("labels", help="reduce raw sweeps to a labels JSONL") + labels.add_argument("--manifest", required=True) + labels.add_argument("--sweep-dir", required=True, nargs="+") + labels.add_argument("--output", required=True) + labels.set_defaults(func=cmd_labels) + return parser + + +def main(argv: list[str]) -> int: + args = build_parser().parse_args(argv) + return args.func(args) + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:])) diff --git a/JPXL/tools/quality_predictor_v2.py b/JPXL/tools/quality_predictor_v2.py new file mode 100644 index 00000000..d81dd3fd --- /dev/null +++ b/JPXL/tools/quality_predictor_v2.py @@ -0,0 +1,1017 @@ +#!/usr/bin/env python3 +"""Train the one-shot program's crossing predictor (PR 3) from oracle labels. + +Consumes the ``jpxl.quality-oracle-labels/1`` JSONL that +``quality_oracle_labels.py labels`` produced and fits, per target knot, a +small transparent linear model in robust-standardized feature space: + +* median crossing (pinball tau = 0.50) of ``ln(effective scale)``; +* candidate crossing (tau = 0.90) — the risk-adjusted rung a one-shot + controller would plan first; +* lower/upper interval (tau = 0.10 / 0.95); +* local loss exponent ``beta`` (least squares on ``ln beta``); +* saturation risk (logistic on the censored indicator). + +Censored rows (``crossing > top``) never enter a crossing fit — they train +only the saturation model — and every family carries equal weight, split +across its rows, so resolution variants of one photograph cannot dominate. + +Evaluation is leave-one-family-out over the calibration+development +families: median/p90/p99 absolute log-scale error of the median model, +first-plan success of the candidate model (candidate scale at or above the +oracle crossing), and simulated byte regression from the priced crossing +neighbourhood. The memo's gates are printed next to the measurements. + +``train`` also emits the generated Rust table for the production feature +set and a JSON report with full provenance. Standard library only; fits are +deterministic (fixed iteration counts, fixed summation order). +""" + +from __future__ import annotations + +import argparse +import datetime +import hashlib +import json +import math +import os +import subprocess +import sys + +TOOL_VERSION = "1.0.0" +# Model/schema identity per emitted feature set. The runtime's feature +# builder must mirror the schema exactly; the generated dim const guards the +# match at compile time. +EMIT_IDENTITY = { + "source": ("qpv2-source-1", "qpv2-source/1"), + "source+transform": ("qpv2-st-1", "qpv2-st/1"), +} +REPORT_SCHEMA = "jpxl.qpv2-report/1" +EPS = 1e-9 +TARGET_KNOTS = [30.0, 50.0, 70.0, 80.0, 85.0, 90.0, 95.0] +QUANTILES = {"lower": 0.10, "median": 0.50, "candidate": 0.90, "upper": 0.95} +RIDGE = 1e-3 +GD_ITERS = 1500 +GD_LR = 0.05 +# Predictions are clamped into the ladder's ln-effective-scale range before +# they are exponentiated or scored; a runaway extrapolation on a held-out +# family lands on the ladder's end, exactly as the runtime rung mapping +# would clamp it. +LN_SCALE_RANGE = (0.0, 16.0) +# The model's declared domain: below this shortest side the metric sits too +# close to its own floor and the runtime routes straight to the exact +# controller (an OOD flag, not a prediction). +MIN_DOMAIN_SIDE = 128 +# Mirrors the navigator: overshoot beyond this band triggers the optional +# byte-tightening attempt, aimed this margin above the target. +MET_OVERSHOOT_BAND = 1.0 +MIN_AIM_MARGIN = 0.25 +# Full-frame reconstructions a routed-to-exact request is charged in the +# simulated work accounting (the current controller's holdout median). +FALLBACK_WORK = 4 + + +def clamp_ln_scale(value: float) -> float: + return min(max(value, LN_SCALE_RANGE[0]), LN_SCALE_RANGE[1]) + + +# -------------------------------------------------------------------------- +# Features +# -------------------------------------------------------------------------- + + +def ln(value: float) -> float: + return math.log(max(value, 0.0) + EPS) + + +FEATURE_NAMES = [ + "ln_luma_q10", + "ln_luma_q50", + "ln_luma_q90", + "ln_chroma_q50", + "flat_fraction", + "ln_edge_proxy", + "log2_pixels", + "ln_aspect", + "grayscale", +] + + +def feature_vector(sf: dict, feature_set: str, extra: dict | None = None) -> list[float]: + """The raw (unstandardized) feature vector of one image.""" + width = float(sf["width"]) + height = float(sf["height"]) + base = [ + ln(sf["luma_variance_q10"]), + ln(sf["luma_variance_q50"]), + ln(sf["luma_variance_q90"]), + ln(sf["chroma_variance_q50"]), + float(sf["flat_fraction"]), + ln(sf["edge_proxy"]), + math.log2(max(width * height, 1.0)), + math.log(max(width, 1.0) / max(height, 1.0)), + 1.0 if sf["grayscale"] else 0.0, + ] + if feature_set == "table2": + return [ln(sf["luma_variance_q50"]), float(sf["flat_fraction"])] + if feature_set == "source": + return base + if feature_set == "source+transform": + if extra is None: + raise ValueError("source+transform needs the per-image transform summary") + # `blocks` duplicates log2_pixels; every other summary field enters raw + # (they are already log/ratio/fraction shaped), in sorted key order. + return base + [float(extra[k]) for k in sorted(extra) if k != "blocks"] + raise ValueError(f"unknown feature set {feature_set}") + + +def robust_standardizer(rows: list[list[float]]) -> tuple[list[float], list[float]]: + """Per-feature (median, scale) with scale = 1.4826 * MAD. + + A feature whose MAD is zero (binary flags, near-constant corpora) keeps + scale 1.0: standardizing by a floored-tiny MAD would blow its values up + by orders of magnitude and let one rare row dominate every fit. + """ + dim = len(rows[0]) + centers, scales = [], [] + for j in range(dim): + values = sorted(row[j] for row in rows) + med = values[len(values) // 2] + mad = sorted(abs(v - med) for v in values)[len(values) // 2] + centers.append(med) + scales.append(1.4826 * mad if mad > 1e-12 else 1.0) + return centers, scales + + +def standardize(row: list[float], centers: list[float], scales: list[float]) -> list[float]: + return [(v - c) / s for v, c, s in zip(row, centers, scales)] + + +# -------------------------------------------------------------------------- +# Deterministic fitting +# -------------------------------------------------------------------------- + + +def solve_linear(matrix: list[list[float]], rhs: list[float]) -> list[float]: + """Gaussian elimination with partial pivoting (small dense systems).""" + n = len(rhs) + a = [row[:] + [rhs[i]] for i, row in enumerate(matrix)] + for col in range(n): + pivot = max(range(col, n), key=lambda r: abs(a[r][col])) + if abs(a[pivot][col]) < 1e-12: + a[col][col] += 1e-9 + else: + a[col], a[pivot] = a[pivot], a[col] + for r in range(col + 1, n): + factor = a[r][col] / a[col][col] + for c in range(col, n + 1): + a[r][c] -= factor * a[col][c] + out = [0.0] * n + for r in range(n - 1, -1, -1): + s = a[r][n] - sum(a[r][c] * out[c] for c in range(r + 1, n)) + out[r] = s / a[r][r] + return out + + +def fit_ols(xs: list[list[float]], ys: list[float], weights: list[float], ridge: float = RIDGE) -> list[float]: + """Weighted ridge least squares with an intercept column prepended.""" + n = len(xs) + dim = len(xs[0]) + 1 + xtx = [[0.0] * dim for _ in range(dim)] + xty = [0.0] * dim + for i in range(n): + row = [1.0] + xs[i] + w = weights[i] + for a in range(dim): + xty[a] += w * row[a] * ys[i] + for b in range(dim): + xtx[a][b] += w * row[a] * row[b] + for a in range(1, dim): + xtx[a][a] += ridge + return solve_linear(xtx, xty) + + +def fit_quantile( + xs: list[list[float]], + ys: list[float], + weights: list[float], + tau: float, + ridge: float = RIDGE, + iterations: int = 40, +) -> list[float]: + """Smoothed pinball-loss linear fit by deterministic IRLS. + + Each round solves a weighted ridge least-squares problem whose per-row + weight is the pinball subgradient magnitude over ``max(|residual|, + delta)`` — the standard iteratively reweighted approximation of quantile + regression. Fixed iteration count and summation order keep the result + reproducible. + """ + delta = 1e-3 + w = fit_ols(xs, ys, weights, ridge=ridge) + for _ in range(iterations): + irls_weights = [] + for i, x in enumerate(xs): + row = [1.0] + x + residual = ys[i] - sum(w[a] * row[a] for a in range(len(w))) + side = tau if residual > 0 else (1.0 - tau) + irls_weights.append(weights[i] * side / max(abs(residual), delta)) + w = fit_ols(xs, ys, irls_weights, ridge=ridge) + return w + + +def fit_logistic(xs: list[list[float]], ys: list[float], weights: list[float]) -> list[float]: + """Weighted logistic regression by deterministic gradient descent. + + Degenerate one-class knots return an intercept-only model saturating at + the observed class (clamped so the probability stays in (0, 1)). + """ + dim = len(xs[0]) + 1 + positives = sum(1 for y in ys if y > 0.5) + if positives == 0 or positives == len(ys): + logit = -6.0 if positives == 0 else 6.0 + return [logit] + [0.0] * (dim - 1) + w = [0.0] * dim + total = sum(weights) + for step in range(GD_ITERS): + lr = 0.5 / (1.0 + step / 200.0) + grad = [0.0] * dim + for i, x in enumerate(xs): + row = [1.0] + x + z = sum(w[a] * row[a] for a in range(dim)) + p = 1.0 / (1.0 + math.exp(-max(min(z, 30.0), -30.0))) + wi = weights[i] / total + for a in range(dim): + grad[a] += wi * (p - ys[i]) * row[a] + for a in range(dim): + if a > 0: + grad[a] += RIDGE * w[a] / max(total, 1.0) + w[a] -= lr * grad[a] + return w + + +def predict(w: list[float], x: list[float]) -> float: + return w[0] + sum(a * b for a, b in zip(w[1:], x)) + + +# -------------------------------------------------------------------------- +# Dataset assembly +# -------------------------------------------------------------------------- + + +def load_labels(path: str) -> tuple[dict, list[dict]]: + with open(path, encoding="utf-8") as fh: + lines = [json.loads(line) for line in fh if line.strip()] + header, rows = lines[0], lines[1:] + if header.get("schema") != "jpxl.quality-oracle-labels/1": + raise ValueError(f"unexpected labels schema in {path}") + return header, rows + + +def family_weights(rows: list[dict]) -> list[float]: + """Equal weight per family, split across that family's rows.""" + counts: dict[str, int] = {} + for row in rows: + counts[row["family_id"]] = counts.get(row["family_id"], 0) + 1 + return [1.0 / counts[row["family_id"]] for row in rows] + + +def rows_for_knot(rows: list[dict], target: float) -> list[dict]: + return [r for r in rows if abs(r["target"] - target) < 1e-9] + + +def load_raw_curves(*sweep_dirs: str) -> dict[str, list[tuple[int, float]]]: + """Per-image measured (effective_scale, score) curves from raw sweeps.""" + curves: dict[str, list[tuple[int, float]]] = {} + for sweep_dir in sweep_dirs: + for name in sorted(os.listdir(sweep_dir)): + if not name.endswith(".jsonl"): + continue + with open(os.path.join(sweep_dir, name), encoding="utf-8") as fh: + raw = json.loads(fh.read()) + if raw.get("schema") != "jpxl.quality-oracle-raw/1": + continue + curves[raw["image"]["id"]] = [ + (p["effective_scale"], p["score"]) for p in raw["points"] + ] + return curves + + +def score_at(curve: list[tuple[int, float]], scale: float) -> float: + """The measured curve's score at ``scale``: piecewise-linear in + ``(ln scale, ln loss)``, clamped at the measured ends.""" + if scale <= curve[0][0]: + return curve[0][1] + if scale >= curve[-1][0]: + return curve[-1][1] + x = math.log(scale) + for (s0, v0), (s1, v1) in zip(curve, curve[1:]): + if s0 <= scale <= s1: + x0, x1 = math.log(s0), math.log(s1) + if x1 <= x0: + return v1 + t = (x - x0) / (x1 - x0) + y = math.log(max(100.0 - v0, 1e-3)) + t * ( + math.log(max(100.0 - v1, 1e-3)) - math.log(max(100.0 - v0, 1e-3)) + ) + return 100.0 - math.exp(y) + return curve[-1][1] + + +def simulate_bytes_ratio(row: dict, predicted_scale: float) -> float | None: + """Bytes(predicted) / bytes(label), from the priced crossing points. + + Interpolates (or locally extrapolates) ``ln bytes`` against ``ln scale`` + over the priced neighbourhood; ``None`` when fewer than two priced points + exist or the label itself was not priced. + """ + priced = [ + (n["effective_scale"], n["bytes"]) + for n in row.get("neighbor_bytes", []) + if n.get("bytes") + ] + if len(priced) < 2 or not row.get("label_bytes"): + return None + priced.sort() + xs = [math.log(s) for s, _ in priced] + ys = [math.log(b) for _, b in priced] + x = math.log(predicted_scale) + if x <= xs[0]: + i = 0 + elif x >= xs[-1]: + i = len(xs) - 2 + else: + i = next(j for j in range(len(xs) - 1) if xs[j] <= x <= xs[j + 1]) + slope = (ys[i + 1] - ys[i]) / (xs[i + 1] - xs[i]) if xs[i + 1] > xs[i] else 0.0 + predicted_bytes = math.exp(ys[i] + slope * (x - xs[i])) + return predicted_bytes / row["label_bytes"] + + +# -------------------------------------------------------------------------- +# Training and evaluation +# +# The crossing curve of one image is close to linear in `ln loss(target)`, +# so instead of fitting each target knot separately (30 rows against 10+ +# parameters — the per-knot ablation was variance-dominated), one pooled +# model is fitted over every uncensored row: +# +# ln scale = A(z) + B(z) * (ln loss(target) - X_T_CENTER) +# +# with `A` affine in all standardized features and `B` affine in a small, +# predeclared slope subset. Folding the pooled fit at a fixed target is +# exactly an affine model over the features again, so the generated per-knot +# Rust table (and the runtime) is unchanged. +# -------------------------------------------------------------------------- + +X_T_CENTER = 3.0 # about ln(100 - 80): centers the slope term near target 80 + +# Raw-feature indices whose interaction with `ln loss(target)` the pooled +# slope may use, per feature set. Small and predeclared, per the memo. +SLOPE_FEATURES = { + "table2": [0], + "source": [1, 4], # ln_luma_q50, flat_fraction + "source+transform": [1, 4, 13], # + ln_ac_y_mean +} + + +def design_row(z: list[float], x_t: float, slope_idx: list[int]) -> list[float]: + xc = x_t - X_T_CENTER + return list(z) + [xc] + [xc * z[i] for i in slope_idx] + + +def fold_at_target(w: list[float], target: float, dim: int, slope_idx: list[int]) -> list[float]: + """The pooled fit as an affine model over the features at one target.""" + xc = math.log(max(100.0 - target, 1e-3)) - X_T_CENTER + intercept = w[0] + w[1 + dim] * xc + coefs = list(w[1 : 1 + dim]) + for j, i in enumerate(slope_idx): + coefs[i] += w[2 + dim + j] * xc + return [intercept] + coefs + + +def pooled_fit(rows: list[dict], feature_set: str, ridge: float) -> dict: + """Standardizer plus pooled quantile/beta fits over all uncensored rows.""" + raw = [ + feature_vector(r["source_features"], feature_set, r.get("transform_features")) + for r in rows + ] + centers, scales = robust_standardizer(raw) + zs = [standardize(r, centers, scales) for r in raw] + z_range = [ + (min(z[j] for z in zs), max(z[j] for z in zs)) for j in range(len(centers)) + ] + slope_idx = SLOPE_FEATURES[feature_set] + crossing = [ + (z, r) for z, r in zip(zs, rows) if r["state"] != "censored" + ] + xs = [ + design_row(z, math.log(max(100.0 - r["target"], 1e-3)), slope_idx) + for z, r in crossing + ] + ys = [math.log(r["crossing_scale"]) for _, r in crossing] + weights = family_weights([r for _, r in crossing]) + fits = { + name: fit_quantile(xs, ys, weights, tau, ridge=ridge) + for name, tau in QUANTILES.items() + } + beta_rows = [ + (x, math.log(r["beta"])) + for x, (_, r) in zip(xs, crossing) + if r.get("beta") + ] + fits["beta"] = ( + fit_ols( + [x for x, _ in beta_rows], + [y for _, y in beta_rows], + [1.0] * len(beta_rows), + ridge=ridge, + ) + if len(beta_rows) >= 8 + else None + ) + return { + "centers": centers, + "scales": scales, + "z_range": z_range, + "slope_idx": slope_idx, + "fits": fits, + "rows": len(crossing), + } + + +def pooled_predict(model: dict, row: dict, feature_set: str, output: str) -> float: + z = standardize( + feature_vector(row["source_features"], feature_set, row.get("transform_features")), + model["centers"], + model["scales"], + ) + x = design_row(z, math.log(max(100.0 - row["target"], 1e-3)), model["slope_idx"]) + return predict(model["fits"][output], x) + + +def train_all(rows: list[dict], feature_set: str, ridge: float = RIDGE) -> dict: + """The pooled fit, folded into the per-knot table the runtime consumes.""" + pooled = pooled_fit(rows, feature_set, ridge) + dim = len(pooled["centers"]) + knots = {} + for target in TARGET_KNOTS: + knot_rows = rows_for_knot(rows, target) + if not knot_rows: + continue + knot: dict = { + "rows": sum(1 for r in knot_rows if r["state"] != "censored"), + "censored": sum(1 for r in knot_rows if r["state"] == "censored"), + } + for name in QUANTILES: + knot[name] = fold_at_target(pooled["fits"][name], target, dim, pooled["slope_idx"]) + if pooled["fits"]["beta"] is not None: + knot["beta"] = fold_at_target(pooled["fits"]["beta"], target, dim, pooled["slope_idx"]) + zs_all = [ + standardize( + feature_vector(r["source_features"], feature_set, r.get("transform_features")), + pooled["centers"], + pooled["scales"], + ) + for r in knot_rows + ] + censored_ys = [1.0 if r["state"] == "censored" else 0.0 for r in knot_rows] + knot["saturation"] = fit_logistic(zs_all, censored_ys, family_weights(knot_rows)) + knots[target] = knot + return { + "centers": pooled["centers"], + "scales": pooled["scales"], + "z_range": pooled["z_range"], + "knots": knots, + } + + +def family_fold(family: str, folds: int) -> int: + """Deterministic fold index of a family (salted hash, scan-order free).""" + digest = hashlib.sha256(b"fold:" + family.encode("utf-8")).digest() + return int.from_bytes(digest[:4], "big") % folds + + +def new_accumulator() -> dict: + return { + "errors": [], + "errors_by_class": {}, + "errors_by_target": {t: [] for t in TARGET_KNOTS}, + "first_plan": [], + "corrected_ok": [], + "fallback_fired": [], + "byte_ratios": [], + "routes": [], + "work": [], + "route_byte_ratios": [], + } + + +def score_rows( + model: dict, + rows: list[dict], + feature_set: str, + curves: dict[str, list[tuple[int, float]]] | None, + acc: dict, +) -> None: + """Score held-out rows against one fitted pooled model, accumulating the + gate metrics and the simulated common-case route (memo section 3.1).""" + errors = acc["errors"] + errors_by_target = acc["errors_by_target"] + first_plan = acc["first_plan"] + corrected_ok = acc["corrected_ok"] + fallback_fired = acc["fallback_fired"] + byte_ratios = acc["byte_ratios"] + routes = acc["routes"] + work = acc["work"] + route_byte_ratios = acc["route_byte_ratios"] + if True: + for row in rows: + if row["state"] == "censored": + continue + oracle = math.log(row["crossing_scale"]) + err = clamp_ln_scale(pooled_predict(model, row, feature_set, "median")) - oracle + errors.append(err) + errors_by_target[row["target"]].append(err) + acc["errors_by_class"].setdefault(row.get("class", "?"), []).append(err) + ln_candidate = clamp_ln_scale(pooled_predict(model, row, feature_set, "candidate")) + candidate_scale = math.exp(ln_candidate) + success = candidate_scale >= row["crossing_scale"] + first_plan.append(success) + ratio = simulate_bytes_ratio(row, candidate_scale) + if ratio is not None and success: + byte_ratios.append(ratio) + + # Runtime fallback simulation: wide interval or out-of-envelope + # feature, matching quality_prediction.rs. + width = clamp_ln_scale( + pooled_predict(model, row, feature_set, "upper") + ) - clamp_ln_scale(pooled_predict(model, row, feature_set, "lower")) + z = standardize( + feature_vector( + row["source_features"], feature_set, row.get("transform_features") + ), + model["centers"], + model["scales"], + ) + ood = any( + v < lo - 0.25 * max(hi - lo, 1e-6) or v > hi + 0.25 * max(hi - lo, 1e-6) + for v, (lo, hi) in zip(z, model["z_range"]) + ) + preflight_fallback = ood or width > 0.405465 + fallback_fired.append(preflight_fallback) + + # Counterfactual common-case route over the measured curve, + # mirroring the memo's section 3.1: one predicted plan, then + # either one byte-tightening attempt (large overshoot), one + # slope correction (miss), or continuation of the exact + # controller. Work is counted in full-frame reconstructions; + # bytes are relative to the oracle crossing (the operating point + # the current controller emits). + curve = (curves or {}).get(row["image_id"]) + if curve is not None: + if model["fits"]["beta"] is not None: + x = design_row( + z, math.log(max(100.0 - row["target"], 1e-3)), model["slope_idx"] + ) + beta = min(max(math.exp(predict(model["fits"]["beta"], x)), 0.2), 3.0) + else: + beta = 0.9 + target = row["target"] + if preflight_fallback: + routes.append("fallback_preflight") + work.append(FALLBACK_WORK) + route_ratio = 1.0 + corrected_ok.append(True) + elif success: + corrected_ok.append(True) + observed = score_at(curve, candidate_scale) + overshoot = observed - target + if overshoot > MET_OVERSHOOT_BAND: + # One coarsening attempt aimed just above the target. + shift = ( + math.log(max(100.0 - observed, 1e-3)) + - math.log(max(100.0 - (target + MIN_AIM_MARGIN), 1e-3)) + ) / beta + tightened = math.exp(clamp_ln_scale(ln_candidate + shift)) + work.append(2) + if tightened >= row["crossing_scale"]: + routes.append("tightened") + route_ratio = simulate_bytes_ratio(row, tightened) + else: + # Keep the verified first plan. + routes.append("tighten_kept_first") + route_ratio = simulate_bytes_ratio(row, candidate_scale) + else: + routes.append("one_shot") + work.append(1) + route_ratio = simulate_bytes_ratio(row, candidate_scale) + else: + observed = score_at(curve, candidate_scale) + shift = ( + math.log(max(100.0 - observed, 1e-3)) + - math.log(max(100.0 - (target + MIN_AIM_MARGIN), 1e-3)) + ) / beta + corrected = math.exp(clamp_ln_scale(ln_candidate + shift)) + if corrected >= row["crossing_scale"]: + corrected_ok.append(True) + routes.append("corrected") + work.append(2) + route_ratio = simulate_bytes_ratio(row, corrected) + else: + corrected_ok.append(False) + routes.append("fallback_exact") + work.append(2 + FALLBACK_WORK) + route_ratio = 1.0 + if route_ratio is not None: + route_byte_ratios.append(route_ratio) + + +def finalize_metrics(acc: dict, families: int) -> dict: + errors = acc["errors"] + errors_by_target = acc["errors_by_target"] + first_plan = acc["first_plan"] + corrected_ok = acc["corrected_ok"] + fallback_fired = acc["fallback_fired"] + byte_ratios = acc["byte_ratios"] + routes = acc["routes"] + work = acc["work"] + route_byte_ratios = acc["route_byte_ratios"] + + def quantile(sorted_values: list[float], q: float) -> float | None: + if not sorted_values: + return None + index = min(int(q * (len(sorted_values) - 1) + 0.9999), len(sorted_values) - 1) + return sorted_values[index] + + abs_sorted = sorted(abs(e) for e in errors) + geomean_bytes = ( + math.exp(sum(math.log(r) for r in byte_ratios) / len(byte_ratios)) + if byte_ratios + else None + ) + return { + "families": families, + "rows_scored": len(errors), + "median_abs_ln_error": quantile(abs_sorted, 0.5), + "p90_abs_ln_error": quantile(abs_sorted, 0.9), + "p99_abs_ln_error": quantile(abs_sorted, 0.99), + "per_target_median_abs": { + str(t): (sorted(abs(e) for e in v)[len(v) // 2] if v else None) + for t, v in errors_by_target.items() + }, + "per_class": { + name: { + "rows": len(v), + "median_abs": sorted(abs(e) for e in v)[len(v) // 2], + "p90_abs": quantile(sorted(abs(e) for e in v), 0.9), + } + for name, v in sorted(acc["errors_by_class"].items()) + }, + "first_plan_success": (sum(first_plan) / len(first_plan)) if first_plan else None, + "first_plan_rows": len(first_plan), + "first_or_correction_success": ( + (sum(corrected_ok) / len(corrected_ok)) if corrected_ok else None + ), + "fallback_rate": ( + (sum(fallback_fired) / len(fallback_fired)) if fallback_fired else None + ), + # Aggregate over the simulated common-case routes (tightening and + # correction included; routed-to-exact rows land on the oracle). + "simulated_byte_geomean": ( + math.exp(sum(math.log(r) for r in route_byte_ratios) / len(route_byte_ratios)) + if route_byte_ratios + else geomean_bytes + ), + "first_plan_byte_geomean": geomean_bytes, + "byte_rows": len(route_byte_ratios) or len(byte_ratios), + "expected_reconstructions": (sum(work) / len(work)) if work else None, + "route_fractions": ( + {name: routes.count(name) / len(routes) for name in sorted(set(routes))} + if routes + else None + ), + } + + +def evaluate_cv( + rows: list[dict], + feature_set: str, + ridge: float = RIDGE, + curves: dict[str, list[tuple[int, float]]] | None = None, + folds: int | None = None, +) -> dict: + """Family-grouped cross-validation of the pooled models. + + Leave-one-family-out when the family count is small; otherwise families + are bucketed into ``folds`` deterministic groups (salted hash of the + family id) so the fit count stays bounded. Rows of one family never + straddle a train/held boundary either way. + """ + families = sorted(set(r["family_id"] for r in rows)) + if folds is None: + folds = 0 if len(families) <= 40 else 10 + if folds and folds < len(families): + groups: dict[int, set[str]] = {} + for family in families: + groups.setdefault(family_fold(family, folds), set()).add(family) + held_sets = [groups[k] for k in sorted(groups)] + else: + held_sets = [{family} for family in families] + acc = new_accumulator() + for held_families in held_sets: + train = [r for r in rows if r["family_id"] not in held_families] + held = [r for r in rows if r["family_id"] in held_families] + model = pooled_fit(train, feature_set, ridge) + score_rows(model, held, feature_set, curves, acc) + return finalize_metrics(acc, len(families)) + + +# Kept as the historical name for the small-corpus path. +evaluate_lofo = evaluate_cv + + +GATES = { + "quick_screen": { + "p90_abs_ln_error": ("<=", 0.45), + "first_plan_success": (">=", 0.80), + "simulated_byte_geomean": ("<=", 1.015), + }, + "production": { + "median_abs_ln_error": ("<=", 0.10), + "p90_abs_ln_error": ("<=", 0.30), + "p99_abs_ln_error": ("<=", 0.70), + }, +} + + +def gate_verdicts(metrics: dict) -> dict: + out = {} + for gate, rules in GATES.items(): + checks = {} + for key, (op, bound) in rules.items(): + value = metrics.get(key) + if value is None: + checks[key] = None + elif op == "<=": + checks[key] = value <= bound + else: + checks[key] = value >= bound + out[gate] = { + "checks": checks, + "passed": all(v for v in checks.values() if v is not None) + and all(v is not None for v in checks.values()), + } + return out + + +# -------------------------------------------------------------------------- +# Rust generation +# -------------------------------------------------------------------------- + + +def rust_array(values: list[float]) -> str: + return "[" + ", ".join(f"{v!r}" for v in values) + "]" + + +def generate_rust(model: dict, provenance: dict, feature_set: str) -> str: + model_version, feature_schema = EMIT_IDENTITY[feature_set] + lines = [ + "//! Generated crossing-predictor model for the one-shot quality program.", + "//!", + f"//! GENERATED by `tools/quality_predictor_v2.py {TOOL_VERSION}` — do not edit", + "//! by hand; retrain and regenerate. The exact controller stays", + "//! authoritative: without the `one-shot-controller` feature this model", + "//! only feeds the shadow trace, never a bitstream.", + "//!", + f"//! model_version: {model_version}", + f"//! feature_schema: {feature_schema}", + f"//! labels: {provenance['labels_sha256']}", + f"//! trained_at: {provenance['generated_at']}", + f"//! git: {provenance['git']['commit']} (dirty: {provenance['git']['dirty']})", + "", + "/// The generated model's version string.", + f'pub const QPV2_MODEL_VERSION: &str = "{model_version}";', + "/// The feature-vector schema the coefficients expect.", + f'pub const QPV2_FEATURE_SCHEMA: &str = "{feature_schema}";', + "/// The feature-vector dimension the coefficient arrays expect. The", + "/// runtime feature builder is sized against this at compile time.", + f"pub const QPV2_FEATURE_DIM: usize = {len(model['centers'])};", + "", + "/// Robust per-feature centers (medians over the training corpus).", + f"pub const QPV2_FEATURE_CENTERS: [f64; {len(model['centers'])}] = {rust_array(model['centers'])};", + "/// Robust per-feature scales (1.4826 x MAD, floored).", + f"pub const QPV2_FEATURE_SCALES: [f64; {len(model['scales'])}] = {rust_array(model['scales'])};", + "/// Standardized-feature range seen in training, for OOD detection.", + f"pub const QPV2_FEATURE_Z_RANGE: [(f64, f64); {len(model['scales'])}] = [" + + ", ".join(f"({lo!r}, {hi!r})" for lo, hi in model["z_range"]) + + "];", + "", + "/// One target knot's fitted linear models over standardized features.", + "/// Every coefficient array is `[intercept, w_0, .., w_n]`; crossing", + "/// outputs are `ln(effective_scale)`, `beta` is `ln(beta)`, and", + "/// `saturation` is a logit.", + "#[derive(Debug, Clone, Copy, PartialEq)]", + "pub struct Qpv2Knot {", + " /// The SSIMULACRA2 target this knot was fitted at.", + " pub target: f64,", + " /// Rows the crossing fits saw (uncensored).", + " pub rows: u32,", + " /// tau = 0.10 crossing quantile.", + f" pub lower: [f64; {len(model['centers']) + 1}],", + " /// tau = 0.50 crossing quantile.", + f" pub median: [f64; {len(model['centers']) + 1}],", + " /// tau = 0.90 crossing quantile (the one-shot candidate).", + f" pub candidate: [f64; {len(model['centers']) + 1}],", + " /// tau = 0.95 crossing quantile.", + f" pub upper: [f64; {len(model['centers']) + 1}],", + " /// ln(beta) least squares, or all-zero when too few rows.", + f" pub beta: [f64; {len(model['centers']) + 1}],", + " /// Saturation-risk logit.", + f" pub saturation: [f64; {len(model['centers']) + 1}],", + "}", + "", + "/// The fitted knots, ascending in target.", + f"pub const QPV2_KNOTS: &[Qpv2Knot] = &[", + ] + for target in TARGET_KNOTS: + knot = model["knots"].get(target) + if not knot or "median" not in knot: + continue + dim = len(model["centers"]) + 1 + beta = knot.get("beta", [0.0] * dim) + lines.extend( + [ + " Qpv2Knot {", + f" target: {float(target)!r},", + f" rows: {knot['rows']},", + f" lower: {rust_array(knot['lower'])},", + f" median: {rust_array(knot['median'])},", + f" candidate: {rust_array(knot['candidate'])},", + f" upper: {rust_array(knot['upper'])},", + f" beta: {rust_array(beta)},", + f" saturation: {rust_array(knot['saturation'])},", + " },", + ] + ) + lines.append("];") + lines.append("") + return "\n".join(lines) + "\n" + + +# -------------------------------------------------------------------------- +# Entry +# -------------------------------------------------------------------------- + + +def sha256_file(path: str) -> str: + h = hashlib.sha256() + with open(path, "rb") as fh: + for chunk in iter(lambda: fh.read(1 << 20), b""): + h.update(chunk) + return h.hexdigest() + + +def git_provenance(root: str) -> dict: + def run(*argv: str) -> str: + return subprocess.run( + ["git", *argv], cwd=root, capture_output=True, text=True, check=True + ).stdout.strip() + + return {"commit": run("rev-parse", "HEAD"), "dirty": run("status", "--porcelain") != ""} + + +def cmd_train(args: argparse.Namespace) -> int: + _, rows = load_labels(args.labels) + # The declared model domain: tiny frames sit against the metric's own + # floor and route straight to the exact controller at runtime, so they + # neither train nor score the model. + in_domain = [ + r + for r in rows + if min(r["source_features"]["width"], r["source_features"]["height"]) + >= MIN_DOMAIN_SIDE + ] + out_of_domain = len(rows) - len(in_domain) + if out_of_domain: + print(f"{out_of_domain} rows below the {MIN_DOMAIN_SIDE}px domain floor route to the exact controller") + rows = in_domain + if args.transform_features: + with open(args.transform_features, encoding="utf-8") as fh: + transform_map = json.load(fh) + for row in rows: + row["transform_features"] = transform_map.get(row["image_id"]) + missing = sum(1 for r in rows if r.get("transform_features") is None) + if missing: + print(f"warning: {missing} label rows lack transform features", file=sys.stderr) + rows = [r for r in rows if r.get("transform_features") is not None] + provenance = { + "generated_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "labels_sha256": sha256_file(args.labels), + "git": git_provenance(args.repo_root), + "tool_version": TOOL_VERSION, + } + + report: dict = { + "schema": REPORT_SCHEMA, + "provenance": provenance, + "feature_sets": {}, + } + chosen_ridge: dict[str, float] = {} + for feature_set in args.feature_sets: + if feature_set == "source+transform" and not args.transform_features: + print("note: source+transform needs --transform-features; skipped") + continue + per_ridge = {} + best_ridge, best_metrics = None, None + curves = load_raw_curves(*args.sweep_dir) if args.sweep_dir else None + train_rows = [r for r in rows if r["split"] != args.blind_split] + blind_rows = [r for r in rows if r["split"] == args.blind_split] + for ridge in args.ridge_grid: + metrics = evaluate_cv( + train_rows, feature_set, ridge, curves=curves, folds=args.cv_folds + ) + per_ridge[str(ridge)] = metrics + if best_metrics is None or metrics["p90_abs_ln_error"] < best_metrics["p90_abs_ln_error"]: + best_ridge, best_metrics = ridge, metrics + chosen_ridge[feature_set] = best_ridge + blind_metrics = None + if blind_rows: + # Never-tuned families: fit on every training row at the chosen + # ridge and score the blind split once. + blind_model = pooled_fit(train_rows, feature_set, best_ridge) + blind_acc = new_accumulator() + score_rows(blind_model, blind_rows, feature_set, curves, blind_acc) + blind_metrics = finalize_metrics( + blind_acc, len(set(r["family_id"] for r in blind_rows)) + ) + report["feature_sets"][feature_set] = { + "ridge_grid": per_ridge, + "chosen_ridge": best_ridge, + "lofo": best_metrics, + "gates": gate_verdicts(best_metrics), + "blind_holdout": blind_metrics, + "blind_gates": gate_verdicts(blind_metrics) if blind_metrics else None, + } + print(f"[{feature_set}] ridge={best_ridge} {json.dumps(best_metrics, indent=2)}") + print(f"[{feature_set}] gates: {json.dumps(gate_verdicts(best_metrics))}") + if blind_metrics: + print(f"[{feature_set}] blind {args.blind_split}: {json.dumps(blind_metrics, indent=2)}") + print(f"[{feature_set}] blind gates: {json.dumps(gate_verdicts(blind_metrics))}") + + # The blind split stays out of the shipped model too, so it remains a + # valid never-tuned holdout for later revisions. + emit_rows = [r for r in rows if r["split"] != args.blind_split] + model = train_all(emit_rows, args.emit_feature_set, chosen_ridge.get(args.emit_feature_set, RIDGE)) + rust = generate_rust(model, provenance, args.emit_feature_set) + with open(args.rust_out, "w", encoding="utf-8", newline="\n") as fh: + fh.write(rust) + report["emitted"] = { + "feature_set": args.emit_feature_set, + "rust": args.rust_out, + "rust_sha256": hashlib.sha256(rust.encode()).hexdigest(), + "knots": {str(t): {"rows": k["rows"], "censored": k["censored"]} for t, k in model["knots"].items()}, + } + with open(args.report_out, "w", encoding="utf-8", newline="\n") as fh: + fh.write(json.dumps(report, indent=2, sort_keys=False) + "\n") + print(f"wrote {args.rust_out} and {args.report_out}") + print("note: run `cargo fmt --all` — the emitted table is not rustfmt-shaped") + return 0 + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="command", required=True) + train = sub.add_parser("train", help="fit, cross-validate, and emit the model") + train.add_argument("--labels", required=True) + train.add_argument("--repo-root", default=".") + train.add_argument( + "--feature-sets", + nargs="+", + default=["table2", "source"], + choices=["table2", "source", "source+transform"], + ) + train.add_argument("--transform-features", default=None, + help="per-image transform-summary JSON map (for source+transform)") + train.add_argument("--ridge-grid", nargs="+", type=float, default=[0.03, 0.3, 3.0], + help="ridge strengths tried per feature set; chosen by LOFO p90") + train.add_argument("--sweep-dir", nargs="+", default=None, + help="raw sweep dir(s); enables the one-correction counterfactual") + train.add_argument("--blind-split", default="ext-holdout", + help="split name held out of all tuning and scored once") + train.add_argument("--cv-folds", type=int, default=None, + help="family-grouped CV folds (default: LOFO up to 40 families, else 10)") + train.add_argument("--emit-feature-set", default="source+transform", + choices=["source", "source+transform"]) + train.add_argument("--rust-out", required=True) + train.add_argument("--report-out", required=True) + train.set_defaults(func=cmd_train) + return parser + + +def main(argv: list[str]) -> int: + args = build_parser().parse_args(argv) + return args.func(args) + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:])) diff --git a/JPXL/tools/tests/test_codec_compare.py b/JPXL/tools/tests/test_codec_compare.py index 781c88c9..9375625b 100644 --- a/JPXL/tools/tests/test_codec_compare.py +++ b/JPXL/tools/tests/test_codec_compare.py @@ -239,7 +239,7 @@ def test_quality_trace_reader(self): json.dumps({"schema": "other/1"}) + "\n" + json.dumps( { - "schema": codec_compare.QUALITY_TRACE_SCHEMA, + "schema": codec_compare.QUALITY_TRACE_SCHEMAS[0], "requested_score": 85.0, "achieved_score": 85.1, "wall_by_phase": { diff --git a/JPXL/tools/tests/test_make_quality_guard_fixtures.py b/JPXL/tools/tests/test_make_quality_guard_fixtures.py index bc42cc72..1d9a1f00 100644 --- a/JPXL/tools/tests/test_make_quality_guard_fixtures.py +++ b/JPXL/tools/tests/test_make_quality_guard_fixtures.py @@ -163,10 +163,36 @@ def test_each_synthetic_class_has_all_three_splits(self): ) def test_split_and_class_counts_are_stable(self): + # 2026-08-24: the one-shot program's coverage expansion added nine + # calibration/development fixtures (noise-lowlight and grayscale from + # more captures; second text/UI/hatch/sky families). + # 2026-08-25: the saturated/gradient corpus-gap closure (memo §10, >=25 + # source families per critical class) added 22 saturated and 17 + # gradient synthetic families across all three splits: +15 calibration, + # +13 development, +7 holdout. by_split = Counter(f.split for f in self.fixtures) - self.assertEqual(by_split["calibration"], 19) - self.assertEqual(by_split["development"], 15) - self.assertEqual(by_split["holdout"], 13) + self.assertEqual(by_split["calibration"], 43) + self.assertEqual(by_split["development"], 32) + self.assertEqual(by_split["holdout"], 20) + + def test_critical_classes_have_25_families(self): + # memo §10: saturated and gradient/banding-stress each need >= 25 + # distinct source families. Synthetic families key on fixture id. + by_class_families: dict[str, set] = defaultdict(set) + for fx in self.fixtures: + by_class_families[fx.klass].add(mqgf.family_fields(fx)["family_id"]) + for klass in ("saturated", "gradient"): + self.assertGreaterEqual( + len(by_class_families[klass]), 25, + f"{klass} has only {len(by_class_families[klass])} families", + ) + + def test_families_never_cross_splits(self): + by_family = defaultdict(set) + for fx in self.fixtures: + by_family[mqgf.family_fields(fx)["family_id"]].add(fx.split) + leaking = {fam: s for fam, s in by_family.items() if len(s) > 1} + self.assertEqual(leaking, {}, "image families must stay in one split") if __name__ == "__main__": diff --git a/JPXL/tools/tests/test_quality_oracle_labels.py b/JPXL/tools/tests/test_quality_oracle_labels.py new file mode 100644 index 00000000..ff0ae64f --- /dev/null +++ b/JPXL/tools/tests/test_quality_oracle_labels.py @@ -0,0 +1,154 @@ +"""Unit tests for the pure helpers of quality_oracle_labels.py.""" + +from __future__ import annotations + +import math +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +import quality_oracle_labels as qol + + +class GridTest(unittest.TestCase): + def test_geometric_grid_includes_endpoints_strictly_increasing(self): + grid = qol.geometric_grid(1, 73728, 1.6) + self.assertEqual(grid[0], 1) + self.assertEqual(grid[-1], 73728) + self.assertTrue(all(a < b for a, b in zip(grid, grid[1:]))) + + def test_geometric_grid_rejects_bad_input(self): + with self.assertRaises(ValueError): + qol.geometric_grid(10, 5, 1.6) + with self.assertRaises(ValueError): + qol.geometric_grid(1, 10, 1.0) + + +class CrossingTest(unittest.TestCase): + POINTS = [(100, 20.0), (200, 40.0), (400, 60.0), (800, 80.0), (1600, 95.0)] + + def test_crossed_picks_the_coarsest_meeting_point(self): + state = qol.crossing_state(self.POINTS, 70.0) + self.assertEqual(state["state"], "crossed") + self.assertEqual(state["below"], (400, 60.0)) + self.assertEqual(state["above"], (800, 80.0)) + + def test_local_reversal_cannot_hide_a_coarser_feasible_point(self): + # 400 meets 55 but 800 dips below it again: the label is still 400. + points = [(100, 20.0), (400, 60.0), (800, 50.0), (1600, 95.0)] + state = qol.crossing_state(points, 55.0) + self.assertEqual(state["above"], (400, 60.0)) + + def test_censored_when_nothing_meets(self): + self.assertEqual(qol.crossing_state(self.POINTS, 99.0)["state"], "censored") + + def test_floor_when_the_first_point_meets(self): + state = qol.crossing_state(self.POINTS, 10.0) + self.assertEqual(state["state"], "floor") + self.assertEqual(state["above"], (100, 20.0)) + + +class RefineTest(unittest.TestCase): + def test_refine_scales_are_strictly_inside_and_geometric(self): + inner = qol.refine_scales(1000, 8000, 3) + self.assertTrue(all(1000 < s < 8000 for s in inner)) + self.assertEqual(inner, sorted(set(inner))) + # Geometric spacing: successive ratios are near-equal. + ratios = [b / a for a, b in zip([1000] + inner, inner + [8000])] + self.assertLess(max(ratios) / min(ratios), 1.05) + + def test_refine_scales_empty_for_adjacent_bracket(self): + self.assertEqual(qol.refine_scales(1000, 1001, 3), []) + + +class InterpTest(unittest.TestCase): + def test_interp_matches_exact_power_law(self): + # loss = 60 * (scale/1000)^-0.8, the curve the encoder tests use. + def score(scale: float) -> float: + return 100.0 - 60.0 * (scale / 1000.0) ** -0.8 + + target = 85.0 + exact = 1000.0 * (60.0 / (100.0 - target)) ** (1.0 / 0.8) + below = (2000, score(2000)) + above = (40000, score(40000)) + got = qol.interp_crossing(below, above, target) + self.assertLess(abs(math.log(got / exact)), 1e-6) + + def test_interp_falls_back_to_midpoint_on_unordered_loss(self): + got = qol.interp_crossing((1000, 90.0), (2000, 80.0), 85.0) + self.assertLess(abs(math.log(got / math.sqrt(1000 * 2000))), 1e-9) + + +class BetaTest(unittest.TestCase): + def test_beta_recovers_the_power_law_exponent(self): + points = [ + (s, 100.0 - 60.0 * (s / 1000.0) ** -0.8) + for s in (2000, 4000, 8000, 16000, 32000) + ] + beta = qol.local_beta(points, 8000.0) + self.assertIsNotNone(beta) + self.assertLess(abs(beta - 0.8), 1e-6) + + def test_beta_is_none_for_flat_or_rising_loss(self): + self.assertIsNone(qol.local_beta([(1000, 50.0), (2000, 40.0)], 1500.0)) + self.assertIsNone(qol.local_beta([(1000, 50.0)], 1000.0)) + + +class MergeTest(unittest.TestCase): + def test_merge_prefers_priced_records_and_sorts_by_rung(self): + first = [{"rung": 10, "effective_scale": 11, "score": 50.0, "bytes": None}] + second = [ + {"rung": 10, "effective_scale": 11, "score": 50.0, "bytes": 123}, + {"rung": 5, "effective_scale": 6, "score": 30.0, "bytes": None}, + ] + merged = qol.merge_points([first, second]) + self.assertEqual([r["rung"] for r in merged], [5, 10]) + self.assertEqual(merged[1]["bytes"], 123) + + def test_merge_never_downgrades_a_priced_record(self): + first = [{"rung": 10, "effective_scale": 11, "score": 50.0, "bytes": 123}] + second = [{"rung": 10, "effective_scale": 11, "score": 50.0, "bytes": None}] + merged = qol.merge_points([first, second]) + self.assertEqual(merged[0]["bytes"], 123) + + +class LabelsTest(unittest.TestCase): + RAW = { + "schema": qol.RAW_SCHEMA, + "targets": [50.0, 99.0], + "image": { + "id": "img", + "family_id": "fam", + "split": "calibration", + "class": "photo", + "variant_id": "img", + "source_capture_id": None, + }, + "points": [ + {"rung": 99, "effective_scale": 100, "score": 20.0, "bytes": None}, + {"rung": 399, "effective_scale": 400, "score": 45.0, "bytes": 900}, + {"rung": 799, "effective_scale": 800, "score": 60.0, "bytes": 1400}, + {"rung": 1599, "effective_scale": 1600, "score": 95.0, "bytes": None}, + ], + } + + def test_crossed_and_censored_rows(self): + rows = qol.labels_for_raw(self.RAW) + crossed = rows[0] + self.assertEqual(crossed["state"], "crossed") + self.assertEqual(crossed["label_rung"], 799) + self.assertEqual(crossed["label_bytes"], 1400) + self.assertTrue(400 < crossed["crossing_scale"] < 800) + self.assertEqual( + [n["bytes"] for n in crossed["neighbor_bytes"]], [900, 1400] + ) + censored = rows[1] + self.assertEqual(censored["state"], "censored") + self.assertIsNone(censored["label_rung"]) + self.assertEqual(censored["top_score"], 95.0) + + +if __name__ == "__main__": + unittest.main() diff --git a/JPXL/tools/tests/test_quality_predictor_v2.py b/JPXL/tools/tests/test_quality_predictor_v2.py new file mode 100644 index 00000000..63b971a8 --- /dev/null +++ b/JPXL/tools/tests/test_quality_predictor_v2.py @@ -0,0 +1,200 @@ +"""Unit tests for the pure fitting helpers of quality_predictor_v2.py.""" + +from __future__ import annotations + +import math +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +import quality_predictor_v2 as qpv2 + + +class SolveTest(unittest.TestCase): + def test_solve_linear_recovers_exact_solution(self): + matrix = [[2.0, 1.0], [1.0, 3.0]] + rhs = [5.0, 10.0] + x = qpv2.solve_linear(matrix, rhs) + self.assertAlmostEqual(x[0], 1.0, places=9) + self.assertAlmostEqual(x[1], 3.0, places=9) + + +class OlsTest(unittest.TestCase): + def test_ols_recovers_a_noiseless_line(self): + xs = [[float(i)] for i in range(10)] + ys = [3.0 + 2.0 * i for i in range(10)] + w = qpv2.fit_ols(xs, ys, [1.0] * 10, ridge=0.0) + self.assertAlmostEqual(w[0], 3.0, places=6) + self.assertAlmostEqual(w[1], 2.0, places=6) + + +class QuantileTest(unittest.TestCase): + def test_median_fit_splits_asymmetric_noise(self): + # y = 5 + x with one gross positive outlier; the median fit should + # stay near the line while OLS is dragged upward. + xs = [[float(i)] for i in range(20)] + ys = [5.0 + i for i in range(20)] + ys[10] += 50.0 + w = qpv2.fit_quantile(xs, ys, [1.0] * 20, tau=0.5) + predicted = qpv2.predict(w, [10.0]) + self.assertLess(abs(predicted - 15.0), 1.0) + + def test_higher_tau_sits_above_lower_tau(self): + # Constant-feature rows with spread: quantiles must order. + xs = [[0.0]] * 50 + ys = [float(i) for i in range(50)] + w50 = qpv2.fit_quantile(xs, ys, [1.0] * 50, tau=0.5) + w90 = qpv2.fit_quantile(xs, ys, [1.0] * 50, tau=0.9) + self.assertGreater(qpv2.predict(w90, [0.0]), qpv2.predict(w50, [0.0]) + 5.0) + + +class LogisticTest(unittest.TestCase): + def test_degenerate_single_class_saturates(self): + w = qpv2.fit_logistic([[0.0]] * 5, [0.0] * 5, [1.0] * 5) + self.assertLess(w[0], -5.0) + w = qpv2.fit_logistic([[0.0]] * 5, [1.0] * 5, [1.0] * 5) + self.assertGreater(w[0], 5.0) + + def test_separable_classes_order_by_feature(self): + xs = [[-1.0]] * 10 + [[1.0]] * 10 + ys = [0.0] * 10 + [1.0] * 10 + w = qpv2.fit_logistic(xs, ys, [1.0] * 20) + self.assertGreater(w[1], 0.5) + + +class StandardizerTest(unittest.TestCase): + def test_median_and_mad(self): + rows = [[float(v)] for v in (1, 2, 3, 4, 100)] + centers, scales = qpv2.robust_standardizer(rows) + self.assertEqual(centers[0], 3.0) + # MAD of [2,1,0,1,97] sorted -> 1; scale = 1.4826. + self.assertAlmostEqual(scales[0], 1.4826, places=4) + + +class BytesTest(unittest.TestCase): + ROW = { + "label_bytes": 1000, + "neighbor_bytes": [ + {"effective_scale": 500, "bytes": 800}, + {"effective_scale": 1000, "bytes": 1000}, + {"effective_scale": 2000, "bytes": 1300}, + ], + } + + def test_interpolates_between_priced_points(self): + ratio = qpv2.simulate_bytes_ratio(self.ROW, 1000.0) + self.assertAlmostEqual(ratio, 1.0, places=6) + finer = qpv2.simulate_bytes_ratio(self.ROW, 2000.0) + self.assertAlmostEqual(finer, 1.3, places=6) + + def test_none_without_enough_pricing(self): + row = {"label_bytes": 1000, "neighbor_bytes": [{"effective_scale": 1000, "bytes": 1000}]} + self.assertIsNone(qpv2.simulate_bytes_ratio(row, 1000.0)) + + +class FamilyWeightTest(unittest.TestCase): + def test_each_family_carries_equal_total_weight(self): + rows = [ + {"family_id": "a"}, + {"family_id": "a"}, + {"family_id": "b"}, + ] + weights = qpv2.family_weights(rows) + self.assertAlmostEqual(weights[0] + weights[1], 1.0) + self.assertAlmostEqual(weights[2], 1.0) + + +class GateTest(unittest.TestCase): + def test_gate_verdicts(self): + metrics = { + "median_abs_ln_error": 0.08, + "p90_abs_ln_error": 0.25, + "p99_abs_ln_error": 0.5, + "first_plan_success": 0.9, + "simulated_byte_geomean": 1.01, + } + verdicts = qpv2.gate_verdicts(metrics) + self.assertTrue(verdicts["quick_screen"]["passed"]) + self.assertTrue(verdicts["production"]["passed"]) + metrics["p90_abs_ln_error"] = 0.5 + verdicts = qpv2.gate_verdicts(metrics) + self.assertFalse(verdicts["quick_screen"]["passed"]) + self.assertFalse(verdicts["production"]["passed"]) + + +class EndToEndTest(unittest.TestCase): + def synthetic_rows(self) -> list[dict]: + # Crossing law: ln(scale) = 6 + 1.5*ln(loss ratio) + 0.8*ln(q50) style — + # exactly linear in the trainer's feature space, so LOFO must nail it. + rows = [] + for fam in range(8): + q50 = 10.0 ** (-6 + fam * 0.4) + flat = 0.1 * fam / 8.0 + sf = { + "width": 640, + "height": 480, + "grayscale": False, + "luma_variance_q10": q50 * 0.5, + "luma_variance_q50": q50, + "luma_variance_q90": q50 * 2.0, + "chroma_variance_q50": q50 * 0.7, + "flat_fraction": flat, + "edge_proxy": q50, + } + for t_idx, target in enumerate(qpv2.TARGET_KNOTS): + # Linear in ln(loss) — the pooled model's family — plus a + # deterministic residual spread so the quantile fits have + # something to separate on. + y = ( + 6.0 + - 0.35 * math.log(qpv2.EPS + q50) + - 0.9 * (math.log(100.0 - target) - qpv2.X_T_CENTER) + + 0.15 * math.sin(2.7 * (fam * 7 + t_idx)) + ) + scale = math.exp(y) + rows.append( + { + "image_id": f"img{fam}", + "family_id": f"fam{fam}", + "split": "calibration", + "class": "photo", + "target": target, + "state": "crossed", + "crossing_scale": scale, + "label_rung": int(scale), + "beta": 0.8, + "label_bytes": 1000, + "neighbor_bytes": [ + {"effective_scale": scale / 1.3, "bytes": 900}, + {"effective_scale": scale, "bytes": 1000}, + {"effective_scale": scale * 1.3, "bytes": 1100}, + ], + "source_features": sf, + } + ) + return rows + + def test_lofo_recovers_a_linear_law(self): + rows = self.synthetic_rows() + metrics = qpv2.evaluate_lofo(rows, "source") + self.assertIsNotNone(metrics["median_abs_ln_error"]) + # The law is representable; the deterministic +-0.15 spread bounds + # the median error from above, and the tau=0.9 candidate should + # clear most of that spread. + self.assertLess(metrics["median_abs_ln_error"], 0.2) + self.assertGreaterEqual(metrics["first_plan_success"], 0.75) + + def test_candidate_quantile_sits_above_the_median(self): + model = qpv2.train_all(self.synthetic_rows(), "source") + for target, knot in model["knots"].items(): + self.assertGreater( + knot["candidate"][0], + knot["median"][0], + f"tau=0.9 must sit above tau=0.5 at target {target}", + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/README.md b/README.md index 8bfe60c0..04c5b722 100644 --- a/README.md +++ b/README.md @@ -121,13 +121,18 @@ reported is the score the written file has: the reconstruction is bit-exact with `jpxl-decode`, and the scorer quantizes to the frame's bit depth the way a viewer would see it. -- **Search.** A calibrated table (`jpxl features` source statistics → starting - quantizer) picks the first probe; the controller brackets the target by - extrapolating the measured loss slope, aims at the log-loss crossing, and - then attaches entropy coding to only the coarsest qualifying candidates and - keeps the smallest exact stream. Probes reuse the first candidate's cover and - chroma-from-luma; a finalist far from that anchor is re-planned fresh and - re-scored. +- **Search.** A trained one-shot model (pooled quantile regression over + `jpxl features` source statistics plus a DCT8 transform summary; the + `one-shot-controller` feature, on by default) picks the first probe when the + input is in its confidence envelope, and falls back to the calibrated + starting table when it is not (tiny frames, out-of-envelope features). From + that seed the controller brackets the target by extrapolating the measured + loss slope, aims at the log-loss crossing, and then attaches entropy coding + to only the coarsest qualifying candidates and keeps the smallest exact + stream. Probes reuse the first candidate's cover and chroma-from-luma; a + finalist far from that anchor is re-planned fresh and re-scored. Every + emission is verified at full resolution against the canonical scorer before + it is reported. - **Budgets.** Fast: at most 3 scored probes and 2 exact prices. Balanced: 5 and 3. These are hard caps; there is no hidden exhaustive fallback. - **Reporting.** Every perceptual encode prints one line, @@ -139,7 +144,7 @@ a viewer would see it. `under_target_work_cap`, `saturated_floor`, `routed_to_lossless` (score 100) and `unsupported_too_small` (below the metric's 8×8 floor). Setting `JPXL_QUALITY_TRACE=<path>` appends a machine-readable - `jpxl.quality-trace/1` record per encode: source features, the predicted + `jpxl.quality-trace/2` record per encode: source features, the predicted rung, every probe's quantizer/score/bytes, and wall time by phase. - **Determinism.** The same input produces the same codestream across worker counts: the metric reduces fixed-size row bands in fixed order, the renderer @@ -204,7 +209,24 @@ The project has two distinct comparison modes. They must not be conflated. a quality claim. The exhaustive JPXL `Quality` preset is deliberately much slower and is not the speed-parity path. -The tracked harness alternates the encoders on identical P6 PPM inputs, +Two reproduction entry points exist. The portable, dependency-light one is +`JPXL/tools/bench_vs_libjxl.sh`: given a corpus and the oracle binaries from +`tools/setup-oracles.sh`, it emits a `bytes / bpp / SSIMULACRA2 / wall` row per +image and setting for both encoders — every stream decoded and scored with the +*same* in-tree production SSIMULACRA2 — behind a provenance header (UTC date, +host, each binary's version and sha256, the exact flags, per-input hashes and +dimensions), with optional `--jsonl` output. It grades JPXL alone, clearly +marked, when no runnable oracle is present on the host. + +```sh +cd JPXL +# JPXL quality targets vs cjxl distances, one identical PPM corpus, 4 threads: +tools/bench_vs_libjxl.sh --quality "70 85 90" --distance "3.0 1.5 1.0" \ + --effort balanced --threads 4 --runs 3 --jsonl bench.jsonl <corpus-dir> +``` + +The fuller, tracked harness (`tools/codec_compare.py`, and the Windows-native +`tools/compare-libjxl.ps1`) alternates the encoders on identical P6 PPM inputs, uses an explicit equal thread count, decodes both streams with `djxl`, and records provenance, hashes, bitrate, quality metrics, and timing dispersion. @@ -250,7 +272,18 @@ committing test images or generated streams. ## License -JPXL is licensed under the [MIT License](LICENSE). +JPXL is licensed under the [MIT License](LICENSE) (mirrored at +[JPXL/LICENSE-MIT](JPXL/LICENSE-MIT)). Every workspace crate declares MIT. + +Every third-party dependency is permissive and MIT-compatible, and the +dependency graph contains no copyleft (no GPL/LGPL/AGPL/MPL/CDDL). The +full crate-by-crate audit — including the handful of dependencies that offer +BSD-2/3-Clause rather than MIT, and the measurement-only metrics that stay off +by default — is in +[JPXL/docs/LICENSING-AUDIT.md](JPXL/docs/LICENSING-AUDIT.md). The gitignored +libjxl oracle checkout (BSD-3-Clause), the ISO/IEC standards documents, and the +local test images are working material only: they are not part of the build and +are not distributed. JPEG XL may be subject to patent claims; this repository’s software license does not provide patent advice or a patent grant. diff --git a/docs/generated/ACTIVE-WORK.md b/docs/generated/ACTIVE-WORK.md index 15fc4998..2d7f239c 100644 --- a/docs/generated/ACTIVE-WORK.md +++ b/docs/generated/ACTIVE-WORK.md @@ -1,5 +1,5 @@ <!-- GENERATED BY AKR — DO NOT EDIT - source-graph: sha256:29947c3fadfc107253854896393926aa0698b43a301121176a032e4f7a7260f4 + source-graph: sha256:2e53454b4abc5a2e7803995f3fb00745b536d02f33e33c9607223cdf0dd3935b tool: akr 0.3.3 --> @@ -41,9 +41,15 @@ Define and screen an independently authored JPXL-side input for normalized seman ## [Perceptual quality controller: SSIMULACRA2 score target for Fast and Balanced, gated Quality effort](ROADMAP.md#perceptual-quality-controller-ssimulacra2-score-target-for-fast-and-balanced-gated-quality-effort) `@jpegxl-rs.track.perceptual-quality-controller/1` +### One-shot program: shadow-first common-case one-shot quality controller + +`active` · `@jpegxl-rs.work.pqc-one-shot-controller/3` · part of `@jpegxl-rs.track.perceptual-quality-controller/1` + +Adopt the 2026-08-24 advisor memo's common-case one-shot design to close the open Balanced wall requirement. PR 1-4 (trace/2, production-endpoint oracle labels, source-only and transform-summary shadow models) are implemented and measured; PR 5 (model-seeded first plan, one slope correction, exact-navigator continuation under unchanged caps with canonical verification before every emission) is implemented and, per the user's 2026-08-24 directive and the 2026-08-25 promotion A/B, is now the DEFAULT controller seed (`one-shot-controller` in the policy crate's default features; building without default features restores the legacy table seed). + ### PQC usable efforts: reduce Fast/Balanced wall time and peak memory -`active` · `@jpegxl-rs.work.pqc-usable-efforts-cost/5` · part of `@jpegxl-rs.track.perceptual-quality-controller/1` +`active` · `@jpegxl-rs.work.pqc-usable-efforts-cost/9` · part of `@jpegxl-rs.track.perceptual-quality-controller/1` · **at risk** Profile and reduce the production Fast and Balanced perceptual path's full-frame render/metric allocation and rescue-probe cost. Work only on usable efforts: do not spend measurement time on the feature-gated Quality reference effort. Pure scorer, renderer, and lifetime changes must preserve Fast/Balanced codestream bytes; any deliberate search-policy change requires the standing Contract B screen. @@ -51,10 +57,29 @@ Profile and reduce the production Fast and Balanced perceptual path's full-frame | Check | Method | Verdict | | --- | --- | --- | -| `memory-12mp` | observation | **satisfied** by `@jpegxl-rs.evidence.pqc-low-memory-12mp-2026-08-23/1` | -| `production-identity` | command | **satisfied** by `@jpegxl-rs.evidence.pqc-low-memory-production-identity-2026-08-23/1` | +| `memory-12mp` | observation | **satisfied** by `@jpegxl-rs.evidence.pqc-fused-render-memory-12mp-2026-08-25/1` | +| `production-identity` | command | **satisfied** by `@jpegxl-rs.evidence.pqc-fused-render-production-identity-2026-08-25/1` | | `wall-anchors` | observation | not satisfied — no evidence | -| `workspace-gates` | command | **satisfied** by `@jpegxl-rs.evidence.pqc-low-memory-workspace-gates-2026-08-23/1` | +| `workspace-gates` | command | **satisfied** by `@jpegxl-rs.evidence.pqc-fused-render-workspace-gates-2026-08-25/1` | + +> **At risk** at depth 1 via `supported_by` → `@jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1` (stale: `watches "JPXL/crates/jpxl-plan-render/src/lib.rs"` was matched by `b0c128c7`, which touched `JPXL/crates/jpxl-plan-render/src/lib.rs`.). See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#pqc-usable-efforts-reduce-fastbalanced-wall-time-and-peak-memory). + +## [VarDCT encoder (M1-M8)](ROADMAP.md#vardct-encoder-m1-m8) `@jpegxl-rs.track.vardct-encoder/1` + +### Lossless JPEG bitstream recompression: carry JPEG1 coefficients into JPEG XL directly instead of decode-and-reencode + +`active` · `@jpegxl-rs.work.jpeg-bitstream-recompression/1` · part of `@jpegxl-rs.track.vardct-encoder/1` + +Carry JPEG1 quantized DCT coefficients into JPEG XL directly (18181-2 s9.11 jbrd + Annex A) instead of decode-and-reencode, through the phased plan in JPXL/docs/jpeg-recompression-plan.md: A jpeg codec roundtrip, B coefficient carriage, C bit-exact reconstruction, D density, E oracle interop. + +**Acceptance** — 1 of 4 satisfied + +| Check | Method | Verdict | +| --- | --- | --- | +| `phase-a-jpeg-codec-roundtrip` | command | **satisfied** by `@jpegxl-rs.evidence.jpeg-phase-a-roundtrip-2026-08-25/1` | +| `phase-b-coefficient-carriage` | command | not satisfied — no evidence | +| `phase-c-bit-exact-reconstruction` | command | not satisfied — no evidence | +| `phase-e-oracle-interop` | command | not satisfied — no evidence | ## Unparented @@ -68,17 +93,17 @@ the optimization decisions and gates. **Plan of record for** [Encoder optimization pass](ROADMAP.md#encoder-optimization-pass) `@jpegxl-rs.track.encoder-optimization/1` -### Publish-ready README, reproducible libjxl comparison, and MIT-only licensing - -`proposed` · `@jpegxl-rs.work.publish-readme-benchmark-mit/1` +### EXIF box and colour-space signalling for the archive pipeline -Create a publish-ready repository surface: an accurate current-state README, a reproducible encoder quality-and-speed comparison with libjxl, MIT-only project licensing, and the local Git remote needed to publish the resulting verified work to LegeApp/JPXL. +`proposed` · `@jpegxl-rs.work.pipeline-metadata-seam/1` -**Acceptance** — 0 of 4 satisfied - -| Check | Method | Verdict | -| --- | --- | --- | -| `comparison-reproducible` | command | not satisfied — no evidence | -| `mit-only` | command | not satisfied — no evidence | -| `readme-current-state` | manual | not satisfied — no evidence | -| `workspace-gates` | command | not satisfied — no evidence | +Give the archive pipeline the two metadata seams it needs from the +encoder: a Part 2 Exif container box exposed as Encoder::with_exif +(validated raw-TIFF payload, container implied), and declarative +colour-space signalling (sRGB, linear sRGB, Display P3, Rec.2020) +through the Annex E ColourEncoding bundle on the lossless Modular +path, with lossy VarDCT rejecting non-sRGB input by typed error +because XYB and the metric are defined on sRGB. Downstream, +raw-autotune supplies the TIFF blob and colour tag and openarc +routes them through archiving; the contract is a bare TIFF blob, +never an APP1 wrapper. diff --git a/docs/generated/CURRENT-STATE.md b/docs/generated/CURRENT-STATE.md index b67169c6..3d12e9a0 100644 --- a/docs/generated/CURRENT-STATE.md +++ b/docs/generated/CURRENT-STATE.md @@ -1,5 +1,5 @@ <!-- GENERATED BY AKR — DO NOT EDIT - source-graph: sha256:29947c3fadfc107253854896393926aa0698b43a301121176a032e4f7a7260f4 + source-graph: sha256:2e53454b4abc5a2e7803995f3fb00745b536d02f33e33c9607223cdf0dd3935b tool: akr 0.3.3 --> @@ -693,6 +693,18 @@ What the corpus patch (K.3) streams contain, and how the patches path was proved > **Stale** — `watches "JPXL/crates/jpxl-decode/**"` was matched by `b12f4fca`, which touched `JPXL/crates/jpxl-decode/src/frame/upsampling.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#patches-k3-corpus-content-and-proof-path). +### PGO gain re-measured at 2-8%; Phase 8.5 magnitudes are historical + +`verified` · `@jpegxl-rs.observation.pgo-gain-remeasured-2026-08-26/1` + +Re-measured at this commit, the combined release-final (fat LTO) +plus PGO cycle gains only 2-8% over the thin-LTO release profile on +the canonical training and unseen images. The Phase 8.5 PGO +magnitudes recorded around 2026-08-14 (36-53%) predate the +Opt-F..P rounds and the 2026-08-25 fused probe-render work and no +longer describe HEAD; treat them as historical, and re-verify +before citing PGO as a major lever. + ### Phase-0 measured multiplicity at 4000x3000 (threads 1 vs auto) `verified` · `@jpegxl-rs.observation.phase0-diag-12mp-2026-08-06/1` @@ -736,7 +748,7 @@ At 95c0217, phase/rung tracing showed that the previously reported 24 Fast price **derived_from** `@jpegxl-rs.work.arch-phase4m-fast-full-handoff/1` -> **Stale** — `watches "JPXL/crates/jpxl-cli/src/main.rs"` was matched by `23635f68`, which touched `JPXL/crates/jpxl-cli/src/main.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#phase-4m-trace-localises-high-rate-undershoot-to-lf-fill-ordering). +> **Stale** — `watches "JPXL/crates/jpxl-cli/src/main.rs"` was matched by `0fd5b3b0`, which touched `JPXL/crates/jpxl-cli/src/main.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#phase-4m-trace-localises-high-rate-undershoot-to-lf-fill-ordering). ### jxl-oxide 0.12.6 narrows LfQuant at the signed-16-bit boundary @@ -758,6 +770,14 @@ Phase 5G promoted AQ Off, quant_lf 8, and no LF-fill after improving Butteraugli > **Stale** — `watches "JPXL/crates/jpxl-encode-policy/src/field.rs"` was matched by `b25beda2`, which touched `JPXL/crates/jpxl-encode-policy/src/field.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#active-epf-remains-production-special-transforms-need-a-better-selector). +### Parallelizing the serial varblock render and linearization cut 12 MP quality wall 21%; ratio 3.11x -> 2.46x on the measuring host + +`verified` · `@jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1` · scope `path "JPXL/crates/jpxl-encode-policy/src/quality.rs"`, `path "JPXL/crates/jpxl-perceptual/src/evaluator.rs"`, `path "JPXL/crates/jpxl-plan-render/**"` + +Per-probe phase attribution on the locked 12 MP anchor (fast-debug, comparing 1-thread vs 8-thread walls to identify serial stages) showed the reference side of SSIMULACRA2 already precomputed once per search (PrecomputedReference with ReferenceRetention::Moments at <=24 MP); the serial per-probe stages were varblock reconstruction (500-730 ms; a 12 MP frame is one LF group, so group-level parallelism contributed nothing) and transfer-curve linearization (105-180 ms). After banding both over the EncodeExecutor (map_ordered varblock chunks of 2048 with serial disjoint scatter; banded LUT lookup), varblocks measure ~235 ms and linearize ~18 ms; per-probe render+metric fell ~1.3 s to ~0.77 s. Release wall on the anchor at 8 threads: 4.923 s -> 3.906 s (-21%), matched-rate ratio 3.11x -> 2.46x on this host (the recorded 6.31x/8.02x baselines were another host/state; the relative move is the reliable signal). Byte-identical streams on six image/target cells, identical canonical scores, cross-thread determinism at threads 1/4/8 in both profiles, peak working set unchanged at 1,899 MB, all touched-crate gates green. Remaining per-probe residual is co-dominated by the still-serial varblock sample scatter (~part of 235 ms) and the already-parallel metric (~380 ms); the largest further headroom is score-changing (navigate on a reduced pyramid or downscaled reconstruction, full-resolution only for final verification), which requires a Contract B promotion A/B. + +> **Stale** — `watches "JPXL/crates/jpxl-plan-render/src/lib.rs"` was matched by `b0c128c7`, which touched `JPXL/crates/jpxl-plan-render/src/lib.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#parallelizing-the-serial-varblock-render-and-linearization-cut-12-mp-quality-wall-21-ratio-311x---246x-on-the-measuring-host). + ### Perceptual encodes of 50 MP sources reach ~12 GB RSS; two concurrent measurement runs OOM-killed a 31 GB host twice `verified` · `@jpegxl-rs.observation.pqc-memory-50mp-oom-2026-08-22/1` · scope `path "JPXL/crates/jpxl-encode-policy/src/quality.rs"`, `path "JPXL/crates/jpxl-perceptual/**"`, `path "JPXL/tools/codec_compare.py"` @@ -768,6 +788,16 @@ The kernel OOM-killed jpxl twice on 2026-08-22 (12:54 anon-rss 9.9 GB; 13:28 ano > **Stale** — `watches "JPXL/crates/jpxl-perceptual/src/ssimulacra2.rs"` was matched by `2425c0b2`, which touched `JPXL/crates/jpxl-perceptual/src/ssimulacra2.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#perceptual-encodes-of-50-mp-sources-reach-12-gb-rss-two-concurrent-measurement-runs-oom-killed-a-31-gb-host-twice). +### The one-shot falsification screen fails on corpus coverage, not on the design: transform features halve the error, the tail is two under-represented classes + +`verified` · `@jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/2` · scope `path "JPXL/crates/jpxl-encode-policy/src/**"`, `path "JPXL/tools/**"` + +On the expanded 56-image / 38-family corpus with production-endpoint oracle labels (fresh Balanced pixel plans over the complete effective ladder; zero censored rows even at target 95, so the old predictor's 81-94% high-target saturation was an artifact of its HfMul=1 ladder cap), the memo's pre-PR5 falsification screen fails: best leave-one-family-out median |ln(predicted/oracle crossing)| is 0.220 with p90 0.805 (bounds: quick p90<=0.45; production median<=0.10, p90<=0.30). Three findings shape the path forward. (1) Transform-domain features materially help, as the memo predicted: source-only p90 1.28 vs source+transform 0.81, median 0.42 vs 0.22, first-or-one-correction success 1.00. (2) The failure is class coverage, not model form: photo, photo-scene, text-screenshot, noise-lowlight, line-art and grayscale classes sit at held-out p90 0.36-0.67 (near or past the quick bound), while saturated (p90 3.58) and sky-noise gradient (p90 4.07) dominate the tail - both effectively class-held-out because their families carry near-unique content. (3) The fallback discipline works: with honest intervals the simulated controller routes 99% of requests to the exact path, keeping bytes at 1.0016x but eliminating the wall win (expected reconstructions 3.98 vs the current median 4). Per the memo, production controller work stops here until the corpus grows (>=25 independent families per critical class; saturated and banding-stress content first); the PR5 mechanism exists behind the off-by-default `one-shot-controller` cargo feature and remains inert without a gate-passing model. + +**supersedes** `@jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/1` · **derived_from** `@jpegxl-rs.work.pqc-one-shot-controller/3` · **verified_by** `@jpegxl-rs.evidence.pqc-one-shot-falsification-screen-2026-08-24/1` + +> **Stale** — `watches "JPXL/crates/jpxl-encode-policy/src/quality_predictor*.rs"` was matched by `b57c9d40`, which touched `JPXL/crates/jpxl-encode-policy/src/quality_predictor_v2.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#the-one-shot-falsification-screen-fails-on-corpus-coverage-not-on-the-design-transform-features-halve-the-error-the-tail-is-two-under-represented-classes). + ### PR 4 on the development split: the score floor holds in 150/150 encodes, photographs beat cjxl -e7 at matched score, synthetic text and line art trail badly, and the quality path costs ~3.5x the rate path's wall time `verified` · `@jpegxl-rs.observation.pqc-pr4-development-split-2026-08-22/1` · scope `path "JPXL/crates/jpxl-encode-policy/src/quality.rs"`, `path "JPXL/crates/jpxl-perceptual/**"`, `path "JPXL/crates/jpxl-plan-render/**"`, `path "JPXL/tools/codec_compare.py"` @@ -786,7 +816,25 @@ The earlier @jpegxl-rs.evidence.pqc-pr4-holdout-byte-neutral-ss2-2026-08-22/1 in **derived_from** `@jpegxl-rs.evidence.pqc-pr4-holdout-byte-neutral-ss2-2026-08-22/1`, `@jpegxl-rs.evidence.pqc-quality-rate-curve-exact-10of13-2026-08-23/1`, `@jpegxl-rs.observation.ssimulacra2-f32-recursion-ripple-2026-08-22/1` -> **Stale** — `watches "JPXL/crates/jpxl-cli/src/main.rs"` was matched by `2425c0b2`, which touched `JPXL/crates/jpxl-cli/src/main.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#pr4s-byte-neutral-rate-curve-conclusion-mixed-ssimulacra2-implementations-and-is-not-a-like-for-like-baseline). +> **Stale** — `watches "JPXL/crates/jpxl-cli/src/main.rs"` was matched by `0fd5b3b0`, which touched `JPXL/crates/jpxl-cli/src/main.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#pr4s-byte-neutral-rate-curve-conclusion-mixed-ssimulacra2-implementations-and-is-not-a-like-for-like-baseline). + +### Full PR7 holdout closes the OOM tail and rejects Quality promotion + +`verified` · `@jpegxl-rs.observation.pqc-pr7-full-holdout-quality-rejected-2026-08-24/1` · scope `path "JPXL/crates/jpxl-encode-policy/src/policy_bank.rs"`, `path "JPXL/crates/jpxl-encode-policy/src/quality.rs"`, `path "JPXL/crates/jpxl-encode-policy/src/reducer.rs"`, `path "JPXL/crates/jpxl-perceptual/src/**"` + +The current low-memory implementation completed the 27 formerly missing Quality cells and 26 companion Balanced rows, closing all 91 cells per effort on the 13-image locked holdout. The terminal reducer is safe and bounded (zero floor/decoder failures, at most six evaluations, final/before bytes geomean 0.991552), but public Quality promotion is rejected: matched-score bytes pass overall at 0.958292, while Contract B fails at Butteraugli mean ratio 1.047881 and worst pnorm3 ratio 1.868818. The added cells make the diagnosis sharper rather than reversing it: photos are nearly byte-neutral at 0.991367 and pass the BA/pnorm3 guards; non-photo byte wins coexist with saturated/tiny/text guard losses. Keep Quality feature-gated and keep its policy bank/reducer out of Fast and Balanced; a future promotion attempt needs a different quality policy or metric guidance, not relaxation of the gate. + +**derived_from** `@jpegxl-rs.evidence.pqc-pr7-quality-promotion-rejected-2026-08-22/1`, `@jpegxl-rs.evidence.pqc-pr7-reducer-holdout-partial-2026-08-22/1` · **verified_by** `@jpegxl-rs.evidence.pqc-pr7-holdout-complete-2026-08-24/1`, `@jpegxl-rs.evidence.pqc-pr7-quality-promotion-full-2026-08-24/1` + +> **Stale** — `watches "JPXL/crates/jpxl-encode-policy/src/quality.rs"` was matched by `b57c9d40`, which touched `JPXL/crates/jpxl-encode-policy/src/quality.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#full-pr7-holdout-closes-the-oom-tail-and-rejects-quality-promotion). + +### Quiet-host wall anchors: 12 MP 5.42x (4t) / 4.25x (8t); the quality path's fixed costs alone sit near 2x, so the 2.0x target is unreachable by search improvements + +`verified` · `@jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25/1` · scope `path "JPXL/crates/jpxl-encode-policy/src/quality.rs"`, `path "JPXL/crates/jpxl-perceptual/src/**"`, `path "JPXL/crates/jpxl-plan-render/src/**"` + +Interleaved matched-score timing (wall_current.py recipe: one warm-up per kind, 10-run schedule, 5 GiB address-space cap, host state per run) on a quiet host with the current release binary: 4.3 MP anchor quality/rate 2.021/0.521 s = 3.88x at 4 threads and 1.824/0.566 s = 3.22x at 8; 12 MP anchor 5.464/1.007 s = 5.42x at 4 threads and 3.988/0.938 s = 4.25x at 8. The 4-thread rate median (1.007 s) reproduces the 2026-08-23 baseline exactly, so the comparable quiet-host series at 12 MP/4t is 8.00x (2026-08-23) -> 5.42x (after the one-shot promotion cut probes 5->3); the 2.46x recorded on 2026-08-25 divided by an inflated ~1.59 s rate denominator measured under load and is not comparable. Phase attribution (12 MP, 8 threads): plan 366 ms, render_metric 1985 ms (three pixel probes at 642-697 ms each), entropy 79 ms, emit 99 ms, two exact prices 89 ms each; ~1.4 s of the 3.99 s wall is outside the controller (source PPM load, metric reference precompute, feature extraction, output IO). Consequence: the quality path's fixed costs excluding all scored probes (~1.83 s at 8 threads) are already 1.95x the rate path, and a perfect single-probe controller would land near 2.7x, so the wall-anchors 2.0x acceptance cannot be met by search improvements alone; it requires reducing per-pixel render+metric cost ~2-3x (score-identity risk) or renegotiating the target. Separately, the band-parallel varblock scatter is exact but wall-neutral at the 12 MP anchor: render_metric 2043/2168 ms (pre-change binary) vs 2107/2057 ms (post), and the 4 MP varblock stage is 45 ms at 4 threads in both trees, so the earlier ~50-100 ms/probe scatter estimate was high. Raw runs, traces, and comparisons are in the kept scratch entry pqc-scatter-20260825 (wall-scatter.json, trace-*.jsonl). + +> **Stale** — `watches "JPXL/crates/jpxl-plan-render/src/lib.rs"` was matched by `c23bc009`, which touched `JPXL/crates/jpxl-plan-render/src/lib.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#quiet-host-wall-anchors-12-mp-542x-4t--425x-8t-the-quality-paths-fixed-costs-alone-sit-near-2x-so-the-20x-target-is-unreachable-by-search-improvements). ### Pre-optimization encode wall-time ladder (release jpxl bench) @@ -891,6 +939,14 @@ tests/rate_proxy_audit.rs now fits actual ~ s*proxy + z*zeros per varblock on th > **Stale** — `watches "JPXL/crates/jpxl-encode-policy/tests/rate_proxy_audit.rs"` was matched by `5fd8e357`, which touched `JPXL/crates/jpxl-encode-policy/tests/rate_proxy_audit.rs`. See [REVIEW-REQUIRED.md](REVIEW-REQUIRED.md#q6-a-per-varblock-interior-zero-term-halves-the-dct8x8-rate-proxy-residuals-p90-but-explains-little-for-dct1632-deferred-rather-than-built-into-the-scoring-kernel). +### Quality-corpus generator now provides 26 saturated and 25 gradient/banding families; locked holdout fixtures byte-unchanged + +`verified` · `@jpegxl-rs.observation.quality-corpus-critical-class-families-2026-08-25/1` · scope `path "JPXL/tools/make-quality-guard-fixtures.py"`, `path "test-set/**"` + +The synthetic quality-guard generator was extended with distinct deterministic content processes so the two critical classes named by the one-shot memo's section 10 now meet its >=25-families floor: saturated 4 -> 26 and gradient/banding-stress 8 -> 25 (each synthetic fixture is its own family). The corpus grew 56 -> 95 fixtures (calibration 43, development 32, holdout 20); the split audit reports no family in more than one split, regeneration is sha256-idempotent on a second run, codec_compare.py manifest-check validates all 95 images, and 16 generator unit tests pass (including a new >=25-families gate). All 13 fixtures pinned by the locked PR4-closure holdout manifest are byte-unchanged, so prior identity and rate-curve evidence remains comparable. SSIMULACRA2 oracle-label sweeps over the new families were deliberately not run; they are the remaining step before the model can be retrained on this coverage. + +**derived_from** `@jpegxl-rs.work.pqc-one-shot-controller/3` + ### The standard's own quant matrices carry a third of the frequency weighting; measuring distortion in quantizer-normalised units cuts Y mispricing 2.65x to 1.86x `verified` · `@jpegxl-rs.observation.quantizer-normalised-distortion-is-the-lever-2026-08-12/1` · scope `path "JPXL/crates/jpxl-conformance/tests/perceptual_frequency.rs"`, `path "JPXL/crates/jpxl-encode-policy/src/lib.rs"`, `path "JPXL/crates/jpxl-encode-policy/src/quantize.rs"` @@ -1197,6 +1253,26 @@ Phase 8.6 rejects raw coefficient-residual allocation as a sufficient perceptual ## Evidence +### jpegxl-rs.evidence.avx2-pooling-bit-identity-2026-08-26 + +`verified` · `@jpegxl-rs.evidence.avx2-pooling-bit-identity-2026-08-26/1` + +accumulate_band's avx2+fma clone is byte-identical to the scalar +walk: the lane test covers remainder lengths 0,1,7,8,9,64,250, +and full quality-path encodes of the canonical images compare +equal with cmp against a pre-change baseline binary. Wall-clock +on the 12 MP quality path improved about 5-8% at 4 threads. + +### One documented command compares JPXL and native cjxl/djxl on the same inputs with recorded provenance + +`verified` · `@jpegxl-rs.evidence.bench-vs-libjxl-reproducible-2026-08-25/1` + +After tools/setup-oracles.sh produced native Linux cjxl/djxl v0.13.0 (196a43d9, pinned in oracle-bin/PINNED_REVISIONS.txt), bench_vs_libjxl.sh emitted a size/bpp/SSIMULACRA2/wall table for jpxl --quality 70/85 and cjxl -d 3.0/1.5/1.0 on three corpus images (2400x1800 photo, 1600x900 text-screenshot, 700x700 radial gradient), every stream decoded and scored by the same in-tree metric, with a provenance header (UTC date, host, binary versions and sha256, flags, per-input sha256/dims) and machine-readable jpxl.bench-vs-libjxl/1 JSONL kept at pqc-scatter-20260825/bench-vs-libjxl.jsonl. Sample: on the photo, jpxl q85 emitted 796,985 bytes at score 85.45 versus cjxl -d1.0 at 804,397 bytes and score 83.41; on text-screenshot cjxl -d3.0 was 4.7x smaller at a higher score, consistent with the recorded VarDCT-on-synthetic gap. + +**Verifies** + +- `completed` `@jpegxl-rs.work.publish-readme-benchmark-mit/1` — check `comparison-reproducible` + ### Focused CLI tests passed: explicit PPM/PGM output uses P6/P5 and incompatible channel layouts error. `verified` · `@jpegxl-rs.evidence.cli-netpbm-subtypes-2026-08-21/1` @@ -1650,6 +1726,16 @@ M1-M8 render as completed in ROADMAP.md and akr check reports no diagnostics and - `completed` `@jpegxl-rs.milestone.history-backfilled/1` — check `history-green` +### jpxl-jpeg roundtrips the full 5,218-file Pol Art archive byte-identically except one provably corrupt file, with typed refusals proven + +`verified` · `@jpegxl-rs.evidence.jpeg-phase-a-roundtrip-2026-08-25/1` + +The deterministic sweep (sorted paths, stride 1 — a superset of the >=500-sample requirement) parsed and re-serialized every JPEG in the archive: 5,217 of 5,218 byte-identical with zero mismatches and zero refusals; coverage of the identical set spans baseline 3,166, progressive 2,049, extended sequential 2, chroma 4:4:4 2,985 / 4:2:0 2,081 / 4:2:2 140 / 4:4:0 6, grayscale, restart intervals 684, Exif 2,419 / ICC 1,558 / XMP 1,396, and 41 files with post-EOI tail data. The single non-roundtripping file is ~15 KB of real data zero-padded to 3.18 MB, rejected by libjpeg's djpeg as premature EOF and refused here with a typed malformed-stream error. All 81 earlier mismatches shared one root cause — libjpeg's non-derivable progressive AC EOB-run splits — fixed by recording the exact run lengths at decode and replaying them flush-on-match at encode; the prog_large.jpg regression fixture proves the recorded runs are load-bearing. Typed refusal of arithmetic, hierarchical, lossless, and 12-bit inputs is proven by 10 dedicated tests; 34 crate tests pass in release. + +**Verifies** + +- `active` `@jpegxl-rs.work.jpeg-bitstream-recompression/1` — check `phase-a-jpeg-codec-roundtrip` + ### M1 - VarDCT plan IR and structural split `verified` · `@jpegxl-rs.evidence.m1-plan-ir/1` @@ -1745,6 +1831,16 @@ docs/generated. - `completed` `@jpegxl-rs.milestone.markdown-retired/1` — check `cutover-wired` +### Workspace licensing surface is MIT-only with no copyleft anywhere in the dependency graph + +`verified` · `@jpegxl-rs.evidence.mit-only-licensing-audit-2026-08-25/1` + +All 12 workspace crates declare MIT via license.workspace; LICENSE and JPXL/LICENSE-MIT carry consistent MIT text; a scan of the full dependency graph (default ~41 external crates plus ~26 behind off-by-default measurement features) found no GPL/LGPL/AGPL/MPL/CDDL/EUPL/SSPL license anywhere; the only non-MIT arms are permissive (moxcms and pxfm BSD-3-Clause/Apache-2.0 on the default path via the image raster adapter; ssimulacra2 BSD-2, butteraugli BSD-3, v_frame BSD-2, imgref CC0/Apache behind feature gates). No ISO text or AGPL-derived content is present in any shipped crate. Full table in JPXL/docs/LICENSING-AUDIT.md. + +**Verifies** + +- `completed` `@jpegxl-rs.work.publish-readme-benchmark-mit/1` — check `mit-only` + ### At 70688ac the lean default (Effort::DEFAULT = 1) reproduces the pre-change default's bytes exactly on all three photo-corpus sizes: small_0p8MP 1,075,465 B sha16 C3C60CEE4D0741D9, mid_4MP 5,157,974 B sha16 5C49AB9020706F0D, large_12MP 9,795,599 B sha16 CA27799D3E7B441D -- all IDENTICAL to the pre-ramp fingerprints. Wall time 267 / 619 / 1165 ms for the default versus 1185 / 5324 / 16919 ms for --effort 7, which reproduces the same three fingerprints and so remains the byte-identical density anchor. Speedup on this corpus is 4.4x / 8.6x / 14.5x for identical output. `verified` · `@jpegxl-rs.evidence.modular-effort-ramp-landed-2026-08-10/1` @@ -4849,6 +4945,57 @@ On the pinned unseen 12 MP photo, finer deterministic quantization scheduling re - `completed` `@jpegxl-rs.work.arch-phase9-quant-scheduling/1` — check `speed` +### jpegxl-rs.evidence.pipeline-metadata-seam-2026-08-26 + +`verified` · `@jpegxl-rs.evidence.pipeline-metadata-seam-2026-08-26/1` + +The metadata seam is verified end to end at this commit: an +appended Exif box round-trips through the decoder's box walk +with the codestream untouched; with_exif rejects payloads +without a TIFF byte-order header; all four ColourSpace values +round-trip through the decoder's ColourEncoding parse; a +greyscale image rejects non-sRGB signalling; lossy VarDCT +returns a typed Unsupported for non-sRGB input. Cross-repo, a +real 25 MB Samsung DNG's 574-byte TIFF blob survived +raw-autotune develop, JPXL encode, openarc archive and extract +byte-identically (recorded on the openarc ledger). + +### Fused-render 12 MP memory: peak 2,008,408 KiB under the 2 GiB ceiling + +`verified` · `@jpegxl-rs.evidence.pqc-fused-render-memory-12mp-2026-08-25/1` + +Balanced q85 on the locked 12 MP anchor after the fused render: peak RSS 2,007,060-2,008,408 KiB over three serialized capped runs, under the 2,097,152 KiB ceiling; output bytes unchanged (1,315,649). Report in kept scratch pqc-metric-cuts-20260825/rss-12mp.json. + +**Verifies** + +- `active` `@jpegxl-rs.work.pqc-usable-efforts-cost/9` — check `memory-12mp` + +### Fused-render production identity: 52/52 matrix and 6/6 pre-change A/B byte-identical + +`verified` · `@jpegxl-rs.evidence.pqc-fused-render-production-identity-2026-08-25/1` + +After the fused depth-quantized probe render, the full locked production-identity matrix is 52/52 byte-identical (threads 1/4, AVX2 on/off) and the 6-cell A/B against the pre-change 1fc3ca7 reference binary is byte-identical across threads 1/4/8 and AVX2 off; the classifier's per-sample equivalence with encode-then-round-trip is additionally pinned by the fused_depth_levels_match_the_srgb_round_trip unit test. Reports in kept scratch pqc-metric-cuts-20260825 (round 2; round-1 reports preserved as *-round1). + +**Verifies** + +- `active` `@jpegxl-rs.work.pqc-usable-efforts-cost/9` — check `production-identity` + +### Fused-render wall anchors: 4.3 MP at 2.50x (8t), 2.0x target still unmet + +`verified` · `@jpegxl-rs.evidence.pqc-fused-render-wall-anchors-2026-08-25/1` + +Interleaved matched-score timing after the fused render: quality/rate 3.06x at 4 threads (load_1m 1.6-1.9) and 2.50x at 8 (load 4.0-4.2) on the 4.3 MP anchor; 4.65x at 4 threads on 12 MP (load 1.9-3.7); the 12 MP 8-thread schedule cell read 3.54x under ambient load 4.0-5.4 and is not representative — quiet manual 12 MP 8-thread quality runs measured 2.39-2.53 s against the schedule's 2.79 s median. Down from 3.40x/2.76x and 5.07x/3.50x before the fusion the same day. The 2.0x wall-anchors target remains unmet on both anchors. Report in kept scratch pqc-metric-cuts-20260825/wall-scatter.json (round 1 preserved as wall-scatter-round1.json). + +### Fused-render workspace gates: fmt, clippy -D warnings, release tests all pass + +`verified` · `@jpegxl-rs.evidence.pqc-fused-render-workspace-gates-2026-08-25/1` + +One strictly chained run at the fused-render tree: cargo fmt --all --check, clippy --workspace --all-targets -D warnings, and the release workspace test suite (76 test groups) all pass, exit 0. + +**Verifies** + +- `active` `@jpegxl-rs.work.pqc-usable-efforts-cost/9` — check `workspace-gates` + ### Low-memory Balanced q85 path meets the 12 MP two-GiB RSS ceiling `verified` · `@jpegxl-rs.evidence.pqc-low-memory-12mp-2026-08-23/1` @@ -4859,7 +5006,8 @@ Kept /usr/bin/time evidence: .agent/scratch/pqc-cost-20260823/large-q85-v5.time. - `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/3` — check `memory-12mp` - `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/4` — check `memory-12mp` -- `active` `@jpegxl-rs.work.pqc-usable-efforts-cost/5` — check `memory-12mp` +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/5` — check `memory-12mp` +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/6` — check `memory-12mp` ### Low-memory quality path passes the full locked production-identity matrix @@ -4871,7 +5019,8 @@ Kept raw rows and report: .agent/scratch/pqc-cost-20260823/README.md. The frozen - `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/3` — check `production-identity` - `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/4` — check `production-identity` -- `active` `@jpegxl-rs.work.pqc-usable-efforts-cost/5` — check `production-identity` +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/5` — check `production-identity` +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/6` — check `production-identity` ### Controlled matched-score wall anchors after low-memory PQC changes @@ -4889,7 +5038,9 @@ The exact chained final-state gate passed after the quality-rate harness and low - `completed` `@jpegxl-rs.work.pqc-quality-rate-curve-closure/5` — check `workspace-gates` - `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/4` — check `workspace-gates` -- `active` `@jpegxl-rs.work.pqc-usable-efforts-cost/5` — check `workspace-gates` +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/5` — check `workspace-gates` +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/6` — check `workspace-gates` +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/7` — check `workspace-gates` ### Low-memory quality path makes 50 MP usable and preserves all xlarge quality streams @@ -4897,6 +5048,70 @@ The exact chained final-state gate passed after the quality-rate harness and low Kept identity audit: .agent/scratch/quality-rate-curve-xlarge-exact-20260823/xlarge-identity-audit.json. Balanced q85 on 8160x6120 completed under a 9 GiB virtual-memory cap at 7,532,608 KiB RSS, replacing the former approximately 12.5 GiB/SIGKILL behavior. Across the three 50 MP images, seven targets, and both production efforts, all 42 current codestream SHA-256 values, byte counts, and controller scores exactly match the historical pre-low-memory rows. +### Balanced q85 on the locked 12 MP anchor peaks at 1.95 GiB RSS, under the 2.0 GiB ceiling + +`verified` · `@jpegxl-rs.evidence.pqc-memory-12mp-2026-08-25/1` + +Three serialized release encodes of the locked 4000x3000 anchor (Balanced q85, 4 threads, RLIMIT_AS 9 GiB, ru_maxrss via wait4) peaked at 2,036,248-2,043,056 KiB, all under the 2,097,152 KiB (2.0 GiB) ceiling, with identical 1,315,649-byte output each run; rows in the kept scratch entry pqc-scatter-20260825/rss-12mp.json. + +**Verifies** + +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/7` — check `memory-12mp` + +### Metric-cuts 12 MP memory: peak 2,008,188 KiB under the 2 GiB ceiling + +`verified` · `@jpegxl-rs.evidence.pqc-metric-cuts-memory-12mp-2026-08-25/1` + +Balanced q85 on the locked 12 MP anchor after the exact metric-path cost cuts: peak RSS 2,007,068-2,008,188 KiB over three serialized capped runs, under the 2,097,152 KiB ceiling and down from 2,043,056 before the cuts; output bytes unchanged (1,315,649). Report in kept scratch pqc-metric-cuts-20260825/rss-12mp.json. + +**Verifies** + +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/8` — check `memory-12mp` + +### Metric-cuts production identity: 52/52 matrix and 6/6 pre-change A/B byte-identical + +`verified` · `@jpegxl-rs.evidence.pqc-metric-cuts-production-identity-2026-08-25/1` + +After the exact metric-path cost cuts, the full locked production-identity matrix is 52/52 byte-identical (threads 1/4, AVX2 on/off), and a 6-cell A/B against the pre-change reference binary (worktree build of 1fc3ca7) is byte-identical across threads 1/4/8 and AVX2 off; reports in kept scratch pqc-metric-cuts-20260825 (identity-scatter-summary.json, identity-ab.json). + +**Verifies** + +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/8` — check `production-identity` + +### Metric-cuts wall anchors: ratios cut ~9-19 percent, 2.0x target still unmet + +`verified` · `@jpegxl-rs.evidence.pqc-metric-cuts-wall-anchors-2026-08-25/1` + +Interleaved matched-score timing after the exact metric-path cost cuts (standing recipe, load_1m 1.8-4.5): quality/rate 3.40x at 4 threads and 2.76x at 8 on the 4.3 MP anchor, 5.07x and 3.50x on 12 MP, from 3.87x/3.22x and 5.42x/4.25x before the cuts (that run at load 5.2-7.7; the ratio columns are the load-robust comparison). Quality medians 1.33/1.17 s (4.3 MP) and 3.65/2.62 s (12 MP). The 2.0x wall-anchors target remains unmet on both anchors. Report in kept scratch pqc-metric-cuts-20260825/wall-scatter.json. + +### Metric-cuts workspace gates: fmt, clippy -D warnings, release tests all pass + +`verified` · `@jpegxl-rs.evidence.pqc-metric-cuts-workspace-gates-2026-08-25/1` + +One strictly chained run at the metric-cuts tree: cargo fmt --all --check, clippy --workspace --all-targets -D warnings, and the release workspace test suite (76 test groups) all pass, exit 0. + +**Verifies** + +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/8` — check `workspace-gates` + +### One-shot falsification screen: fails on corpus coverage, transform features validated + +`verified` · `@jpegxl-rs.evidence.pqc-one-shot-falsification-screen-2026-08-24/1` + +The memo's pre-PR5 falsification screen fails on the expanded 38-family corpus: best model (pooled quantile regression, source+transform features, ridge 3.0) reaches leave-one-family-out median |ln err| 0.220 and p90 0.805 vs the quick-screen p90<=0.45 and production median<=0.10/p90<=0.30 bounds. First-plan success 0.88 and simulated common-case bytes 1.0016 pass their checks, and first-or-one-correction success is 1.00, but the honest uncertainty intervals route 99% of requests to the exact controller (expected reconstructions 3.98, no wall win). Tail is concentrated in saturated (p90 3.58) and sky-noise gradient (p90 4.07) classes - single-content-family coverage - while photo/scene/text/noise/line-art/grayscale sit at p90 0.36-0.67. Standalone transform-summary cost on the 12 MP anchor is 21.9% of a matched-rate encode (fails the 5% bolt-on gate; in-search shared-cache marginal cost unmeasured). Report kept in .agent/scratch/one-shot-qpv2-20260824/qpv2-report.json. + +### One-shot PR 1-5 implementation: all workspace gates pass + +`verified` · `@jpegxl-rs.evidence.pqc-one-shot-pr1-5-gates-2026-08-24/1` + +The one-shot program's PR 1-5 implementation passes every workspace gate: full test suite green (71 suites) in the default build and 155 policy tests under the one-shot-controller feature (including the PR 5 routing test and the trace/2 assertions), clippy clean in both configurations, formatting clean, 75 Python tool tests green (trainer, oracle-label, fixture-generator suites), and the fixture generator's check mode reproduces all 56 corpus PPMs byte-identically with the family split-hygiene guard passing. Default-build behavior is unchanged: the shadow predictor only logs, and the one-shot start is compiled out. + +### One-shot promotion A/B: floor held, bytes and work strictly better in aggregate + +`verified` · `@jpegxl-rs.evidence.pqc-one-shot-promotion-ab-2026-08-25/1` + +Promotion A/B of the one-shot-controller build against the default controller, interleaved on the same machine. Locked 13-image holdout (91 cells): zero floor violations both arms, byte geomean 0.99744 (bound 1.007), wall geomean 0.989, reconstructions 348 vs 339; worst cells are three sub-kilobyte saturated fixtures (max ratio 1.51 = +131 bytes absolute). Never-tuned ext-holdout paintings (50 families, 350 cells): zero floor violations, byte geomean 0.99018, reconstructions 1287 to 1056 (-18%), wall geomean 0.838, worst cell 1.117 at target 95 with a higher achieved score. Achieved-score dips above the floor exist in both arms (1 default, 2 one-shot) - a pre-existing bounded-search artifact, no hard inversion below target anywhere. Wall anchors at q85 vs matched-rate: 4.3 MP 1.36x both arms; 12 MP 8.02x default vs 6.31x one-shot (metric/render remains the dominant residual, as the memo predicted). Raw rows kept in .agent/scratch/one-shot-qpv2-20260824/ (ab-locked-holdout.json, ab-ext-holdout.json). + ### The generator reproduces 47 fixtures (47 PPM + 29 PNG + 47 JSON provenance sidecars, idempotent sha256) and the jpxl.codec-corpus/1 manifest test-set/quality-corpus.json (gitignored) validates; splits are calibration 19, development 15, holdout 13 with no source family in two splits; classes text-screenshot 5, line-art 5, gradient 6, saturated 4, tiny 6, noise-lowlight 1, grayscale 2, photo-scene 7, photo 11. 14 generator unit tests pass. `verified` · `@jpegxl-rs.evidence.pqc-pr0-corpus-manifest-2026-08-22/1` @@ -4907,6 +5122,12 @@ The generator reproduces 47 fixtures (47 PPM + 29 PNG + 47 JSON provenance sidec - `completed` `@jpegxl-rs.work.pqc-pr0-provenance-corpus-calibration/1` — check `corpus-manifest` +### PR0 hard floor: targeted miss-path tests pass + +`verified` · `@jpegxl-rs.evidence.pqc-pr0-hard-floor-tests-2026-08-24/1` + +New miss-path tests all pass: policy-layer UnderTargetWorkCap with rescue accounting (pixel_probes 1 + 1 rescue), facade refuse/lossless/best-effort on the deterministic 64x64 gradient miss at 99.5 Fast (TargetNotMet with LadderSaturated, FallbackLossless bytes equal to the lossless encoder's, best-effort re-scored independently), and 8/8 CLI quality tests including exit-1-with-no-output-file refusal and --quality-fallback validation. + ### 256 fixed-quantizer points (16 calibration images x 16 global_scale rungs 400..73728) scored with the reference SSIMULACRA2 in 604 s produced quality_predictor.rs: 56 table cells over 5 luma-variance x 3 flat-fraction buckets plus a fallback log fit [15.62, -1.60, 0.0099, -2.57]. Leave-one-image-out |ln(pred/actual)| median 0.30, p90 2.68 (cells hold 1-4 images); loss-vs-scale slope median -2.34; saturation at 73728 is 0 up to target 70 and 0.06/0.19/0.81/0.94 at 80/85/90/95. `verified` · `@jpegxl-rs.evidence.pqc-pr0-predictor-table-2026-08-22/1` @@ -4927,6 +5148,12 @@ Both advice documents are registered under sources/external with content hashes - `completed` `@jpegxl-rs.work.pqc-pr0-provenance-corpus-calibration/1` — check `sources-registered` +### PR0 hard floor: workspace and tools gates pass + +`verified` · `@jpegxl-rs.evidence.pqc-pr0-workspace-gates-2026-08-24/1` + +Workspace gates pass after the hard-floor change: full test suite green in both debug and fast-debug profiles (zero failures), clippy emits no warnings, formatting is clean, and the 46 Python tool tests pass (codec_compare now opts into --quality-fallback best-effort so curve points on an unmet target stay measurable). + ### 6/6 facade target tests pass: score 100 is byte-identical to the lossless path, scores outside 0..=100 are rejected, conflicting targets (quality+bpp, bpp+global_scale) are rejected, a perceptual score below 100 runs the quality controller, --global-scale encodes and decodes, and the rate report matches the bytes. `verified` · `@jpegxl-rs.evidence.pqc-pr1-api-semantics-2026-08-22/1` @@ -5126,12 +5353,58 @@ With one shared CandidateSearchContext and the baseline anchor reused by quantiz The single-evaluation Balanced reducer (first batch = half the ranked candidates, max 16384 edits) saves 0.4% bytes geomean at targets 70 and 85 (photos 0.4-0.5%, best 2.9%) with zero floor violations, at +21% / +16% wall geomean but up to +62% on one small image. Average inside the +25% Balanced wall budget, worst case outside, for a small gain: Balanced keeps the reducer off until the locked-holdout gate (matched-score bytes, Butteraugli/PSNR guards) is run and the evaluation made cheaper; the feature-gated Quality effort runs the full reducer. +### PR7 holdout closure workspace and feature gates pass + +`verified` · `@jpegxl-rs.evidence.pqc-pr7-closure-workspace-gates-2026-08-24/1` + +Workspace build, full release suite, strict Clippy, formatting, and the quality-effort+perceptual feature-enabled CLI release tests all pass after the policy note and regression-test update. + +**Verifies** + +- `completed` `@jpegxl-rs.work.pqc-pr7-holdout-closure/1` — check `workspace-gates` + ### Development split, Balanced with the reducer on versus off: bytes geomean 0.9825 at target 70 (photos 0.9903, min 0.888) and 0.9912 at 85 (photos 0.9886, min 0.9835); zero floor violations in 30 cells (the reduced stream is kept only when its exact size is smaller and every accepted batch is re-scored canonically); at most 6 evaluations; wall geomean 1.53x / 1.63x. Bytes down at matched-or-better score with bounded work, but the wall exceeds Balanced's +25% budget, so the reducer ships default-off on Balanced (BALANCED_DEFAULT_REDUCER = None) and on for the feature-gated Quality effort; the locked-holdout gate is still to be measured. `verified` · `@jpegxl-rs.evidence.pqc-pr7-dev-split-2026-08-22/1` Development split, Balanced with the reducer on versus off: bytes geomean 0.9825 at target 70 (photos 0.9903, min 0.888) and 0.9912 at 85 (photos 0.9886, min 0.9835); zero floor violations in 30 cells (the reduced stream is kept only when its exact size is smaller and every accepted batch is re-scored canonically); at most 6 evaluations; wall geomean 1.53x / 1.63x. Bytes down at matched-or-better score with bounded work, but the wall exceeds Balanced's +25% budget, so the reducer ships default-off on Balanced (BALANCED_DEFAULT_REDUCER = None) and on for the feature-gated Quality effort; the locked-holdout gate is still to be measured. +### Full PR7 locked-holdout Quality/reducer sweep completes after low-memory changes + +`verified` · `@jpegxl-rs.evidence.pqc-pr7-holdout-complete-2026-08-24/1` + +Combined historical checkpoints with 53 resumed rows to close 91/91 Quality and 91/91 Balanced cells on all 13 locked holdout images. All 182 successful rows had zero canonical-floor and decoder failures; reducer work stayed at <=6 evaluations, saved 263,266 exact bytes total, and produced 0.991552 final/before geomean. The formerly blocked 50 MP Quality rows completed serially under a 12 GiB cap; worst measured peak was 11,163,372 KiB. Kept report: .agent/scratch/pr7-holdout-closure-20260824/report.json (SHA-256 fd78df93...f282e). + +**Verifies** + +- `completed` `@jpegxl-rs.work.pqc-pr7-holdout-closure/1` — check `full-holdout` + +### Production budgets keep rejected Quality-only work disabled + +`verified` · `@jpegxl-rs.evidence.pqc-pr7-policy-gating-test-2026-08-24/1` + +The new policy regression test passes and proves Fast/Balanced default budgets retain zero policy trials and no reducer while the feature-gated Quality budget retains its bank and terminal reducer. + +**Verifies** + +- `completed` `@jpegxl-rs.work.pqc-pr7-holdout-closure/1` — check `resulting-disposition` + +### Full Contract B result is dispositioned without weakening the public gate + +`verified` · `@jpegxl-rs.evidence.pqc-pr7-promotion-disposition-2026-08-24/1` + +The full 79-cell common-score report was evaluated against the standing Quality promotion conditions. Overall matched bytes pass at 0.958292, but Butteraugli mean 1.047881 and worst pnorm3 1.868818 fail their 1.02/1.05 limits, so the existing quality-effort feature gate remains in force. No search-policy change is justified; the source note now records the completed gate and a regression test pins Quality-only work out of Fast/Balanced. + +**Verifies** + +- `completed` `@jpegxl-rs.work.pqc-pr7-holdout-closure/1` — check `promotion-contract` + +### Full locked holdout rejects public Quality promotion under Contract B + +`verified` · `@jpegxl-rs.evidence.pqc-pr7-quality-promotion-full-2026-08-24/1` + +Across 79 common-score cells, Quality/Balanced bytes pass at 0.958292 geomean, but Butteraugli mean ratio is 1.047881 (>1.02) and worst pnorm3 ratio is 1.868818 (>1.05), so Contract B fails. Photos are nearly byte-neutral at 0.991367 with BA mean 1.000790 and worst pnorm3 1.042495; non-photo savings drive the aggregate while saturated, tiny, and text cells drive the guard failure. Quality remains feature-gated. + ### Partial locked holdout definitively rejects public Quality promotion `verified` · `@jpegxl-rs.evidence.pqc-pr7-quality-promotion-rejected-2026-08-22/1` @@ -5144,6 +5417,12 @@ Stopped by user after 64/91 Quality cells (10/13 images; 55 decoded-score-matche Across 64 completed Quality cells the reducer had zero canonical floor violations, at most 6 evaluations, saved 147845 exact bytes total, and produced a 0.98925 final/before geomean (0.98626 on 50 eligible cells). The reducer-off decoded guard baseline and 27 remaining Quality cells were not run; reducer-gate remains open and is deferred with no more Quality testing requested. +### Expanded-corpus qpv2 model: blind holdout hits production p90; transform features win + +`verified` · `@jpegxl-rs.evidence.pqc-qpv2-expanded-corpus-gate-2026-08-25/1` + +Corpus expanded with 211 independent painting families from the user's collection (104 calibration / 57 development / 50 never-tuned ext-holdout, deterministic split by family hash, originals untouched); 103 swept within the 2-hour budget for 994 total label rows (zero censored). Retrained pooled model, 10-fold family-grouped CV: source+transform wins (CV median 0.121 / p90 0.461 vs source-only 0.176/0.595); blind never-tuned ext-holdout: median 0.113, p90 0.258 (production p90 bound met), p99 0.478, first-plan 0.925, first-or-one-correction 1.00. Remaining CV tail is entirely the saturated (p90 3.58) and sky-noise gradient classes, still effectively class-held-out. Emitted runtime model qpv2-st-1 (19 features: 9 source + 10 DCT8-summary); the in-search transform summary measured -3.9% wall at 4.3 MP (shares the cover's cache; standalone 7.8% after parallelizing the fill from 21.9% serial). Artifacts in .agent/scratch/one-shot-qpv2-20260824/. + ### Corrected common-axis rate curve is exact on 10/13 holdout images; xlarge quality remains memory-blocked `verified` · `@jpegxl-rs.evidence.pqc-quality-rate-curve-exact-10of13-2026-08-23/1` @@ -5185,6 +5464,16 @@ All 20 codec_compare unit tests passed, covering matched-score interpolation, BD The complete chained workspace build, release-test, clippy-with-warnings-denied, and formatting gate exited successfully. +### Production identity holds after the band-parallel scatter: 52/52 locked cells identical across threads 1/4 and AVX2 on/off + +`verified` · `@jpegxl-rs.evidence.pqc-scatter-production-identity-2026-08-25/1` + +The full locked 13-image matrix (both production efforts, q70 and q85, serialized under a 9 GiB address-space cap) produced 52/52 byte-identical cells across threads 1, threads 4, and threads 4 with JPXL_DISABLE_AVX2=1 on the band-parallel-scatter binary; separately, six image/target cells (both anchors at q85/q70, Fast and Balanced, text-screenshot and gradient) hashed identically between this binary and a reference binary built from commit 1fc3ca7 in an isolated worktree, and within the new binary at threads 1/4/8 and AVX2 off. Rows and summaries are in the kept scratch entry pqc-scatter-20260825 (identity-scatter-summary.json, identity-ab.json). + +**Verifies** + +- `superseded` `@jpegxl-rs.work.pqc-usable-efforts-cost/7` — check `production-identity` + ### Release test suite green across the workspace (the one failure seen in the background run was the PR 1 placeholder CLI test, rewritten in the same tree), clippy clean under -D warnings with default and extended feature sets, fmt clean. Rate-mode production streams on mid.ppm (--bpp 1.0, 4 threads) hash 05bae79d4c96f77b2bb6bd3b1ad6a794331323d11e7c55903db4c4359798701b (balanced) and 07d71108de1fc69fd4fa5cb9e0e71917ec0887510bd21bc7db5eab4a0bf6bc6b (fast), identical to the pre-change binary. `verified` · `@jpegxl-rs.evidence.pqc-workspace-gates-2026-08-22/1` @@ -5653,6 +5942,16 @@ All required Rust workspace gates pass on native Windows. The conformance link m - `completed` `@jpegxl-rs.work.quality-q9-production-presets/1` — check `workspace-gates` +### README describes the current supported scope, exclusions, clean-room policy, build steps, and the recorded benchmark + +`verified` · `@jpegxl-rs.evidence.readme-current-state-2026-08-25/1` + +The README's claims were verified against JPXL/docs/CONFORMANCE.md and the actual CLI contracts (jpxl --help, encode, compare); the quality-search section now correctly describes the promoted one-shot seed (qpv2-st-1 pooled quantile regression behind the default one-shot-controller feature, with the calibrated-table fallback for out-of-envelope inputs) and the full-resolution canonical verification before every emission, replacing the stale claim that the calibrated table picks the first probe; the benchmark section documents the reproducible bench_vs_libjxl.sh entry point, and the license section states the no-copyleft position with a pointer to the audit. + +**Verifies** + +- `completed` `@jpegxl-rs.work.publish-readme-benchmark-mit/1` — check `readme-current-state` + ### Release-mode workspace tests pass `verified` · `@jpegxl-rs.evidence.release-workspace-tests-2026-08-21/1` @@ -5820,3 +6119,13 @@ jpegxl-rs.decision.ledger-conventions fixes namespace, scope convention, and the **Verifies** - `completed` `@jpegxl-rs.milestone.spine-bootstrap/1` — check `conventions` + +### Workspace release gates pass with jpxl-jpeg, the scatter change, and the publish surface in the tree + +`verified` · `@jpegxl-rs.evidence.workspace-gates-2026-08-25/1` + +Under pipefail in one chained run: cargo test --workspace --release reported 76 test-result groups, all ok with zero failures (including the new jpxl-jpeg suite); cargo clippy --workspace --all-targets -D warnings finished clean; cargo fmt --all --check passed. + +**Verifies** + +- `completed` `@jpegxl-rs.work.publish-readme-benchmark-mit/1` — check `workspace-gates` diff --git a/docs/generated/DECISION-HISTORY.md b/docs/generated/DECISION-HISTORY.md index 56070d25..d2244d76 100644 --- a/docs/generated/DECISION-HISTORY.md +++ b/docs/generated/DECISION-HISTORY.md @@ -1,5 +1,5 @@ <!-- GENERATED BY AKR — DO NOT EDIT - source-graph: sha256:29947c3fadfc107253854896393926aa0698b43a301121176a032e4f7a7260f4 + source-graph: sha256:2e53454b4abc5a2e7803995f3fb00745b536d02f33e33c9607223cdf0dd3935b tool: akr 0.3.3 --> @@ -242,6 +242,18 @@ Normal lossy encoding is specified by a minimum perceptual score, not a byte bud **supersedes** `@jpegxl-rs.decision.lossy-production-presets/1` +## jpegxl-rs.decision.quality-miss-fallback-semantics + +### Revision 1 — An under-target perceptual result is refused at the facade by default; lossless and best-effort are explicit opt-in fallbacks + +`proposed` · `@jpegxl-rs.decision.quality-miss-fallback-semantics/1` · scope `path "JPXL/crates/jpxl-cli/**"`, `path "JPXL/crates/jpxl-encode-policy/src/quality.rs"`, `path "JPXL/crates/jpxl/**"` + +An under-target perceptual result is never an ordinary success at the public facade. When the bounded controller stops at SaturatedTop or UnderTargetWorkCap, the default is to refuse: Encoder returns Error::TargetNotMet carrying a QualityMiss (kind LadderSaturated | WorkBudgetExhausted, requested and best verified scores, probe/price counts, metric version, quality trace), and the CLI exits 1 without creating the output file. Two explicit fallbacks exist via Encoder::with_quality_fallback / --quality-fallback: Lossless emits a mathematically lossless stream reported as PerceptualStatus::FallbackLossless with achieved score 100, and BestEffort emits the finest canonically verified under-target stream under its true status and score. SaturatedFloor remains a success. The policy layer (search_frame_perceptual) still returns every terminal status with bytes; the facade owns the refusal. The rescue probe stays one probe beyond the navigation cap and is documented as an observable per-solve maximum of pixel_probes + 1; folding it inside the cap is a deliberate search-policy change that would require the standing Contract B screen. + +**Context.** The 2026-08-24 one-shot advisor memo's first code finding, verified against quality.rs, jpxl/src/lib.rs and jpxl-cli/src/main.rs: the facade wrapped SaturatedTop and UnderTargetWorkCap outcomes as Ok and the CLI wrote the output file and exited 0, so a scripted caller checking only the exit code received a silently under-target stream — incompatible with the hard-floor reading of the perceptual-quality contract. The completed 91-cell locked holdout contains no under-target cells, so refusing changes no canonical-sweep behavior. + +**Consequences.** Met, MetAdjacentRungs, MetWorkCap, SaturatedFloor and RescuedFreshStructure outputs stay byte-identical. The public API adds Error::TargetNotMet, QualityMiss, QualityMissKind, QualityFallback, PerceptualStatus::FallbackLossless and Encoder::with_quality_fallback; the CLI adds --quality-fallback lossless|best-effort. A refused encode still appends its jpxl.quality-trace/1 record to JPXL_QUALITY_TRACE so failed searches remain calibration input. Facade, policy-layer and CLI tests now pin the miss path, which previously had no coverage anywhere. + ## jpegxl-rs.decision.quant-bias-defaults ### Revision 1 — quant_bias defaults are 1 minus x diff --git a/docs/generated/OPEN-QUESTIONS.md b/docs/generated/OPEN-QUESTIONS.md index c1f038b8..6a00ff60 100644 --- a/docs/generated/OPEN-QUESTIONS.md +++ b/docs/generated/OPEN-QUESTIONS.md @@ -1,5 +1,5 @@ <!-- GENERATED BY AKR — DO NOT EDIT - source-graph: sha256:29947c3fadfc107253854896393926aa0698b43a301121176a032e4f7a7260f4 + source-graph: sha256:2e53454b4abc5a2e7803995f3fb00745b536d02f33e33c9607223cdf0dd3935b tool: akr 0.3.3 --> diff --git a/docs/generated/PAPERCUTS.md b/docs/generated/PAPERCUTS.md index 05265cb9..7fcc8c7f 100644 --- a/docs/generated/PAPERCUTS.md +++ b/docs/generated/PAPERCUTS.md @@ -1,5 +1,5 @@ <!-- GENERATED BY AKR — DO NOT EDIT - source-graph: sha256:29947c3fadfc107253854896393926aa0698b43a301121176a032e4f7a7260f4 + source-graph: sha256:2e53454b4abc5a2e7803995f3fb00745b536d02f33e33c9607223cdf0dd3935b tool: akr 0.3.3 --> @@ -7,6 +7,7 @@ Small frictions hit while working, logged in the moment (D-027). None of these blocked anything; together they show where the project needs sanding down. Newest first. +- 2026-08-24 [codex] The frozen PR7 holdout parser assumed `ssimulacra2` immediately followed `psnr_db`; the current compare line inserts `ssimulacra2_jpxl` and its version first, so an otherwise successful 12 MP resumed cell was needlessly rerun. Parse named fields independently or share the checked-in compare parser. `@jpegxl-rs.papercut.the-frozen-pr7-holdout-parser-assumed/1` - 2026-08-21 [Codex] The Phase-32 corpus harness resolves JPXL_PHASE32_PPM paths from the package test working directory, so repo-relative paths that exist from the cargo invocation directory fail with an unlabelled Os NotFound. The harness should report the failing path or document that paths must be absolute. `@jpegxl-rs.papercut.the-phase-32-corpus-harness-resolves-jpxl/1` - 2026-08-19 [fugu-ultra] The native Windows workspace test gate treated the conformance corpus's 166-byte `IntxLNK` symlink placeholder as a downloaded `reference_image.npy` and failed with `BadMagic`; the test's documented missing-reference skip handled absent/empty files but not Windows symlink placeholders. `@jpegxl-rs.papercut.the-native-windows-workspace-test-gate-treated/1` - 2026-08-19 [fugu-ultra] The checked-in quality-track README points ladder.ps1 at `.agent/scratch/quality-track/inputs`, but the current `mid-photo.ppm` there is not a valid P5/P6 image and the run fails immediately; prior handoff outputs exist, so the fixture path or scratch retention is stale and needs a regeneration note/check. `@jpegxl-rs.papercut.the-checked-in-quality-track-readme-points/1` @@ -27,6 +28,10 @@ Small frictions hit while working, logged in the moment (D-027). None of these b Frictions with something else — a tool, a harness — hit while working here. They are logged where they were hit; `akr papercut collate --about <subject>` is how the project that owns the subject gathers them. +- 2026-08-25 [claude] (akr) knowledge.get with detail canonical on jpegxl-rs.work.pqc-usable-efforts-cost/7 truncated at the tool's token cap and the suggested continuation (detail summary) cannot return the acceptance block's source text either, so reproducing checks verbatim for a revision required reading .akr/records/jpegxl-rs/work.akr by hand. A continuation or an acceptance-only detail level would remove the need to touch the records directory. `@jpegxl-rs.papercut.knowledge-get-with-detail-canonical-on-jpegxl/1` +- 2026-08-25 [claude] (akr) Amending an akr-generated commit message (to replace the generic "chore:" subject with a descriptive one while keeping the AKR trailers byte-identical) is rejected by the commit-msg hook with AKR-C031 because the change transaction closes at akr git commit; the only way through is git commit --amend --no-verify. Either akr git commit could accept a subject/body override, or the hook could allow amends whose trailers match the last commit. `@jpegxl-rs.papercut.amending-an-akr-generated-commit-message-to/1` +- 2026-08-25 [claude] (akr) A work revision that cites already-committed evidence in verified_by lands "not satisfied - predates the last change" unless the evidence rows are committed in the very same commit as the revision (the current-together grace); revising a record in a later .akr-only commit therefore un-satisfies checks whose measurements are still perfectly valid, and there is no way to re-land unchanged evidence. Retargeting a check citation onto fresh evidence needs either an observed_at at-or-after the revision commit (chicken-and-egg) or duplicated evidence records. Hit while pointing pqc-usable-efforts-cost's workspace-gates check at the 2026-08-25 gates evidence; repaired by dropping the revision commit. `@jpegxl-rs.papercut.a-work-revision-that-cites-already-committed/1` +- 2026-08-24 [codex] (akr) knowledge.propose's generic `slots` schema does not expose that an observation's `method` must be one of manual/command/instrumented/observation. V-022 said to add `method`; supplying explanatory prose then produced dozens of parser diagnostics before the enum expectation appeared. `@jpegxl-rs.papercut.knowledge-propose-s-generic-slots-schema-does/1` - 2026-08-23 [codex-gpt-5] (akr) knowledge.propose exposes `topic` for every record kind, but a work proposal containing it fails only after full validation with AKR-T034 because topic is normative-only. The tool schema or preflight should state/reject this earlier. `@jpegxl-rs.papercut.knowledge-propose-exposes-topic-for-every/1` - 2026-08-23 [codex] (akr) knowledge.propose exposes scope as Array<unknown>; passing intuitive {path: ...} objects fails only with 'unknown scope form'. The accepted MCP shape is {form: 'path', glob: ...}, which should be expressed in the tool schema or error. `@jpegxl-rs.papercut.knowledge-propose-exposes-scope-as-array/1` - 2026-08-23 [codex] (akr) After `akr scratch keep` visibly added a new entry, knowledge.evidence_add still rejected that kept artifact from its cached workspace. The CLI fallback revalidated ~230 unrelated historical scratch citations lacking current keep markers and refused an otherwise valid atomic write; omitting the typed artifact was the only scoped path forward. `@jpegxl-rs.papercut.after-akr-scratch-keep-visibly-added-a-new/1` diff --git a/docs/generated/REVIEW-REQUIRED.md b/docs/generated/REVIEW-REQUIRED.md index 704d5ff6..73ce5e72 100644 --- a/docs/generated/REVIEW-REQUIRED.md +++ b/docs/generated/REVIEW-REQUIRED.md @@ -1,5 +1,5 @@ <!-- GENERATED BY AKR — DO NOT EDIT - source-graph: sha256:29947c3fadfc107253854896393926aa0698b43a301121176a032e4f7a7260f4 + source-graph: sha256:2e53454b4abc5a2e7803995f3fb00745b536d02f33e33c9607223cdf0dd3935b tool: akr 0.3.3 --> @@ -7,7 +7,7 @@ What should not be trusted without re-checking: records the build flagged `stale` or `at_risk`. Neither flag means a record is wrong (D-003); both mean look at it. This view is generated on every successful build, including one that exits 0 with a long queue (D-024). An empty file on an active project is more often a sign the `watches` globs are wrong than a sign the knowledge is perfect. -## Stale (46) +## Stale (50) ### The cover/CfL objective misprices Y-channel error by 2.65x across DCT8x8 frequency; the mispricing is in the ruler, not the lever @@ -93,6 +93,18 @@ What should not be trusted without re-checking: records the build flagged `stale **Cause** — `watches "JPXL/crates/jpxl-encode-policy/src/rate.rs"` was matched by `08f6c4a0`, which touched `JPXL/crates/jpxl-encode-policy/src/rate.rs`. +### Phase 4M trace localises high-rate undershoot to LF-fill ordering + +`verified` · `@jpegxl-rs.observation.phase4m-lf-fill-direction-2026-08-11/1` · observation · **stale** · [Phase 4M trace localises high-rate undershoot to LF-fill ordering](CURRENT-STATE.md#phase-4m-trace-localises-high-rate-undershoot-to-lf-fill-ordering) + +**Cause** — `watches "JPXL/crates/jpxl-cli/src/main.rs"` was matched by `0fd5b3b0`, which touched `JPXL/crates/jpxl-cli/src/main.rs`. + +### PR4's byte-neutral rate-curve conclusion mixed SSIMULACRA2 implementations and is not a like-for-like baseline + +`verified` · `@jpegxl-rs.observation.pqc-pr4-rate-curve-axis-audit-2026-08-23/1` · observation · **stale** · [PR4's byte-neutral rate-curve conclusion mixed SSIMULACRA2 implementations and is not a like-for-like baseline](CURRENT-STATE.md#pr4s-byte-neutral-rate-curve-conclusion-mixed-ssimulacra2-implementations-and-is-not-a-like-for-like-baseline) + +**Cause** — `watches "JPXL/crates/jpxl-cli/src/main.rs"` was matched by `0fd5b3b0`, which touched `JPXL/crates/jpxl-cli/src/main.rs`. + ### I.4 context model proved by ANS exhaustion `verified` · `@jpegxl-rs.observation.i4-context-model/1` · observation · **stale** · [I.4 context model proved by ANS exhaustion](CURRENT-STATE.md#i4-context-model-proved-by-ans-exhaustion) @@ -105,12 +117,6 @@ What should not be trusted without re-checking: records the build flagged `stale **Cause** — `watches "JPXL/crates/jpxl-encode-policy/src/rate.rs"` was matched by `23635f68`, which touched `JPXL/crates/jpxl-encode-policy/src/rate.rs`. -### Phase 4M trace localises high-rate undershoot to LF-fill ordering - -`verified` · `@jpegxl-rs.observation.phase4m-lf-fill-direction-2026-08-11/1` · observation · **stale** · [Phase 4M trace localises high-rate undershoot to LF-fill ordering](CURRENT-STATE.md#phase-4m-trace-localises-high-rate-undershoot-to-lf-fill-ordering) - -**Cause** — `watches "JPXL/crates/jpxl-cli/src/main.rs"` was matched by `23635f68`, which touched `JPXL/crates/jpxl-cli/src/main.rs`. - ### Q3: at matched bytes JPXL leads cjxl -e7 on SSIMULACRA2 in every cell and trails on Butteraugli and PSNR; the deficit sits in low-to-mid activity blocks, worst where an edge meets flat content `verified` · `@jpegxl-rs.observation.q3-butteraugli-deficit-localisation-2026-08-18/3` · observation · **stale** · [Q3: at matched bytes JPXL leads cjxl -e7 on SSIMULACRA2 in every cell and trails on Butteraugli and PSNR; the deficit sits in low-to-mid activity blocks, worst where an edge meets flat content](CURRENT-STATE.md#q3-at-matched-bytes-jpxl-leads-cjxl--e7-on-ssimulacra2-in-every-cell-and-trails-on-butteraugli-and-psnr-the-deficit-sits-in-low-to-mid-activity-blocks-worst-where-an-edge-meets-flat-content) @@ -141,12 +147,6 @@ What should not be trusted without re-checking: records the build flagged `stale **Cause** — `watches "JPXL/crates/jpxl-encode-policy/src/quality.rs"` was matched by `2425c0b2`, which touched `JPXL/crates/jpxl-encode-policy/src/quality.rs`. -### PR4's byte-neutral rate-curve conclusion mixed SSIMULACRA2 implementations and is not a like-for-like baseline - -`verified` · `@jpegxl-rs.observation.pqc-pr4-rate-curve-axis-audit-2026-08-23/1` · observation · **stale** · [PR4's byte-neutral rate-curve conclusion mixed SSIMULACRA2 implementations and is not a like-for-like baseline](CURRENT-STATE.md#pr4s-byte-neutral-rate-curve-conclusion-mixed-ssimulacra2-implementations-and-is-not-a-like-for-like-baseline) - -**Cause** — `watches "JPXL/crates/jpxl-cli/src/main.rs"` was matched by `2425c0b2`, which touched `JPXL/crates/jpxl-cli/src/main.rs`. - ### The reference SSIMULACRA2 implementations' f32 recursive Gaussian leaves a ripple that inflates near-lossless scores' error on flat content, growing with image size; the in-tree metric runs the recursion in f64 `verified` · `@jpegxl-rs.observation.ssimulacra2-f32-recursion-ripple-2026-08-22/1` · observation · **stale** · [The reference SSIMULACRA2 implementations' f32 recursive Gaussian leaves a ripple that inflates near-lossless scores' error on flat content, growing with image size; the in-tree metric runs the recursion in f64](CURRENT-STATE.md#the-reference-ssimulacra2-implementations-f32-recursive-gaussian-leaves-a-ripple-that-inflates-near-lossless-scores-error-on-flat-content-growing-with-image-size-the-in-tree-metric-runs-the-recursion-in-f64) @@ -171,6 +171,12 @@ What should not be trusted without re-checking: records the build flagged `stale **Cause** — `watches "JPXL/crates/jpxl-encode-policy/tests/rate_proxy_audit.rs"` was matched by `5fd8e357`, which touched `JPXL/crates/jpxl-encode-policy/tests/rate_proxy_audit.rs`. +### Parallelizing the serial varblock render and linearization cut 12 MP quality wall 21%; ratio 3.11x -> 2.46x on the measuring host + +`verified` · `@jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1` · observation · **stale** · [Parallelizing the serial varblock render and linearization cut 12 MP quality wall 21%; ratio 3.11x -> 2.46x on the measuring host](CURRENT-STATE.md#parallelizing-the-serial-varblock-render-and-linearization-cut-12-mp-quality-wall-21-ratio-311x---246x-on-the-measuring-host) + +**Cause** — `watches "JPXL/crates/jpxl-plan-render/src/lib.rs"` was matched by `b0c128c7`, which touched `JPXL/crates/jpxl-plan-render/src/lib.rs`. + ### Cropped frames, orientation, kBlack channels `verified` · `@jpegxl-rs.observation.cropped-frames-orientation-kblack/1` · observation · **stale** · [Cropped frames, orientation, kBlack channels](CURRENT-STATE.md#cropped-frames-orientation-kblack-channels) @@ -279,13 +285,31 @@ What should not be trusted without re-checking: records the build flagged `stale **Cause** — `watches "JPXL/crates/jpxl-encode-policy/src/field.rs"` was matched by `b25beda2`, which touched `JPXL/crates/jpxl-encode-policy/src/field.rs`. +### The one-shot falsification screen fails on corpus coverage, not on the design: transform features halve the error, the tail is two under-represented classes + +`verified` · `@jpegxl-rs.observation.pqc-one-shot-gate-fails-on-corpus-coverage-2026-08-24/2` · observation · **stale** · [The one-shot falsification screen fails on corpus coverage, not on the design: transform features halve the error, the tail is two under-represented classes](CURRENT-STATE.md#the-one-shot-falsification-screen-fails-on-corpus-coverage-not-on-the-design-transform-features-halve-the-error-the-tail-is-two-under-represented-classes) + +**Cause** — `watches "JPXL/crates/jpxl-encode-policy/src/quality_predictor*.rs"` was matched by `b57c9d40`, which touched `JPXL/crates/jpxl-encode-policy/src/quality_predictor_v2.rs`. + +### Full PR7 holdout closes the OOM tail and rejects Quality promotion + +`verified` · `@jpegxl-rs.observation.pqc-pr7-full-holdout-quality-rejected-2026-08-24/1` · observation · **stale** · [Full PR7 holdout closes the OOM tail and rejects Quality promotion](CURRENT-STATE.md#full-pr7-holdout-closes-the-oom-tail-and-rejects-quality-promotion) + +**Cause** — `watches "JPXL/crates/jpxl-encode-policy/src/quality.rs"` was matched by `b57c9d40`, which touched `JPXL/crates/jpxl-encode-policy/src/quality.rs`. + +### Quiet-host wall anchors: 12 MP 5.42x (4t) / 4.25x (8t); the quality path's fixed costs alone sit near 2x, so the 2.0x target is unreachable by search improvements + +`verified` · `@jpegxl-rs.observation.pqc-wall-quiet-host-attribution-2026-08-25/1` · observation · **stale** · [Quiet-host wall anchors: 12 MP 5.42x (4t) / 4.25x (8t); the quality path's fixed costs alone sit near 2x, so the 2.0x target is unreachable by search improvements](CURRENT-STATE.md#quiet-host-wall-anchors-12-mp-542x-4t--425x-8t-the-quality-paths-fixed-costs-alone-sit-near-2x-so-the-20x-target-is-unreachable-by-search-improvements) + +**Cause** — `watches "JPXL/crates/jpxl-plan-render/src/lib.rs"` was matched by `c23bc009`, which touched `JPXL/crates/jpxl-plan-render/src/lib.rs`. + ### Phase Q9 fixes Windows AVX2/fallback cube-root determinism without moving AVX2 production hashes `verified` · `@jpegxl-rs.observation.windows-msvc-avx2-fallback-not-bit-identical-2026-08-18/3` · observation · **stale** · [Phase Q9 fixes Windows AVX2/fallback cube-root determinism without moving AVX2 production hashes](CURRENT-STATE.md#phase-q9-fixes-windows-avx2fallback-cube-root-determinism-without-moving-avx2-production-hashes) **Cause** — `watches "JPXL/crates/jpxl-core/src/color.rs"` was matched by `deed1f65`, which touched `JPXL/crates/jpxl-core/src/color.rs`. -## At risk (7) +## At risk (8) ### Assess the 2026-08-21 libjxl-gap bridge against current JPXL @@ -323,6 +347,12 @@ What should not be trusted without re-checking: records the build flagged `stale **Via** `supported_by` → `@jpegxl-rs.observation.libjxl-comparison-2026-08-18/2` (stale: `watches "JPXL/tools/compare-libjxl.ps1"` was matched by `4f528696`, which touched `JPXL/tools/compare-libjxl.ps1`.) +### PQC usable efforts: reduce Fast/Balanced wall time and peak memory + +`active` · `@jpegxl-rs.work.pqc-usable-efforts-cost/9` · work · **depth 1** · [PQC usable efforts: reduce Fast/Balanced wall time and peak memory](ACTIVE-WORK.md#pqc-usable-efforts-reduce-fastbalanced-wall-time-and-peak-memory) + +**Via** `supported_by` → `@jpegxl-rs.observation.pqc-large-frame-render-parallel-2026-08-25/1` (stale: `watches "JPXL/crates/jpxl-plan-render/src/lib.rs"` was matched by `b0c128c7`, which touched `JPXL/crates/jpxl-plan-render/src/lib.rs`.) + ### Execute the evidence-gated G0-G6 libjxl-gap bridge `active` · `@jpegxl-rs.decision.encoder-architecture-phases/2` · decision · **depth 2** · [Execute the evidence-gated G0-G6 libjxl-gap bridge](DECISION-HISTORY.md#revision-2--execute-the-evidence-gated-g0-g6-libjxl-gap-bridge) diff --git a/docs/generated/ROADMAP.md b/docs/generated/ROADMAP.md index ff6fc9b0..68852d41 100644 --- a/docs/generated/ROADMAP.md +++ b/docs/generated/ROADMAP.md @@ -1,5 +1,5 @@ <!-- GENERATED BY AKR — DO NOT EDIT - source-graph: sha256:29947c3fadfc107253854896393926aa0698b43a301121176a032e4f7a7260f4 + source-graph: sha256:2e53454b4abc5a2e7803995f3fb00745b536d02f33e33c9607223cdf0dd3935b tool: akr 0.3.3 --> @@ -359,7 +359,8 @@ Deliver the perceptual quality contract of jpegxl-rs.decision.perceptual-quality **Work items** -- `active` [PQC usable efforts: reduce Fast/Balanced wall time and peak memory](ACTIVE-WORK.md#pqc-usable-efforts-reduce-fastbalanced-wall-time-and-peak-memory) `@jpegxl-rs.work.pqc-usable-efforts-cost/5` +- `active` [One-shot program: shadow-first common-case one-shot quality controller](ACTIVE-WORK.md#one-shot-program-shadow-first-common-case-one-shot-quality-controller) `@jpegxl-rs.work.pqc-one-shot-controller/3` +- `active` [PQC usable efforts: reduce Fast/Balanced wall time and peak memory](ACTIVE-WORK.md#pqc-usable-efforts-reduce-fastbalanced-wall-time-and-peak-memory) `@jpegxl-rs.work.pqc-usable-efforts-cost/9` — **at risk** ### VarDCT encoder (M1-M8) @@ -368,7 +369,9 @@ Deliver the perceptual quality contract of jpegxl-rs.decision.perceptual-quality The VarDCT encoder, built as milestones M1-M8 (PLAN.md slices 11-18) and backfilled from git history as legacy-sourced completed records (D-028). -**Work items** — _(none)_ +**Work items** + +- `active` [Lossless JPEG bitstream recompression: carry JPEG1 coefficients into JPEG XL directly instead of decode-and-reencode](ACTIVE-WORK.md#lossless-jpeg-bitstream-recompression-carry-jpeg1-coefficients-into-jpeg-xl-directly-instead-of-decode-and-reencode) `@jpegxl-rs.work.jpeg-bitstream-recompression/1` ## Summary diff --git a/sources/catalog.json b/sources/catalog.json index 35d05563..e93798c7 100644 --- a/sources/catalog.json +++ b/sources/catalog.json @@ -23,6 +23,17 @@ "scope": "JPXL/**", "title": "Outside advice: a codec-optimization perceptual metric for JPXL (JPXL-PQ / JPXL-PCost)" }, + { + "added_at": "2026-08-24", + "availability": "full", + "byte_len": 53129, + "content_hash": "sha256:ab931f55436c9b169b1369718c4eaa1187e8d007ff7ed9d430f280d01a8bf991", + "id": "jpxl-one-shot-quality-controller-memo-2026-08-24", + "media_type": "text/markdown", + "origin": "external", + "path": "sources/external/jpxl-one-shot-quality-controller-memo-2026-08-24--ab931f55.md", + "title": "One-shot quality-controller design memo (advisor response, 2026-08-24)" + }, { "added_at": "2026-08-22", "availability": "full", diff --git a/sources/external/jpxl-one-shot-quality-controller-memo-2026-08-24--ab931f55.md b/sources/external/jpxl-one-shot-quality-controller-memo-2026-08-24--ab931f55.md new file mode 100644 index 00000000..9b39322e --- /dev/null +++ b/sources/external/jpxl-one-shot-quality-controller-memo-2026-08-24--ab931f55.md @@ -0,0 +1,1185 @@ +# Design memo: JPXL score-targeted encoding + +## Executive verdict + +JPXL should implement **common-case one shot**, not claim true one shot. + +A source- or transform-derived statistical predictor cannot guarantee the canonical decoded SSIMULACRA2 score of an arbitrary image. A conservative reserve can reduce the miss rate, but it cannot turn an empirical predictor into a proof. The only generally defensible true-one-shot guarantees would be: + +1. encode losslessly, or +2. derive a proven worst-case relationship between quantization and SSIMULACRA2. + +Neither is presently available as a byte-competitive lossy controller. + +The existing hard-floor contract can nevertheless remain intact: + +> A successful result is emitted only after its actual reconstructed pixels have been canonically scored at or above the requested target. + +The normal path should perform one predicted pixel plan, one reconstruction/metric evaluation, and one entropy attachment/emission. A miss should receive one slope-based corrective plan. Broad uncertainty, out-of-distribution content, likely saturation, or a second miss should route into the existing bounded controller while reusing all work already performed. + +This directly preserves the handoff’s requirement that successful output not silently fall below the target, while keeping saturation and work exhaustion explicit. + +--- + +## Review scope and limitations + +I reviewed: + +* the attached advisor handoff; +* `quality.rs`, `quality_features.rs`, `quality_predictor.rs`, `rate.rs`; +* the public `jpxl` facade and CLI quality path; +* `CandidateSearchContext`, `CandidateForwardCache`, the cover/CfL planning order, and entropy attachment; +* `calibrate_initial_rung.py`; +* `quality-corpus.json`; +* the included AKR decisions and evidence summaries. + +I did not inspect or derive anything from the included libjxl source. That remains consistent with the clean-room constraint. + +The raw `.agent/scratch` evidence named in the handoff was excluded from the agent pack, so I could not independently recalculate those rows from their JSONL files. I treated the handoff and AKR records as authoritative for those measurements. The environment also lacks the Rust toolchain, so this is a static code and evidence review rather than a fresh test run. + +--- + +# 1. Immediate code findings that should be fixed first + +## 1.1 The public facade currently violates the hard-floor contract + +This is the most important finding. + +In `crates/jpxl-encode-policy/src/quality.rs:1109-1128`, when no probe meets the requested score, the controller deliberately selects the finest verified under-target probe and exact-prices it: + +```rust +// Nothing met the target: emit the finest verified probe and say so +under_target = true; +... +ordered.push(index); +``` + +At `quality.rs:1684-1716`, both `SaturatedTop` and `UnderTargetWorkCap` still produce a normal `QualityOutcome` containing `codestream: winner.bytes`. + +The public facade then wraps those bytes and returns `Ok` at: + +* `crates/jpxl/src/lib.rs:692-736` +* especially `lib.rs:708` and `lib.rs:723-736`. + +The public status documentation is itself explicit: + +```rust +/// The bounded controller ran out of probes before any candidate met the +/// score; the emitted stream's `achieved_score` is below the request. +UnderTargetWorkCap, +``` + +Finally, `crates/jpxl-cli/src/main.rs:2455-2495` treats every `Ok(Perceptual(...))` as a successful encode and returns the bytes for writing. + +That is incompatible with the handoff’s stated minimum-score semantics. + +### Required correction + +The default hard-floor API must not return an under-target codestream as a successful result. + +A minimally disruptive API would be: + +```rust +pub enum PerceptualFailureKind { + LossyLadderSaturated, + WorkBudgetExhausted, +} + +pub struct PerceptualFailure { + pub kind: PerceptualFailureKind, + pub requested_score: f64, + pub best_verified_score: f64, + pub best_rung: u32, + pub probes: u32, + pub structural_builds: u32, + pub metric_version: MetricVersion, + pub trace_json: Option<String>, +} +``` + +Then either: + +```rust +Error::PerceptualTargetNotMet(PerceptualFailure) +``` + +or a dedicated result enum should be returned. The CLI should exit nonzero and should not create or replace the requested output file. + +Two explicitly different fallback modes may be offered: + +* `--quality-fallback=lossless`: emit a lossless stream and report that the lossy ladder saturated. +* `--quality-fallback=best-effort`: knowingly emit the best verified under-target stream, with a distinct contract and status. + +Neither should happen silently. + +`SaturatedFloor` remains a successful result. `SaturatedTop` and `UnderTargetWorkCap` do not. + +--- + +## 1.2 The nominal probe budget is not actually hard + +`quality.rs:907-921` defines: + +```rust +/// One probe beyond the budget +fn rescue_probe(...) +``` + +The status documentation at `quality.rs:265` also refers to “the probe budget (plus its one rescue probe).” + +This conflicts with comments and tests that describe `pixel_probes` as a hard cap. It also makes the requested work-count table misleading. + +The correction should be one of: + +* reserve the final slot for rescue inside `pixel_probes`, or +* expose separate counters and limits such as `navigation_probes` and `rescue_probes`. + +I recommend the first. Balanced should have a **total maximum of five canonical pixel probes**, not five plus a conditionally hidden sixth. The new predictive path should continue the same total budget rather than receiving one or two prediction attempts and then restarting a five-probe controller. + +--- + +## 1.3 The current predictor is trained on the wrong endpoint + +The current generated predictor is not merely too small. Its labels do not correspond to the production decision that the proposed model must make. + +`quality_predictor.rs:8-11` states that it predicts fixed-quantizer `global_scale` with `HfMul = 1`. + +`calibrate_initial_rung.py:34-46` confirms that the calibration ladder ends at `global_scale = 73728`, while the production effective-scale ladder continues far beyond that by increasing `HfMul`. + +The production decision also includes quantizer-dependent structure: + +```text +quantizer + -> AQ setup + -> HF quantizers + -> cover selection + -> selected transforms + -> CfL + -> quantization + -> reconstruction +``` + +A fixed-global-scale calibration point is therefore not necessarily the same operating point as the final production Balanced stream, especially when a fresh-structure rescue changes cover or CfL. + +The replacement labels should be: + +> The coarsest effective rung whose **fresh production Balanced pixel plan** meets the target, over the complete effective-scale ladder, with saturation represented as censoring rather than as a clamped crossing. + +The current final corrected result can be recorded as a secondary label, but the oracle training label should not inherit the bounded navigator’s own approximation errors. + +--- + +## 1.4 The runtime predictor ignores most of its available features + +Although `SourceFeatures` contains: + +* dimensions; +* grayscale; +* luma q10, q50, and q90; +* chroma q50; +* flat fraction; +* edge proxy; + +`predicted_effective_scale()` in `quality.rs:613-663` uses only: + +* `luma_variance_q50`; +* `flat_fraction`; +* target score. + +The generated table’s `support` field is also unused. A cell supported by one image receives the same runtime trust as a well-supported cell. + +This explains why the current predictor is useful as a navigation seed but not as a final controller. Its recorded median absolute log-scale error is 0.30, while p90 is 2.68—a roughly 14.6× scale-ratio error at the tail. High targets are also heavily censored by the fixed-scale ceiling. + +The replacement should not be an expanded bucket table. It needs a calibrated curve prediction, uncertainty, and explicit saturation/OOD handling. + +--- + +## 1.5 Current “cheap” source features are not entirely free + +`quality_features.rs:91-139` creates two full vectors of per-atom values and sorts both to obtain quantiles. + +For a 50 MP image there are approximately 781,250 8×8 atoms. Two full `f32` arrays are not a major part of the encoder’s total memory, but two `O(n log n)` sorts are unnecessary for a supposedly cheap controller feature stage. + +Use deterministic fixed-bin log histograms or a deterministic selection algorithm. Histograms are preferable because they also provide useful tail and distribution features without additional storage. + +--- + +## 1.6 Existing controller tests are too idealized for this redesign + +Most `quality.rs` controller tests use `CurveEvaluator`, where score is a monotone quantizer-only power law. That validates navigation mechanics but cannot expose: + +* cover/CfL discontinuities; +* fresh-versus-reused structure changes; +* local score nonmonotonicity; +* saturation; +* content-class OOD; +* entropy-size inversions; +* public under-target success behavior. + +The public quality test uses one synthetic image at targets 50, 70, and 85. It does not test the public API’s terminal miss semantics. + +Add real-trace replay tests and explicit public tests asserting: + +```rust +assert!(matches!( + result, + Err(Error::PerceptualTargetNotMet(...)) +)); +``` + +for both work exhaustion and lossy-ladder saturation. + +--- + +## 1.7 The literal byte-minimization wording overstates the current controller + +The current controller exact-prices at most the two coarsest feasible finalists: + +```rust +let max_finalists = budget.exact_prices.clamp(1, 2); +``` + +It selects the smallest exact stream among those retained candidates, not the globally smallest JPXL codestream satisfying the score. + +This is reasonable for a bounded-effort encoder, but the contract should say so: + +> Produce a canonically verified stream meeting the requested score. Within the selected effort’s declared policy family and bounded evaluated candidate set, select the smallest exact-priced feasible finalist. + +If “minimize exact codestream bytes” is interpreted literally over all possible quantizers, structures, restoration settings, and entropy outcomes, neither the current controller nor a one-shot predictor satisfies it. + +This matters because exact entropy size can itself be locally nonmonotone. A one-shot controller can aim for the coarsest feasible rung, but it cannot prove that an unpriced neighboring rung would not entropy-code smaller. + +--- + +# 2. Contract comparison + +| Approach | Existing hard floor preserved? | Main limitation | Recommendation | +| ------------------------------------------------------------- | ----------------------------------------------------------------- | --------------------------------------------------------------------------------------- | ----------------------------------------------- | +| Conservative source-only prediction plus reserve | **No**, not by itself | Reserve gives empirical coverage, not a per-image guarantee; large reserve wastes bytes | Do not use as the correctness mechanism | +| One predicted encode, verify once, explicit failure on miss | **Yes for successful outputs** | No repair; potentially poor completion rate | Useful diagnostic mode, not primary public mode | +| Common-case one shot with one corrective re-plan | **Yes** | Some inputs still require two plans or exact fallback | **Recommended** | +| Reduced-resolution or cheap pilot, then one final full encode | Only if the final result is also verified and failure is explicit | Pilot/full-resolution relationship is content-dependent; still adds work | Test only if transform-derived prediction fails | +| Existing bounded exact controller for uncertainty/OOD | **Yes after the facade fix** | Expensive but bounded | Recommended fallback | +| Percentile or expected-score promise without verification | **No** | Changes `--quality` semantics | Only as a separately named API contract | +| Always route uncertain cases to lossless | Yes | Potentially catastrophic byte expansion | Explicit opt-in fallback only | + +A “score reserve” is still useful for choosing the first candidate or correction aim. It simply must not be described as the reason the floor is guaranteed. The canonical verification gate is the guarantee. + +--- + +# 3. Proposed runtime architecture + +## 3.1 High-level flow + +```text +source pixels + -> PreparedFrame + production AnalysisAtlas + -> deterministic source summary + -> request-scoped CandidateSearchContext + -> fill/reuse quantizer-independent transform candidate cache + -> deterministic transform summary + -> QualityPredictionV2(target, effort, source, transform) + outputs: + median crossing + risk-adjusted candidate crossing + calibrated interval + local loss slope + saturation risk + OOD flags + structural-instability risk + -> route: + uncertain / OOD / likely saturated + -> existing bounded exact navigator, seeded by prediction + otherwise + -> one fresh predicted pixel plan + -> full reconstruction + canonical SSIMULACRA2 + meets tightly + -> one entropy attachment + -> one emission + misses + -> one slope-based corrective plan on warm transform cache + -> reconstruct + score + -> emit if feasible + -> otherwise continue existing navigator with existing observations + meets but substantially overshoots + -> optional one coarsening attempt when predicted byte saving is material + -> retain first feasible plan as fallback + -> if terminal result is below target: + explicit failure, lossless fallback, or explicit best-effort mode +``` + +The current architecture already provides most of the needed mechanical separation: + +* `CandidateSearchContext::pixel_plan_for()` builds scored pixels without entropy. +* `attach_entropy_for()` trains entropy without changing pixels. +* `CandidateForwardCache` is request-scoped and quantizer-independent. +* cover/CfL rebuilds can reuse cached DCT coefficients. + +That means a failed prediction does not need entropy training and does not need to recompute already cached transforms. + +--- + +## 3.2 Suggested pseudocode + +```rust +fn encode_score_targeted( + ctx: &mut CandidateSearchContext<'_>, + source: SourceFeatureSummary, + target: f64, + budget: QualityBudget, +) -> Result<MetStream, PerceptualFailure> { + // Uses the same cache that final cover/CfL and quantization will consume. + let transform = ctx.prepare_transform_summary()?; + + let prediction = QUALITY_MODEL.predict( + target, + ctx.request().rate_preset, + &source, + &transform, + ); + + if prediction.ood.any() + || prediction.interval_log_width > FALLBACK_LOG_WIDTH + || prediction.saturation_risk.is_high() + { + return continue_exact_controller( + ctx, + target, + prediction.median_rung, + Vec::new(), + budget, + ); + } + + let first_rung = prediction.candidate_rung.ceil_finer(); + let first = plan_and_score_fresh(ctx, first_rung, target)?; + + if first.score >= target { + if first.score - target <= MET_OVERSHOOT_BAND + || prediction.predicted_tightening_saving < 0.03 + { + return entropy_attach_and_emit(ctx, first); + } + + // Optional rare byte-tightening attempt. Keep `first` until the + // coarser candidate has been verified. + let tighter_rung = + correction_rung(first_rung, first.score, target, prediction.local_beta, false); + + if tighter_rung < first_rung { + let tighter = plan_and_score(ctx, tighter_rung, ReusePolicy::ByDistance)?; + if tighter.score >= target { + return entropy_attach_and_emit(ctx, tighter); + } + } + + return entropy_attach_and_emit(ctx, first); + } + + let corrected_rung = + correction_rung(first_rung, first.score, target, prediction.local_beta, true); + + let corrected = plan_and_score(ctx, corrected_rung, ReusePolicy::ByDistance)?; + + if corrected.score >= target { + return entropy_attach_and_emit(ctx, corrected); + } + + // Do not restart. Seed the existing navigator with both measured probes, + // the warm transform cache, and the remaining total budget. + continue_exact_controller( + ctx, + target, + prediction.median_rung, + vec![first.into_probe(), corrected.into_probe()], + budget.remaining_after(2), + ) +} +``` + +The correction formula should use the same perceptual-loss domain already used by the navigator: + +```text +L(q) = max(100 - q, 1e-3) +β = -d ln(L) / d ln(scale), β > 0 + +ln(scale₂) = + ln(scale₁) + [ln L(observed_score) - ln L(aim_score)] / β +``` + +Then: + +* round to the finer legal rung when correcting a miss; +* clamp the jump to the existing bounded ratio; +* fall back to the current extrapolation logic when `β` is invalid; +* rebuild cover/CfL when the scale displacement exceeds the existing structural threshold or the model marks the region as structurally unstable. + +Unlike the current first-rung predictor, this model supplies the local slope needed to use the first measured score efficiently rather than geometrically exploring again. + +--- + +# 4. Exact work counts + +“One forward transform” here means one request-scoped population of the candidate transform banks: each legal `(LF group, origin, transform type)` is computed at most once and reused. It does not mean that the image contains only one DCT operation. + +| Runtime route | Transform-bank fills | Pixel plans | Full-frame reconstructions | Canonical metric evaluations | Entropy trainings | Codestream emissions/exact prices | Fresh cover/CfL builds | +| ------------------------------------------------------ | ---------------------: | -----------: | -------------------------: | ---------------------------: | ----------------: | --------------------------------: | ---------------------: | +| Predicted normal success | 1 | 1 | 1 | 1 | 1 | 1 | 1 | +| Predicted miss, one successful correction | 1 shared | 2 | 2 | 2 | 1 | 1 | 1–2 | +| Predicted overshoot, one byte-tightening attempt | 1 shared | 2 | 2 | 2 | 1 | 1 | 1–2 | +| Preflight uncertainty/OOD to Balanced exact controller | 1 shared | ≤5 | ≤5 | ≤5 | ≤2 | ≤2 | ≤2 | +| Prediction followed by exact continuation | 1 shared | **≤5 total** | **≤5 total** | **≤5 total** | ≤2 total | ≤2 total | ≤2 total | +| Saturation or work-cap failure under hard-floor API | 1 shared | ≤5 | ≤5 | ≤5 | 0 | 0 | ≤2 | +| Explicit lossless fallback | Separate lossless path | — | — | optional verification | lossless backend | 1 | — | + +Important implementation details: + +* Prediction attempts consume the normal total probe budget. +* There is no hidden sixth rescue probe. +* Failed plans receive no entropy training. +* The normal path never holds two frame-sized pixel plans. +* The byte-tightening path may temporarily retain two plans, but only under a rare explicit trigger. +* An under-target terminal result is not exact-priced unless the caller explicitly requested best effort. + +The handoff shows that current Balanced usually performs four probes and two exact prices, while the wall ratio is 3.07× to 8.00× the matched-rate path. Removing approximately three reconstructions/metric evaluations from the median request is therefore the correct performance target. + +--- + +# 5. What the model should predict + +## Primary output: the image-specific score/scale crossing curve + +Do not predict bitrate as the primary output. + +Use: + +```text +x = ln(effective_scale) +z = ln(max(100 - SSIMULACRA2, 1e-3)) +``` + +For the seven target knots 30, 50, 70, 80, 85, 90, and 95, predict: + +1. median fresh-structure crossing `x50`; +2. risk-adjusted candidate crossing, initially approximately `x90`; +3. lower and upper calibrated crossing quantiles; +4. local positive loss exponent `β`; +5. saturation/reachability risk; +6. out-of-distribution flags; +7. optional estimated exact bytes at the crossing; +8. optional structural-instability risk. + +For arbitrary target values, interpolate between target knots in log perceptual loss, not directly in score. + +A suitable runtime result is: + +```rust +pub struct QualityPredictionV2 { + pub median_rung: Rung, + pub candidate_rung: Rung, + pub interval_low: Rung, + pub interval_high: Rung, + pub local_loss_exponent: f32, + pub saturation_risk: SaturationRisk, + pub structural_risk: StructuralRisk, + pub ood: OodFlags, + pub predicted_bytes: Option<u64>, + pub model_version: QualityModelVersion, +} +``` + +## Why not predict bitrate? + +The existing rate controller still performs multiple exact writer prices and bounded corrections. Passing a predicted bitrate to it would reorganize the search rather than remove it. + +It also leaves two mappings to solve: + +```text +source + score target -> bitrate +bitrate -> quantizer/policy +``` + +Direct effective-rung prediction solves the actual expensive decision. + +The rate module remains useful for: + +* legal rung/effective-scale mappings; +* quantizer construction; +* auxiliary byte labels; +* comparative evaluation. + +It should not be the normal score-target backend. + +## Why not predict a local or per-frequency quantization field yet? + +That introduces many more degrees of freedom and changes the codec’s psychovisual policy at the same time as the controller is being replaced. It would make failures difficult to attribute. + +First establish that a scalar effective-rung predictor can remove repeated score probes without losing the byte advantage. Local/per-frequency optimization can then become a separate quality-efficiency project. + +## Why not predict policy choices initially? + +Keep the production Balanced policy fixed: + +* restoration fixed by the request/default; +* hierarchical cover; +* normal CfL behavior; +* Balanced entropy effort; +* no Quality policy bank. + +Entropy policy does not affect reconstructed pixels and therefore does not need score prediction. + +A later model may select between a very small, independently validated set of source-only policies, but each policy would need its own calibrated crossing curve. The failed Quality promotion screen is a reason not to make policy prediction part of the first controller. + +--- + +# 6. Model form and deterministic implementation + +## Recommended model + +Use a small, code-generated monotone generalized additive model: + +```text +crossing_k = + target_intercept_k + + Σ piecewise_linear_feature_term_jk(feature_j) + + a small number of predeclared interactions +``` + +Fit separate models for: + +* the median crossing; +* the candidate quantile; +* the upper uncertainty bound; +* local slope; +* saturation risk. + +The first implementation should contain no more than a small set of measured interactions, such as: + +* noise × flatness; +* high-frequency energy × bit depth; +* flatness × orientation coherence; +* chroma/luma energy ratio × CfL correlation. + +Do not add interactions simply because the trainer can fit them. + +## Monotonicity + +For each image, predict the seven target-knot crossings, then project them to a nondecreasing effective-scale sequence: + +```text +x30 <= x50 <= x70 <= x80 <= x85 <= x90 <= x95 +``` + +Round the candidate scale toward the finer legal rung. + +The same projection must be applied to median and upper-quantile curves. + +This guarantees that the **chosen predicted rung** does not become coarser as target increases. It does not provide a mathematical proof that canonical SSIMULACRA2 itself is monotone across every independently encoded request, because the real score curve can have local reversals. Promotion should therefore require zero achieved-score inversions on the full target grid, while the documentation should not describe this as a theorem for arbitrary images. + +## Determinism + +Model inference should: + +* run in scalar Rust; +* use generated constants; +* reduce per-group feature accumulators in fixed LF-group order; +* avoid worker-order floating-point reductions; +* use explicit rounding before converting to a rung; +* include exact model and feature-schema versions in traces. + +Fixed-point model constants are worth considering if f64 boundary behavior differs across supported architectures. A scalar f64 implementation with conservative rung rounding may already be sufficient, but this must be tested across SIMD modes and worker counts. + +--- + +# 7. Feature design + +The current production atlas is a useful foundation. The diagnostic `AnalysisAtlasV2` also already contains several promising feature candidates, but it currently performs a separate pass and stores a large diagnostic structure. It should be used for offline ablation first, not installed wholesale in the production path. + +## Proposed features and costs + +| Feature group | Purpose | Source-only analysis | Final forward transforms | Candidate reconstruction | Entropy/emission | +| --------------------------------------------------------------------------------------- | ------------------------------------------- | -------------------: | -----------------------: | -----------------------: | ---------------: | +| Width, height, log area, aspect ratio, edge-partial-atom fraction, bit depth, grayscale | Size and metric-scale effects | Yes | No | No | No | +| Luma variance histogram and q10/q50/q90/q99 | Flat/texture distribution | Yes | No | No | No | +| Chroma variance and chroma/luma ratios | Colour complexity | Yes | No | No | No | +| Flat fraction plus low-variance tail | Smooth fields and gradients | Yes | No | No | No | +| Channel clipping/saturation fractions and dynamic range | High-target and saturated-content risk | Yes | No | No | No | +| Horizontal/vertical gradient energy and cross term | Edge orientation and anisotropy | Yes | No | No | No | +| Laplacian energy and affine-plane residual | Detail versus smooth ramp separation | Yes | No | No | No | +| Noise MAD and noise-to-edge ratio | Low-light/noise quantization sensitivity | Yes | No | No | No | +| Orientation coherence and flat-side asymmetry | Text, line art, and edge leakage risk | Yes | No | No | No | +| Channel covariance | CfL benefit and chroma reconstruction risk | Yes | No | No | No | +| DCT energy by channel and radial frequency band | Direct quantization sensitivity | No | Yes | No | No | +| Frequency-energy q50/q90/q99 and tail ratios | Sparse detail versus broadband texture | No | Yes | No | No | +| Directional AC energy | Lines, hatching, directional textures | No | Yes | No | No | +| DCT8/DCT16/DCT32 candidate energy or cost gaps | Cover preference and structural instability | No | Yes | No | No | +| Coefficient zero/nonzero counts at a few reference scales | Approximate scale sensitivity | No | Yes | No | No | +| Transform-domain luma/chroma correlation | CfL sensitivity | No | Yes | No | No | +| Cheap token/run histogram proxy | Auxiliary byte prediction | No | Yes | No | No | + +No proposed final-decision feature requires a candidate reconstruction, canonical score, entropy training, or exact emission. + +## Production feature extraction rules + +1. Replace full vector sorting with deterministic histograms. +2. Accumulate source features during the existing atlas pass where practical. +3. Use `AnalysisAtlasV2` only to identify useful signals initially. +4. Fuse winning diagnostic signals into a compact streaming production summary. +5. Build transform summaries directly from `CandidateForwardCache`; do not copy all coefficient banks into a second representation. +6. Reduce transform summaries per LF group and combine them in fixed raster order. +7. Reject any feature set whose additional normal-path wall exceeds 5% of the matched-rate path. + +--- + +# 8. One final transform can support prediction and emission + +The precise dependency cycle in the current planner is: + +```text +quantizer + -> AqSetup::build + -> LfQuantizer and HfQuantizers + -> quantizer-dependent cover objective + -> selected transforms + -> CfL over selected coefficients + -> quantization +``` + +Therefore, the final cover and CfL cannot be selected before the quantizer is known. + +However, the forward transform coefficients themselves are quantizer-independent. The existing `CandidateForwardCache` is explicitly designed so that a given `(origin, transform)` is computed at most once per request. `ensure_cover_candidates_cached()` can populate aligned DCT8, DCT16, and DCT32 candidates. + +The cycle can therefore be broken as follows: + +```text +source + -> quantizer-independent candidate transform bank + -> aggregate transform features + -> predict quantizer + -> quantizer-dependent AQ and cover selection + -> selected transforms read from same bank + -> CfL + -> quantization +``` + +This is genuinely one transform bank supporting both prediction and final emission. + +### Required refactor + +Add a method approximately like: + +```rust +impl CandidateSearchContext<'_> { + pub(crate) fn prepare_quality_transform_summary( + &mut self, + ) -> Result<TransformFeatureSummary>; +} +``` + +It should: + +* prepare geometry; +* fill the transform candidates required by the Balanced cover; +* calculate compact feature accumulators; +* leave all cache entries available to the later pixel planner. + +There is one caveat: eagerly filling every candidate may compute more than the current serial lazy cover path on some images. The existing parallel hierarchical path already has a complete-cache mechanism, but the exact incremental cost and 50 MP memory behavior must be measured. If full prefill is too expensive, begin with DCT8 summaries or another subset that the final cover path already necessarily computes. + +Restoration policy must remain fixed in the first version because Gaborish changes the transform frame itself. Predicting restoration would move the policy decision before this shared transform stage and reopen the dependency problem. + +--- + +# 9. Training and calibration procedure + +## 9.1 Training labels + +Create a new oracle dataset from the actual production Balanced pixel policy. + +For each independent source family: + +1. build source and transform features once; +2. evaluate the full effective-scale ladder, including `HfMul` segments; +3. adaptively densify around target crossings; +4. rebuild cover/CfL fresh for oracle crossing points; +5. record canonical SSIMULACRA2; +6. record the coarsest fresh rung meeting each target; +7. record local score/loss slope; +8. exact-price the crossing and nearby feasible rungs; +9. record top-rung score and whether the target is reachable; +10. optionally record anchor-reuse versus fresh-structure differences. + +The seven target rows from one image are correlated observations, not seven independent samples. Weight each source family equally, dividing its weight across targets and derived resolutions. + +## 9.2 Saturation is censored data + +Do not assign the top rung as if it were the actual crossing when the top score misses. + +Record: + +```text +crossing > top_rung +``` + +and train a separate transparent reachability/saturation model. + +At runtime: + +* model saturation risk is only a routing hint; +* actual `SaturatedTop` requires canonically scoring the real top candidate below target; +* uncertainty and saturation remain separate concepts. + +This is especially important because the current calibration reports saturation fractions of 0.81 and 0.94 at targets 90 and 95. + +## 9.3 Loss functions + +Fit at least three crossing models: + +### Median crossing + +Use Huber or median quantile loss on: + +```text +ln(required_effective_scale) +``` + +### Candidate crossing + +Initially use quantile loss at approximately `τ = 0.90`: + +```text +ρτ(error) +``` + +At `τ = 0.90`, predicting too coarse is penalized approximately nine times as much as predicting too fine. + +This quantile is not a safety guarantee. It is a wall-versus-byte operating point. The verification gate supplies safety. + +### Upper uncertainty bound + +Fit a higher quantile such as `τ = 0.95` and calibrate its empirical coverage by whole family on development data. + +### Byte-regret term + +Add a modest penalty for unnecessarily fine predictions: + +```text +λbytes * max(0, ln(bytes(predicted_rung) / bytes(crossing_rung))) +``` + +This prevents the asymmetric floor penalty from collapsing into a permanently overfine controller. + +### Slope model + +Use robust regression for: + +```text +β = -d ln(perceptual_loss) / d ln(effective_scale) +``` + +with a positivity constraint and bounded runtime range. + +## 9.4 Uncertainty and OOD detection + +Use both model residuals and feature support. + +Fallback should fire when any of the following occurs: + +* calibrated log-scale interval width exceeds `ln(1.5) ≈ 0.405`; +* a hard feature lies outside the calibrated production envelope; +* distance to the nearest training prototype exceeds the development-set 99th percentile; +* predicted upper crossing reaches the top 2% of the effective ladder; +* saturation risk exceeds the development-calibrated threshold; +* transform-size preference is unusually ambiguous; +* target/bit-depth/content combination has insufficient family support. + +The runtime OOD implementation can remain simple: + +* robust min/max ranges; +* median and MAD standardization; +* a small code-generated set of feature-space medoids; +* deterministic L1 or diagonal-distance calculation. + +Do not describe these intervals as universal confidence guarantees. They are empirical in-domain calibration used to decide whether the cheap path is worthwhile. + +## 9.5 Generated-model provenance + +Every generated model should embed or accompany: + +* model schema version; +* feature schema version; +* SSIMULACRA2 metric version; +* encoder git revision; +* trainer git revision; +* corpus-manifest hash; +* split-manifest hash; +* label-generation command; +* label dataset hash; +* training configuration; +* generated Rust checksum; +* full development and holdout report. + +A single reproducible command should regenerate both the Rust constants and the report. Generated diffs should be human-reviewable. + +--- + +# 10. Corpus assessment and expansion + +The current corpus contains 47 images split 19/15/13 across calibration, development, and holdout. It includes the requested broad classes, but it is not sufficient to calibrate production uncertainty. + +It is sufficient for: + +* rejecting obviously inadequate features; +* comparing source-only versus transform-derived predictors; +* determining whether the median and tail errors improve materially; +* building the first shadow model. + +It is not sufficient for: + +* a trustworthy 90th or 95th percentile crossing model; +* OOD thresholds; +* high-target saturation calibration; +* public promotion across every content class. + +## Required manifest changes + +Add: + +```json +{ + "family_id": "...", + "variant_id": "...", + "generator_family": "...", + "source_capture_id": "..." +} +``` + +All crops, resolutions, colour variants, recompressions, and derivatives of one source must remain in the same split. + +The existing manifest includes multiple resolutions of individual photographs. They appear to remain within one split, but without `family_id` that cannot be mechanically audited or weighted correctly. + +## Expansion targets + +For a first feature-gated implementation: + +* at least **150 independent source families**; +* at least 25 independent families in each critical non-photo class; +* at least 30 blind holdout families. + +For public promotion: + +* at least **300 independent source families**; +* at least 60 never-tuned promotion-holdout families; +* 8-, 10-, 12-, and 16-bit inputs; +* dimensions from metric minimums through 50 MP; +* multiple camera and rendering sources; +* real and synthetic text/UI; +* line art and diagrams; +* smooth ramps and sky gradients; +* saturated wide-colour stress; +* low-light and structured noise; +* grayscale; +* tiny images; +* high-detail natural photographs. + +The current 13-image locked holdout should remain untouched as a legacy regression holdout. The larger expansion should create a new blind promotion holdout rather than recycling those 13 images for threshold selection. + +--- + +# 11. Quantitative experiment and promotion gates + +The handoff requires Balanced wall no more than 2.0× the matched-rate path on the two anchors. + +The following thresholds should be frozen before running the experiment. + +## 11.1 Prediction quality, non-saturated family holdouts + +| Metric | Required | | | +| ---------------------------------------------------------- | ---------------------------------------: | - | ----- | +| Median ` | ln(predicted crossing / oracle crossing) | ` | ≤0.10 | +| p90 absolute log error | ≤0.30 | | | +| p99 absolute log error | ≤0.70 | | | +| First-plan success, reachable in-domain targets 30–90 | ≥90% | | | +| First or one-correction success | ≥98% | | | +| Exact-controller continuation on reachable in-domain cells | ≤5% | | | +| Successful outputs below requested score | **0** | | | + +Target 95 should be reported separately until its current saturation rate is reduced or properly routed. + +A quick falsification gate may be looser: + +* p90 log error ≤0.45; +* at least 80% first-plan success; +* no more than 1.5% simulated byte regression. + +Failure there means the transparent one-shot model is not yet viable and production controller work should stop. + +## 11.2 Byte efficiency + +The current Balanced score controller has a 0.99281 geometric-mean ratio to the interpolated rate curve. That leaves only approximately 0.724% before the ratio reaches 1.0. + +Therefore the public promotion limits should be: + +| Metric | Required | +| ---------------------------------------------------- | -------: | +| Geometric-mean bytes versus current score controller | ≤1.007 | +| Geometric-mean bytes versus same-effort rate curve | ≤1.000 | +| Mean per-image BD-rate versus same-effort rate curve | ≤0% | +| p95 per-cell byte ratio versus current controller | ≤1.03 | +| Worst reachable cell byte ratio | ≤1.08 | +| Median score overshoot on first-plan successes | ≤0.35 | +| p90 score overshoot | ≤1.0 | + +A shadow prototype may initially allow a 1.5% geometric-mean regression, but that is not enough for promotion because it would discard the controller’s existing aggregate rate advantage. + +## 11.3 Wall + +| Measurement | Acceptance | Design target | +| ----------------------------------------- | ---------: | ------------: | +| 4.3 MP anchor / matched-rate | ≤2.0× | ≤1.50× | +| 12 MP anchor / matched-rate | ≤2.0× | ≤1.75× | +| Incremental source/transform feature wall | ≤5% | ≤3% | +| Normal-path full reconstructions | 1 | 1 | +| Normal-path entropy trainings/emissions | 1 | 1 | + +A canonical metric evaluation remains unavoidable. Therefore the practical lower bound is: + +```text +one Balanced pixel plan ++ one full reconstruction ++ one canonical metric ++ one entropy attachment/emission +``` + +If that isolated path still exceeds 2.0×, the remaining problem is metric/render cost rather than controller navigation. That should be measured directly rather than hidden behind further prediction work. + +## 11.4 Memory + +| Measurement | Required | +| --------------------------------------------- | -------------------------------: | +| Normal-path peak RSS versus matched-rate path | ≤1.10× | +| Fallback peak RSS | no worse than current controller | +| Normal retained frame-sized pixel plans | 1 | +| Normal duplicate transform payloads | 0 | + +## 11.5 Determinism and monotonicity + +Promotion requires: + +* byte-identical output at supported worker counts; +* identical decision path and status across worker counts; +* identical result across supported SIMD modes; +* nondecreasing predicted rung across the full target grid; +* zero achieved-score inversions on the promotion corpus; +* zero byte inversions on the target grid unless explicitly justified by a smaller exact stream with a higher achieved score; +* independent decoding and re-scoring equal to the reported score within the established metric tolerance. + +--- + +# 12. Minimal implementation sequence + +## PR 0: repair the existing contract + +This is separate from the predictive experiment. + +1. Do not return under-target bytes as `Ok`. +2. Make saturation and work exhaustion structured failures. +3. Make CLI output atomic and avoid creating an output file on failure. +4. Add explicit lossless and best-effort fallback policies if desired. +5. remove the hidden extra rescue probe or put it inside the documented cap. +6. Add public API tests for saturation and work exhaustion. + +This changes behavior only where the current implementation already contradicts the public hard-floor contract. + +## PR 1: trace and corpus schema, no encoding change + +Create `jpxl.quality-trace/2` with: + +```text +model_version +feature_schema +median_rung +candidate_rung +interval_low/high +local_loss_exponent +saturation_risk +ood_flags +fallback_reason +first_observed_score +correction_rung +decision_path +total pixel plans +total reconstructions +total metric evaluations +total entropy trainings +total emissions +``` + +Add `family_id` and related fields to the corpus manifest. + +The current controller remains authoritative. + +## PR 2: production-endpoint label generator + +Replace or supplement `calibrate_initial_rung.py` with a trainer that: + +* operates over the full effective-scale ladder; +* labels production Balanced fresh-structure crossings; +* represents saturation as censoring; +* records local slopes and neighboring exact bytes; +* splits and weights by family; +* emits reproducible dataset and model provenance. + +Existing controller traces can be used as provisional labels for the first falsification pass, but not as final oracle labels. + +## PR 3: source-only shadow model + +Implement `QualityPredictionV2`, but only log its counterfactual decision. + +Compare: + +* current two-feature table; +* all existing `SourceFeatures`; +* histogram-based source features; +* selected `AnalysisAtlasV2` aggregates. + +Do not let the model affect a bitstream. + +## PR 4: transform-summary shadow model + +Add the request-scoped transform-summary API to `CandidateSearchContext`. + +Again, only log: + +* candidate rung; +* interval; +* predicted correction; +* whether exact fallback would have fired; +* counterfactual one-shot work counts; +* predicted byte regret. + +The current exact controller still chooses output. + +## Gate before production integration + +Proceed only when family-held-out results satisfy: + +* median log-scale error ≤0.10; +* p90 ≤0.30; +* at least 80% first-plan success in the quick screen; +* no more than 1.5% simulated byte regression; +* incremental feature cost ≤5%; +* no opaque model required. + +If transform features cannot reach that gate, do not move candidate reconstructions into the feature extractor and call the result one shot. + +## PR 5: feature-gated common-case path + +Only after the shadow gate: + +* make the predicted plan the first real plan; +* use canonical verification before entropy; +* add one slope correction; +* continue the existing navigator from measured observations; +* maintain the same total work caps; +* retain current exact controller as the default fallback. + +Public promotion then uses the stricter wall and byte gates above. + +--- + +# 13. Expected tradeoffs + +## Wall + +The proposed normal route eliminates: + +* geometric expansion probes; +* repeated bracket tightening; +* repeated candidate reconstruction; +* one of the usual exact finalist prices. + +Given the current median of four probes and two exact prices, it should materially approach the single Balanced path, although one full canonical metric remains unavoidable. + +## Memory + +Normal-path memory should improve because only one pixel plan needs to survive until entropy attachment. The transform cache is already request-scoped and shared. + +The main memory risk is eagerly completing all transform candidates before prediction on 50 MP inputs. That must be measured before promotion. + +## Score floor + +The score floor becomes stronger than the current public implementation because: + +* under-target streams are no longer returned as ordinary successes; +* every emitted lossy stream has passed canonical verification; +* uncertainty changes the work path, not the correctness promise. + +## Bytes + +Byte efficiency is the hard part. + +A highly conservative predictor can easily turn a target of 85 into an achieved score above 90 and waste substantial bytes. The model therefore needs both: + +* an asymmetric candidate quantile to keep first-plan misses uncommon; +* an explicit byte-regret term and optional rare coarsening correction. + +The existing controller’s modest rate advantage leaves little room for aggregate regression. A model that is fast but routinely overfine should not be promoted. + +## Complexity + +Runtime complexity remains modest: + +* one compact feature summary; +* one generated additive model; +* one prediction structure; +* one continuation path into the existing navigator. + +No new heavyweight dependency or neural runtime is justified at this stage. + +--- + +# 14. Principal risks + +1. **Family leakage.** Resolutions or synthetic variants of one source can make prediction appear substantially better than it is. + +2. **High-target censoring.** Treating top-rung saturation as an ordinary crossing corrupts both the scale model and uncertainty. + +3. **Structural discontinuities.** Cover and CfL can change around the crossing, making a smooth source-only score curve inaccurate. + +4. **Metric nonmonotonicity.** Running-max training curves hide local reversals that the runtime may still encounter. + +5. **Entropy nonmonotonicity.** The coarsest feasible rung is not always the smallest exact stream. + +6. **Overconservative calibration.** A large reserve can meet the floor while silently discarding the present byte advantage. + +7. **Understated feature cost.** A second source pass, full coefficient copy, or eager transform work can erase the saved controller wall. + +8. **Transform-cache memory at 50 MP.** Full candidate prefill may alter peak RSS even when it avoids recomputation. + +9. **Model staleness.** Changes to quantization, AQ, cover cost, CfL, restoration, metric version, or entropy policy invalidate labels. + +10. **Misrepresented uncertainty.** Empirical quantiles do not provide arbitrary-image guarantees. The canonical score gate must remain authoritative. + +11. **Hidden fallback work.** Restarting the exact controller after one or two prediction attempts would make the apparent one-shot path more expensive than the current controller. + +12. **Benchmark-specific routing.** Tuning fallback thresholds to the 13 locked images would create a demonstration rather than a general controller. + +13. **Literal contract mismatch.** Neither current bounded search nor one-shot prediction globally minimizes all possible exact codestreams. The bounded effort domain must be explicit. + +--- + +# Final recommendation + +Implement this in three conceptual layers: + +1. **Correctness layer:** under-target output is never a normal success. +2. **Prediction layer:** a transparent monotone model predicts the production Balanced fresh-structure crossing, local slope, interval, saturation risk, and OOD state from source plus shared transform features. +3. **Execution layer:** one predicted plan is canonically verified; one correction is allowed; the existing navigator continues from those observations under the same total cap. + +Do not route through the bitrate controller, do not predict a per-frequency field yet, and do not make policy-bank selection part of the first model. + +The key design principle is: + +> The predictor determines where JPXL should look first. Canonical verification determines whether JPXL is allowed to emit. The existing bounded controller remains the recovery mechanism, not the normal path. + +That is the narrowest design that can plausibly move Balanced from four full-frame probes toward one while honestly preserving the hard per-image score floor and the encoder’s existing byte-efficiency advantage.