diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index da795a1af3..9fe2cb1453 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -76,7 +76,7 @@ producing an invalid bitstream. They are separate from the reconstruction-input ### Architecture and verification findings -Fresh correction verification on 2026-09-05: +Reference-edge correction checkpoint `578ec34d9`, verified on 2026-09-05: - Final Release .NET 11 build after removing the screening hook: zero errors and zero reported warnings on that incremental build. The preceding test compilation reported 1,009 existing warnings. @@ -96,6 +96,32 @@ Fresh correction verification on 2026-09-05: output parity, complete encoder control flow, or decoder-wide conformance. No benchmark was run. Temporary launch scripts, native comparison output, logs, and reports remain local and are excluded from commits. +Sequence-construction ownership correction, verified subsequently on 2026-09-05: + +- At checkpoint `578ec34d9`, `Av1FrameEncoder.cs:1492-1560,1668-1713,1778-1824` allocated common state and + frame owners without unwinding partial construction. `Av1EncoderPictureBuffer.cs:87-164` and + `Av1SymbolEncoder.cs:270-274` had the same problem inside their multi-owner constructors. A failed constructor + never returns an instance to the caller's `using` statement. This is a demonstrated lifetime defect, independent + of encoder quality. Reference `av1/encoder/encoder.c:1480-1499` clears compressor state and invokes + `av1_remove_compressor` if construction fails. +- The new production-factory regression failed before the correction: rejecting the second allocation left the + first allocation unreturned. The first failure stopped VSTest as configured; evidence is + `D:\GitHub\ynse01\av1-takeover-20260905\allocation-failure-before.trx`. +- Common sequence state, both sample-width frame constructors, picture state, and symbol state now unwind + completed owners at their own construction boundaries. Picture context views borrow its two owners + (`Av1NeighborArrayUnit.cs:65-70,135-141`), so failed picture construction releases those owners directly. + Common cleanup avoids dispatching into derived frame disposal before derived construction begins. + Nullable disposal checks are restricted to owners that can be absent after an allocation failure; no new + owners, buffers, copies, or native dependencies were introduced. +- Six color/alpha cases at 8/10/12 bits reject every allocator request in turn and require exactly one return + for every earlier successful request. All six pass. The final affected frame, superblock, and public encoder + set passes 257/257 cases through serialized Release .NET 11 Visual Studio VSTest with stop-on-failure. + The report is `D:\GitHub\ynse01\av1-takeover-20260905\ownership-final.trx`. +- The final incremental Release .NET 11 build reports zero warnings and errors; Roslynk reports zero compiler + errors. The twelve regenerated color sequences and two mixed-partition streams were compared again with the + optimized native decoder: 23,396 samples, maximum error 0, zero samples exceeding one. This remains bounded + same-stream decoder/reconstruction evidence, not separate-encoder parity. No benchmark was run. + - PNG resolves options before converted metadata and sanitizes incompatible output combinations (`src/ImageSharp/Formats/Png/PngEncoderCore.cs:1632-1675`). TIFF follows the same precedence and converts unsupported combinations (`src/ImageSharp/Formats/Tiff/TiffEncoderCore.cs:111-175,371-457`). @@ -119,6 +145,47 @@ Fresh correction verification on 2026-09-05: ### Required completion gates +Motion-controller investigation continued after correction checkpoint `578ec34d9`: + +- Managed `Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:1278-1493` uses the same normalized squared-error + plus complete mode/vector RD cost for integer and fractional candidates. The integer operators at + `Av1IntraSuperblockEncoder.Operator.cs:466-493,985-1014` compute squared error, not SAD or centered variance. +- Reference `av1/encoder/mcomp.c:72-130,184-238,314-384,644-664` separates full-pixel SAD cost, variance cost, + SAD-per-bit scaling, error-per-bit scaling, and their motion limits. Its full-pixel dispatcher + (`mcomp.c:1768-1903`) owns the configured diamond/hexagonal/pattern search and conditional mesh search, + including downsampled-SAD fallback. +- Reference `av1/encoder/motion_search_facade.c:150-324,347-488` derives the starting search step, considers + configured start candidates, retains a second full-pixel candidate, prunes repeated dynamic-reference searches, + and can refine and compare both candidates. Fractional search + (`mcomp.c:3266-3337`) has configured precision, iteration count, repeated-position tracking, and a second-level + check. The managed one-ring-per-scale controller does not implement that path. +- The next motion implementation must reconcile configuration, limits, start-candidate lifetime, search costs, + full-pixel traversal, fractional traversal, and winner publication together. Substituting SAD or variance alone, + increasing the radius, or adding isolated search points would not establish that contract. No benchmark or + motion-search implementation change has been made from this follow-up investigation. + +- Reference good-quality speed policy is layered rather than a radius lookup. The defaults at + `av1/encoder/speed_features.c:2353-2364` select NSTEP, full eighth-pixel precision, two subpixel iterations, + and eight-tap search. Good-quality overrides at `speed_features.c:1248-1250,1308-1312,1370-1409` change + iteration count, search range, full-pixel and fractional methods, second-candidate refinement, and mesh pruning. + Resolution-dependent speed-six overrides at `speed_features.c:1029-1076` also select block-size-dependent + search and reference-candidate pruning. These source observations do not make existing managed effort values + equivalent to native cpu-used values. +- Encoder intra-edge filtering requires the complete candidate prediction path. Reference + `av1/common/reconintra.c:958-986,1204-1243,1512-1548` derives neighbour-dependent strength, filters the corner + and required edges, then upsamples before directional prediction. The decoder already performs these stages + at `src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PredictionDecoder.cs:915-966`. Encoder luma, chroma, and + tiled candidates must consume the same filtered-reference contract before the sequence flag can be enabled. + The decoder's chroma smooth-neighbour test at `Av1PredictionDecoder.cs:1820-1837` does not repeat the native + inter-block check, but `Av1TileReader.cs:2037-2042` resets every inter block's UV mode to DC. That owning + invariant prevents stale smooth modes; no redundant guard or numerical defect is justified here. +- Sparse inverse dispatch remains incomplete. Reference + `av1/common/x86/highbd_inv_txfm_avx2.c:4088-4180` derives separate horizontal and vertical nonzero extents + from EOB and chooses low-one, low-eight, low-sixteen, or full DCT/ADST axis kernels as applicable. + Managed `Av1InverseTransformerFactory.cs:47-60,98-112` only distinguishes DC-only DCT from the full lossy + transform. The DC arithmetic at `Av1Inverse2dTransformer.cs:44-57` retains separate axis scaling and rounding; + agreement with the managed full path still does not independently establish all native sparse cases. + - [ ] Finish the full production-path and complete upstream-diff audit, including conversion, animation, ownership, filters, decoder SIMD, and independent validity of the claimed tests. - [ ] Reconcile frame configuration and encoder decision policy with the reference before isolated pruning changes. @@ -144,11 +211,11 @@ Fresh correction verification on 2026-09-05: Reference checkout evidence refreshed on 2026-09-05: - The current encoder source comparison uses the official libaom `main` revision `d565eec60f084421fa34fc0534b760c6452b6a6c`, exported at `D:\GitHub\ynse01\aom-d565eec6-source`. -- The current independently built reference decoder is `D:\GitHub\ynse01\aom-d565eec6-build-generic2\aomdec.exe`. Its CMake cache identifies the current source export above, and it reports version 3.15.0. On 2026-09-05 it accepted both frames of each retained-reference sequence at efforts five, seven, eight, and nine. This establishes syntax acceptance for those four streams, not complete interpolation or codec conformance. +- The earlier generic reference decoder was `D:\GitHub\ynse01\aom-d565eec6-build-generic2\aomdec.exe`; the fresh correction comparisons use the optimized x64 Release build recorded above. Its CMake cache identifies the current source export above, and it reports version 3.15.0. On 2026-09-05 it accepted both frames of each retained-reference sequence at efforts five, seven, eight, and nine. This establishes syntax acceptance for those four streams, not complete interpolation or codec conformance. ## Status notation -- [x] Verified: the current behavior has exact evidence from the current libaom `main` tree and the evidence proves the production contract. +- [x] Recorded checkpoint: the associated dated report claims focused verification. Historical marks outside the takeover section have not been accepted by the fresh audit and do not establish current-tree completeness. - [~] Locally implemented, checkpoint open: production source exists, but current-tree verification is missing or a known audit issue invalidates the checkpoint. - [ ] Remaining: the production behavior is absent, incomplete, or has not reached its required implementation boundary. @@ -169,14 +236,14 @@ Reconciled with the worktree on 2026-09-05. ## Immediate execution queue -Current interpolation-search checkpoint, implemented on 2026-09-05 with focused verification in progress: +Historical interpolation-search reports from before the takeover follow. Their timings and test counts apply only to the trees identified by those reports. The takeover audit and required completion gates above determine current work. -- [x] Complete elementary-stream decoder comparison now includes parsing, reconstruction, output allocation, RGB conversion, and disposal for both ImageSharp and optimized current-main libaom. Eight-bit Kodak and ten-bit Cosmos inputs match every `Rgb48` sample before timing. Initial warmed means are 15.130 versus 5.423 ms and 27.655 versus 10.246 ms respectively. These expose an open decoder gap; they are not container-load measurements. Reproduction and limitations are recorded in the benchmark README. -- [x] Shared inverse reconstruction now uses EOB to select a DC-only DCT path, preserving both axis roundings, rectangular normalization, input clamps, and final clipping through the existing semantic output operators. Vector512 output was added to that operator contract. Every transform size, signed boundary, padded separate/in-place destination, and supported sample precision is checked against the full transform. The post-change photographic encoder payload remained byte-identical; decoder RGB output remained exact against libaom. Short decoder measurements do not yet establish a statistically significant improvement. -- [~] Fixed 8x8 lossy luma search now screens candidates with the reference's unnormalized Hadamard magnitude and 1.5x best-model threshold. Existing tensor and transpose APIs vectorize the operation over frame-reused scratch; no per-candidate allocation or custom hardware-width operator was introduced. A five-warmup, ten-measurement repeat records 1,363.09 ms versus native 71.73 ms, about 34% faster than the initial managed baseline but still approximately 19x behind native. Output increased slightly to 11.577 KiB and aggregate native-plane PSNR declined from 37.124 to 37.069 dB. This tradeoff does not close the performance/compression gate. Larger-block screening, top-ranked pruning, transform bounds, winner refinement, filter decisions, and allocation attribution remain open. Evidence: `artifacts/BenchmarkDotNet/av1-screen-verified-short-20260905/20260905-140358`. -- [x] The current screening/DC/native-profile/moving-color subset passes 25 cases in each of three separately configured VSTest hardware tiers. Normal-path verification passes 84 screening/DC/superblock cases and 165 frame/public encoder cases. The photographic benchmark now additionally requires exact ImageSharp/libaom RGB agreement across all three dependent frames before timing. No expected image or golden output was changed. Release test/benchmark builds have zero errors and their existing 1,009/39 warning baselines. Evidence and the corrected process-local VSTest environment handling are recorded in `tests/ImageSharp.Benchmarks/Codecs/Heif/README.md` and `artifacts/TestResults/av1-screen-20260905`. +- [~] The earlier elementary-stream decoder comparison included parsing, reconstruction, output allocation, RGB conversion, and disposal for both ImageSharp and optimized current-main libaom. Eight-bit Kodak and ten-bit Cosmos inputs match every `Rgb48` sample before timing. Initial warmed means are 15.130 versus 5.423 ms and 27.655 versus 10.246 ms respectively. These expose an open decoder gap; they are not container-load measurements. Reproduction and limitations are recorded in the benchmark README. +- [~] Uncommitted shared inverse reconstruction uses EOB to select a DC-only DCT path, preserving both axis roundings, rectangular normalization, input clamps, and final clipping through the existing semantic output operators. Vector512 output was added to that operator contract. Every transform size, signed boundary, padded separate/in-place destination, and supported sample precision is checked against the full transform. The post-change photographic encoder payload remained byte-identical; decoder RGB output remained exact against libaom. Short decoder measurements do not yet establish a statistically significant improvement. +- [ ] The isolated fixed-8x8 Hadamard screening hook was removed in correction checkpoint `578ec34d9`. The following is a historical experiment, not an active implementation or an accepted improvement. Existing tensor and transpose APIs vectorize the operation over frame-reused scratch; no per-candidate allocation or custom hardware-width operator was introduced. A five-warmup, ten-measurement repeat records 1,363.09 ms versus native 71.73 ms, about 34% faster than the initial managed baseline but still approximately 19x behind native. Output increased slightly to 11.577 KiB and aggregate native-plane PSNR declined from 37.124 to 37.069 dB. This tradeoff does not close the performance/compression gate. Larger-block screening, top-ranked pruning, transform bounds, winner refinement, filter decisions, and allocation attribution remain open. Evidence: `artifacts/BenchmarkDotNet/av1-screen-verified-short-20260905/20260905-140358`. +- [~] The earlier screening/DC/native-profile/moving-color subset passed 25 cases in each of three separately configured VSTest hardware tiers. Normal-path verification passes 84 screening/DC/superblock cases and 165 frame/public encoder cases. The photographic benchmark now additionally requires exact ImageSharp/libaom RGB agreement across all three dependent frames before timing. No expected image or golden output was changed. Release test/benchmark builds have zero errors and their existing 1,009/39 warning baselines. Evidence and the corrected process-local VSTest environment handling are recorded in `tests/ImageSharp.Benchmarks/Codecs/Heif/README.md` and `artifacts/TestResults/av1-screen-20260905`. -- [x] The reference benchmark now measures identical RGB-to-OBU boundaries, including pixel conversion for every frame on both sides. A benchmark-only C adapter links the optimized current-main libaom build in-process and borrows the existing converted plane storage; production remains fully managed. Setup and file I/O are excluded equally. Native quantizer bounds are fixed to the managed base index, with independent speed settings and explicit size/quality reporting. Reproduction, native build provenance, lifetime documentation, and exact output hashes are in `tests/ImageSharp.Benchmarks/Codecs/Heif/README.md`. +- [~] The earlier reference benchmark measured RGB-to-OBU boundaries, including pixel conversion for every frame on both sides. A benchmark-only C adapter links the optimized current-main libaom build in-process and borrows the existing converted plane storage; production remains fully managed. Setup and file I/O are excluded equally. Native quantizer bounds are fixed to the managed base index, with independent speed settings and explicit size/quality reporting. Reproduction, native build provenance, lifetime documentation, and exact output hashes are in `tests/ImageSharp.Benchmarks/Codecs/Heif/README.md`. - [ ] The corrected benchmark exposes a substantial remaining performance and compression gap. For three photographic 256x256 frames, ImageSharp effort seven takes 2,053.31 ms and writes 11.54 KiB at 37.124 dB aggregate native YUV PSNR; current-main libaom cpu-used six takes 70.67 ms and writes 8.27 KiB at 38.942 dB. Both include RGB conversion and both measured outputs decode to all three complete frames. This is fixed-base-quantizer evidence, not equal-quality evidence. The managed path records 9.38 MiB of managed allocations per operation, which still requires attribution; native memory is not measured by that counter. The Short-run evidence is `artifacts/BenchmarkDotNet/av1-sequence-rgb-fixed-q-short-20260905/20260905-131149`. Power-plan and CPU-query warnings remain documented. The new benchmark files build in net11.0 Release and have no Roslyn compiler/analyzer diagnostics. Do not close the interpolation-performance gate or advance to additional reference tools until the gap is addressed. - [x] The requested in-progress tree was committed as `433afd1a9` before further encoder work. That commit is a checkpoint, not a claim of completed interpolation or codec delivery. - [x] The subsequent partition/interpolation/lossless correction was committed as `7cf7fc4` after the 225-case affected encoder run and exact current-main native comparison. @@ -898,7 +965,7 @@ Final decoder allocation, lifetime, precision, architecture, and test-validity a parallelism disabled and stop-on-failure enabled: 8,746 of 8,746 cases passed. The touched `HeifDecoderTests` and `HeifSequenceParserTests` add 104 of 104 passing integration cases. Focused CDEF, restoration, film-grain, copy-ownership, and reference-isolation runs also pass 15 of - 15 cases. No test-host crash or Windows application-error dialog occurred. + 15 cases. The historical report recorded successful process exits; it did not independently establish absence of Windows application-error dialogs. Final decoder stream, presentation, and public-registration evidence on 2026-09-01: @@ -916,8 +983,7 @@ Final decoder stream, presentation, and public-registration evidence on 2026-09- `HeifSequenceParserTests` set with the registration contract: 115 of 115. The final explicit no-encoder registration assertion passes 1 of 1 after its final edit. - [x] The net11.0 Release test project builds with zero errors, Roslynk reports zero compiler errors, - `git diff --check` passes, and `.gitattributes` is unchanged. Every VSTest invocation returned - normally with no surviving test host and no Windows application-error dialog. + `git diff --check` passes, and `.gitattributes` is unchanged. The historical report recorded normal VSTest exits and no surviving test host; that does not establish absence of Windows application-error dialogs. SIMD traversal consistency evidence on 2026-09-02: @@ -981,15 +1047,15 @@ Encoder verification contract: - [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder operation now instantiates matching source and reconstruction owners with ordinary `using` lifetimes and converts packed pixels directly into the source owner before extension. - [~] Temporal delimiter, sequence header, frame header, combined-frame tile-group writing, uniform multi-tile layout, and reduced and non-reduced frame operations now exist locally. The remaining codec-tool and verification work is tracked below. - [~] Implement superblock and partition analysis for every permitted block size and partition. Efforts zero through eight deliberately split every in-frame node to 8x8 blocks. Effort nine performs recursive live rate-distortion selection at complete 8x8 and 16x16 nodes, while effort ten extends the same search to complete 32x32, 64x64, and 128x128 nodes. Candidate order matches current libaom: `PARTITION_NONE`, `PARTITION_SPLIT`, `PARTITION_HORZ`, `PARTITION_VERT`, the four asymmetric partitions, then `PARTITION_HORZ_4` and `PARTITION_VERT_4`; the two 1-to-4 partitions are excluded at 128x128 as required by current libaom. Invalid chroma geometries are excluded before evaluation. Each candidate saves and restores the exact partition, coefficient, transform, and palette neighbor edges in one aligned block-workspace owner; trials neither allocate nor copy probability state. Recursive split trials publish each selected child's decoded mode, transform, coefficient, and palette contexts before evaluating its next sibling. Coefficient contexts are published per retained transform rather than broadcasting the first transform over an entire partition leaf. Large luma and chroma leaves are evaluated as bounded-64, raster-ordered transform tiles in the existing aligned workspace, and each winning plane is copied to retained storage once. Production picture state retains the compact 8x8 mode allocation below effort nine and explicitly selects 4x4 allocation granularity when sub-8x8 partitions are enabled. Effort-dependent pruning remains. -- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes. Palette entropy, retained state, production syntax, exhaustive luma and paired chroma palette selection, adaptive screen-content activation, and joint intra-block-copy mode selection are complete. +- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes. Palette entropy, retained state, production syntax, luma and paired chroma palette selection, screen-content activation, and joint intra-block-copy mode selection exist, but their full reference decision policy and separate-encoder parity remain unverified. - [~] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. The sequence encoder retains the preceding reconstruction and, from effort six, searches a bounded full-pixel frame translation against that LAST_FRAME reference. Candidate discovery uses the existing SIMD-first squared-error kernels over a central analysis window, validates the winner over the complete coded luma plane, and charges its exact uncompressed-header bit count in the inter-frame rate-distortion domain. Pure translation is signaled as an identity-scale rotation/zoom model, matching current libaom's workaround for the AV1 translation-only axis defect. Each 8x8 inter-frame block first retains the complete intra candidate, then compares NEARESTMV, all three legal NEARMV dynamic-list entries, GLOBALMV, and all three legal NEWMV dynamic-list entries against it with live intra/inter, single-reference, mode, DRL, differential-vector, skip, transform, coefficient, and distortion costs. The initial NEWMV search retains the reference's cheap prediction-error stage, but every surviving mode now owns a complete transform, coefficient, skip, and distortion evaluation before mode selection; selected and candidate workspace views exchange ownership only on strict improvement. Inter trials remain in the existing shared workspace until they strictly beat the intra result, so losing trials require no backup buffer or copy. The tile writer emits the matching DRL path and normative context-selected `LAST_FRAME` reference tree instead of forcing every block through a segmentation feature. The selected DRL index reuses the filter-intra byte because those block syntax branches are mutually exclusive, preserving the existing packed state size. Packed short vectors reuse the existing picture-state owner. Effort six keeps a full-pixel fast path; effort seven refines each selected NEWMV through half- and quarter-pixel eight-tap prediction; effort eight adds the final eighth-pixel stage. The frame header advertises the matching precision, and the shared motion-vector entropy path emits fractional and high-precision symbols only when that precision permits them. Roslynk reports no compiler or scoped analyzer diagnostics, but runtime verification is pending. Additional retained reference pictures, compound prediction, and the remaining inter tools remain. - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. High-bit-depth paths widen before multiplying instead of applying the eight-bit coefficient clamp. Lossless blocks use the AV1 4x4 Walsh-Hadamard transform, exact lossless quantization and dequantization, four-by-four-only transform syntax, and non-skipped residual coding. Transform search and coefficient optimization remain. - [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma, filter-intra, palette, and intra-block copy, now perform live rate-distortion selection. Public quality mapping and effort tiers through exhaustive uniform luma mode/transform search are implemented. Effort nine adds exact recursive 8x8 and 16x16 partition rate-distortion selection, and effort ten extends it through 128x128; effort-dependent pruning and the remaining sequence searches remain. - [~] Frame effort now progressively expands the available current search: zero is DC-only, one adds every zero-angle spatial mode, two adds every legal directional adjustment, three refines the preliminary luma winner's transform type, four adds filter-intra and chroma-from-luma, and five adds adaptive palette and intra-block-copy analysis. Lower tiers do not signal unavailable sequence or frame tools, and tiers below five skip the whole-frame screen-content scan. Effort six enables `TX_MODE_SELECT` and compares the winning ordinary spatial or filter-intra luma mode as one 8x8 transform against four raster-ordered 4x4 transforms; each luma palette candidate owns that size comparison from effort six onward. Effort seven searches every legal 8x8 transform type inside every ordinary spatial candidate rather than refining only the preliminary winner. Effort eight also performs the 8x8-versus-four-4x4 comparison inside every ordinary spatial and filter-intra candidate, matching current libaom's per-candidate uniform-transform ownership. Effort nine additionally searches every legal partition at complete 8x8 and 16x16 nodes in current-libaom order, and effort ten extends that recursive search through 128x128. Prediction and residual construction run once per mode and are reused across its legal transform types. A 128x128 leaf evaluates four 64x64 luma transforms and as many as sixteen 32x32 transforms per 4:4:4 chroma plane, retaining sparse state at coefficient-area offsets. Residual emission follows AV1's bounded-region order, completing Y, U, and V for each 64x64 luma region before advancing. Every 4x4 transform searches all legal types with live coefficient contexts and reconstructed intra references. The search reuses the aligned block workspace, preserves only global improvements, and performs no per-block, per-partition, or per-transform rent. Non-skipped intra-block copy writes and costs the current-libaom unsplit variable-transform root; skipped intra-block copy emits no transform-partition symbol. Effort-dependent model and transform pruning remain. Decoder-visible production cases inspect the emitted restrictions and frame state and decode the produced streams, including real effort-nine streams selecting sub-8x8 and 8x16 rectangular blocks. The complete non-HEVC HEIF/AV1 namespace passes 9,077 of 9,077 through one foreground net11 Release VSTest run. The last independently built `aomdec`, from the then-current `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` snapshot, accepts the previously generated effort-eight and effort-ten payloads as well as the existing palette and intra-block-copy payloads. The affected encoder, partition, and workspace surface passes 139 of 139 through one foreground net11 Release VSTest run. The net11 Release build and Roslynk compiler and analyzer passes report zero errors. -- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these reusable workspace-backed selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma and chroma candidate scratch is partitioned from the encoder's single aligned reusable block workspace; transform-size search uses that owner for four retained 4x4 transform states, local coefficient contexts, and the compact trial reconstruction needed to preserve the best result. No candidate path rents a buffer per block or per transform. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. Luma palette selection now evaluates dominant-color and one-dimensional K-means candidates for every legal size, snaps near-cache colors with the reference threshold and tie order, removes duplicate snapped colors, extends boundary maps from active samples, and performs complete transform rate-distortion search. Ordinary DC and filter-intra candidates pay the palette-disabled symbol whenever screen-content syntax is enabled. The exact net11 Release rebuild reports 1,992 test-project warnings and zero errors, all 58 intra-superblock cases pass, all 8,935 AVIF cases pass, and all 230 HEIF cases pass. Remaining mode decision work includes transform-size coverage for filter-intra and palette, broader joint mode/transform refinement, and effort-dependent pruning. Non-empty intra blocks deliberately remain non-skipped, matching current libaom; later inter mode selection owns its distinct skip-transform RD decision. +- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. The corrected prepared reference edges retain the common-corner prefix and width-plus-height extent required by rectangular directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these reusable workspace-backed selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma and chroma candidate scratch is partitioned from the encoder's single aligned reusable block workspace; transform-size search uses that owner for four retained 4x4 transform states, local coefficient contexts, and the compact trial reconstruction needed to preserve the best result. No candidate path rents a buffer per block or per transform. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. Luma palette selection now evaluates dominant-color and one-dimensional K-means candidates for every legal size, snaps near-cache colors with the reference threshold and tie order, removes duplicate snapped colors, extends boundary maps from active samples, and performs complete transform rate-distortion search. Ordinary DC and filter-intra candidates pay the palette-disabled symbol whenever screen-content syntax is enabled. The exact net11 Release rebuild reports 1,992 test-project warnings and zero errors, all 58 intra-superblock cases pass, all 8,935 AVIF cases pass, and all 230 HEIF cases pass. Remaining mode decision work includes transform-size coverage for filter-intra and palette, broader joint mode/transform refinement, and effort-dependent pruning. Non-empty intra blocks deliberately remain non-skipped, matching current libaom; later inter mode selection owns its distinct skip-transform RD decision. - [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One picture owner packs segmentation, every tile's partition, luma, chroma, and transform edges, CDEF state, preceding quantizer, and encoded payload bounds into one clean byte allocation with typed non-owning views; together with the separately typed packed mode-information owner, the complete picture state uses two allocator rents rather than seven. Each encoded tile has independent neighbor and probability state while sharing the bounded output owner. Earlier aligned-length and balanced-return coverage exists; runtime allocation verification of the current multi-tile layout remains pending. - [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains. -- [~] The superblock decision and palette-map workspace uses one reusable 40.3 KiB ImageSharp allocator owner. Its aligned 8.3 KiB decision region contains 1,024 explicitly packed 8-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree; its remaining 32 KiB contains the fixed 128x128 luma and chroma palette maps. This improves on libaom's separate compressor-state allocation lifetimes without changing their layouts or introducing a frame-time rent. Palette colors have their own current-block value and are copied only to the picture edges that later blocks can reference, so enabling palette mode does not add 50 bytes to every final-block entry. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled quantizer, prediction, partition, and current-palette bytes cannot leak into the next decision pass. Roslynk reports zero compiler errors for the current one-owner refactor; runtime allocation verification remains pending. +- [~] The superblock decision and palette-map workspace uses one reusable 40.3 KiB ImageSharp allocator owner. Its aligned 8.3 KiB decision region contains 1,024 explicitly packed 8-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree; its remaining 32 KiB contains the fixed 128x128 luma and chroma palette maps. This combines storage held separately by libaom; exact lifetime and size reconciliation remains part of the fresh allocation audit, and fewer owners alone does not establish an improvement. Palette colors have their own current-block value and are copied only to the picture edges that later blocks can reference, so enabling palette mode does not add 50 bytes to every final-block entry. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled quantizer, prediction, partition, and current-palette bytes cannot leak into the next decision pass. Roslynk reports zero compiler errors for the current one-owner refactor; runtime allocation verification remains pending. - [~] Finalized transform coefficients and packed EOB/type state now use raster-ordered, per-superblock plane segments matching current libaom's coefficient-pool geometry. One ImageSharp allocator owner replaces libaom's separate coefficient, EOB, and entropy-context allocations while preserving the full 1024 luma and 256-per-chroma 4x4 state capacity of a 128x128 4:2:0 superblock. The fixed 8x8 DC-intra traversal populates the owner's quantized coefficient and state slices while updating the caller-owned reconstruction plane directly, and a real tile-writer integration check proves that both sides consume identical luma and chroma areas. Complete mode decision still remains. - [~] Tile partition writing now follows current libaom's recursive `write_modes_sb` preorder traversal and `update_ext_partition_context` edge updates directly. Bottom-edge blocks use the horizontal-alike partition CDF and right-edge blocks use the vertical-alike CDF; byte-exact regressions cover both paths after the previous calls were found reversed. Lossless chroma-from-luma availability now uses the subsampled plane block size shared with the decoder instead of the lossy 32x32 limit, preserving the correct UV-mode alphabet for each segment. The obsolete SVT-derived global geometry catalog and its unimplemented lookup are removed; transform geometry is derived in libaom's bounded 64x64 residual order, fixed intra transform-size symbols use the reference depth and neighbor contexts, and each derived transform size is persisted to the frame-owned mode information before the entropy snapshot and coefficient traversal consume it. Frame-edge and segmentation syntax use mode-information units, and 128x128 CDEF units use libaom's 0-to-3 indexing and first-block strength ownership. The focused transform-state regression passes 3 of 3 direct net11 VSTest cases in Release. Writer, entropy, and OBU coverage passes 1,957 of 1,957 direct net11 VSTest cases in Release, with 20 of 20 focused encoder and decoder chroma-from-luma cases. Partition and mode analysis still need to populate these retained decisions; variable inter-transform syntax remains part of later inter-frame support. - [ ] Implement legal deblocking, CDEF, restoration, super-resolution, and film-grain signaling decisions. @@ -1016,7 +1082,7 @@ Encoder verification contract: - [~] Paired chroma palette clustering now preserves current libaom's squared two-component distance, first-centroid tie order, independently rounded U/V means, paired deterministic empty-cluster replacement, preceding-state retention on increased distortion, and 50-iteration limit. The source planes remain separate, with Vector512, Vector256, Vector128, then scalar dispatch through ImageSharp's shared vector-count helpers. An end-to-end improvement over the native implementation has not been established. Three independent tests cover exact paired convergence, midpoint initialization, 12-bit distance and index parity, untouched destination bounds, and every intrinsic tier. The exact Release test-project build reports 1,992 baseline warnings and zero errors; the focused three-case set, complete 8,934-case AVIF set, and complete 230-case HEIF set pass direct foreground net11 Release VSTest. Roslynk reports zero compiler errors and no touched-file analyzer warnings. Candidate integration and production activation remain in the open chroma-palette checkpoint. - [~] Live paired chroma palette selection now follows current libaom's complete 2-through-8 color-size search, U-plane neighbor-cache snapping, stable U-ordered color pairs, shared U/V index map, implicit DCT-DCT transform, and strict rate-distortion winner replacement. It omits the reference's early header-cost pruning, keeps planar U/V source data separate, and reuses the existing prediction, residual, transform, quantization, and reconstruction operators. Omitting pruning is an unresolved decision-policy deviation, not an established improvement. The production tile regression proves both palette-mode probability branches, exact paired colors and indices, coefficient-free reconstruction, and nonempty syntax. The complete 58-case intra-superblock set, 8,935-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains the next checkpoint. - [x] Production screen-content activation now matches current libaom's default good-quality detector: it scans only complete 16x16 luma blocks, normalizes palette samples to eight bits, admits 2-through-4-color blocks, and uses the reference's strict greater-than-ten-percent frame-area threshold. The same pass accumulates centered sums and squared sums at native precision, applies libaom's exact 10-bit and 12-bit variance rounding, and enables intra-block copy only when positive rounded per-pixel variance exceeds its strict one-twelfth frame-area threshold. A 256-bit stack bitset and fifth-color early exit replace libaom's larger per-block histogram without a second source scan or allocation. The adaptive sequence flag remains enabled and both frame flags are fixed before picture-state allocation. Focused regressions prove strict palette-threshold equality, high-bit-depth normalization, the exact variance rounding boundary, five-color rejection, emitted frame-header activation, actual production IBC selection, and production decode. The exact Release test-project build reports 1,992 baseline warnings and zero errors; all 9,242 non-HEVC HEIF/AV1 cases pass direct foreground net11 Release VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 31 payloads regenerated by the current test tree, including an actual IBC-coded 328x16 stream with decoded MD5 `677435e5af39c930af1178f91c34af6a`. Roslynk reports zero compiler errors and no touched-file analyzer warnings. -- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Paths above the speed-zero screen-content threshold continue through libaom's 256-pixel, one-pixel-step exhaustive mesh. Four adjacent byte or high-bit-depth candidates share each SIMD source load, strict row-major tie ordering is retained, and the final legal tail column remains searchable where libaom's current four-wide remainder loop omits it. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. Production mode decision now derives the same spatial displacement reference used by the writer, deduplicates hash and full-pixel finalists in search order, and evaluates every surviving vector through complete luma and chroma transform RD. This intentionally improves on libaom's preliminary-error pruning by permitting a hash and pixel finalist from the same search region to compete using their final syntax and reconstruction costs. Prediction is prepared once per plane and vector, including integer or half-sample chroma phase, then reused across every legal inter transform without an allocator rent or frame copy. The joint comparison includes the live intra-block-copy flag, discounted displacement rate, skip flag, coefficient syntax, and normalized Y/U/V distortion; an empty transform alternative can win only when its complete skip cost is strictly lower, while conventional intra and earlier vectors retain tie precedence. Winning reconstruction, coefficients, transform state, DC modes, cleared palette/filter/CfL state, and displacement are copied once into the existing retained stores. Production regressions force the path at 8 and 12 bits and force 4:2:0 horizontal half-sample chroma with an unaligned reference. The former bulk local workspace occupied 2.75 KiB for byte samples or 3.375 KiB for high-bit-depth samples. Prediction, candidate, winning reconstruction, residual, and coefficient scratch now occupy one naturally aligned 3.125 KiB extension of the existing frame-reused block-workspace owner, matching libaom's reusable macroblock-scratch lifetime without adding an allocation; only the 128-byte reference, weight, and finalist arrays remain on the stack. The net11 Release solution build reports zero errors; all 2,082 focused transform, entropy, intra-block-copy, intra-superblock, and frame-encoder cases and all 9,242 non-HEVC HEIF/AV1 cases pass through direct foreground VSTest, with tiered compilation disabled only for the full allocation-sensitive suite. Adaptive production activation is complete, and the emitted frame flag remains authoritative for the complete frame rather than being invalidated after tile coding. +- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Paths above the speed-zero screen-content threshold continue through libaom's 256-pixel, one-pixel-step exhaustive mesh. Four adjacent byte or high-bit-depth candidates share each SIMD source load, strict row-major tie ordering is retained, and the final legal tail column remains searchable where libaom's current four-wide remainder loop omits it. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. Production mode decision now derives the same spatial displacement reference used by the writer, deduplicates hash and full-pixel finalists in search order, and evaluates every surviving vector through complete luma and chroma transform RD. This differs from libaom's preliminary-error pruning by permitting a hash and pixel finalist from the same search region to compete using final syntax and reconstruction costs. It is an unresolved controller deviation, with no established quality or performance improvement. Prediction is prepared once per plane and vector, including integer or half-sample chroma phase, then reused across every legal inter transform without an allocator rent or frame copy. The joint comparison includes the live intra-block-copy flag, discounted displacement rate, skip flag, coefficient syntax, and normalized Y/U/V distortion; an empty transform alternative can win only when its complete skip cost is strictly lower, while conventional intra and earlier vectors retain tie precedence. Winning reconstruction, coefficients, transform state, DC modes, cleared palette/filter/CfL state, and displacement are copied once into the existing retained stores. Production regressions force the path at 8 and 12 bits and force 4:2:0 horizontal half-sample chroma with an unaligned reference. The former bulk local workspace occupied 2.75 KiB for byte samples or 3.375 KiB for high-bit-depth samples. Prediction, candidate, winning reconstruction, residual, and coefficient scratch now occupy one naturally aligned 3.125 KiB extension of the existing frame-reused block-workspace owner, matching libaom's reusable macroblock-scratch lifetime without adding an allocation; only the 128-byte reference, weight, and finalist arrays remain on the stack. The net11 Release solution build reports zero errors; all 2,082 focused transform, entropy, intra-block-copy, intra-superblock, and frame-encoder cases and all 9,242 non-HEVC HEIF/AV1 cases pass through direct foreground VSTest, with tiered compilation disabled only for the full allocation-sensitive suite. Adaptive production activation is complete, and the emitted frame flag remains authoritative for the complete frame rather than being invalidated after tile coding. - [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips. - [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips. - [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips. diff --git a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs index 52d2b50067..feeb409917 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs @@ -268,11 +268,21 @@ internal sealed class Av1SymbolEncoder : IDisposable // Transform dimensions are bounded by the AV1 coefficient-coding rules, so the complete entropy scratch // is known with the tile output capacity and remains valid for every transform in every sequence sample. this.levels = new Av1LevelBuffer(configuration); - this.coefficientContexts = - configuration.MemoryAllocator.Allocate(MaximumCoefficientContextCount); + try + { + this.coefficientContexts = + configuration.MemoryAllocator.Allocate(MaximumCoefficientContextCount); - this.writer = new(configuration, bufferLength, updateCdf); - this.baseQIndex = qIndex; + this.writer = new(configuration, bufferLength, updateCdf); + this.baseQIndex = qIndex; + } + catch + { + // The level buffer is already owned here; a later allocation failure cannot be unwound by the caller. + this.coefficientContexts?.Dispose(); + this.levels.Dispose(); + throw; + } } /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs index 6cc6151393..19cc1f2611 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs @@ -1519,44 +1519,54 @@ internal static class Av1FrameEncoder encodeAlpha, usesHighBitDepth); - bool allocateScreenContentState = effort >= 5; - bool allocateIntraBlockCopySearch = - allocateScreenContentState && - !this.FrameHeader.CodedLossless; - - // Sequence geometry and maximum tool capacity are fixed before the first sample. Reusing this owner - // avoids renting the complete mode grid and optional screen-content index for every frame. - this.PictureBuffer = new Av1EncoderPictureBuffer( - configuration, - this.SequenceHeader, - this.FrameHeader, - width, - height, - disallow4x4AllFrames: !this.FrameHeader.CodedLossless && effort < 9, - allocateScreenContentState: allocateScreenContentState, - allocateMotionVectorState: true, - allocateIntraBlockCopySearch: allocateIntraBlockCopySearch); + try + { + bool allocateScreenContentState = effort >= 5; + bool allocateIntraBlockCopySearch = + allocateScreenContentState && + !this.FrameHeader.CodedLossless; + + // Sequence geometry and maximum tool capacity are fixed before the first sample. Reusing this owner + // avoids renting the complete mode grid and optional screen-content index for every frame. + this.PictureBuffer = new Av1EncoderPictureBuffer( + configuration, + this.SequenceHeader, + this.FrameHeader, + width, + height, + disallow4x4AllFrames: !this.FrameHeader.CodedLossless && effort < 9, + allocateScreenContentState: allocateScreenContentState, + allocateMotionVectorState: true, + allocateIntraBlockCopySearch: allocateIntraBlockCopySearch); - this.Coefficients = new Av1EncoderCoefficientBuffer( - configuration, - this.SequenceHeader, - width, - height); + this.Coefficients = new Av1EncoderCoefficientBuffer( + configuration, + this.SequenceHeader, + width, + height); - this.SuperblockWorkspace = new Av1EncoderSuperblockWorkspace(configuration); + this.SuperblockWorkspace = new Av1EncoderSuperblockWorkspace(configuration); - this.TileWorkspace = new Av1EncoderTileWorkspace(this.FrameHeader, this.SuperblockWorkspace); - this.BlockWorkspace = new Av1EncoderBlockWorkspace(configuration); + this.TileWorkspace = new Av1EncoderTileWorkspace(this.FrameHeader, this.SuperblockWorkspace); + this.BlockWorkspace = new Av1EncoderBlockWorkspace(configuration); - // Tile probabilities adapt within a sample, while error-resilient frame headers prohibit carrying - // those updates into the next sample. The retained encoder is therefore reset before each frame. - this.SymbolEncoder = new Av1SymbolEncoder( - configuration, - this.TileBufferLength, - qIndex, - updateCdf: true); + // Tile probabilities adapt within a sample, while error-resilient frame headers prohibit carrying + // those updates into the next sample. The retained encoder is therefore reset before each frame. + this.SymbolEncoder = new Av1SymbolEncoder( + configuration, + this.TileBufferLength, + qIndex, + updateCdf: true); - this.ObuWriter = new ObuWriter(configuration); + this.ObuWriter = new ObuWriter(configuration); + } + catch + { + // The caller receives no encoder when construction fails. Release only completed common owners; + // derived frame construction has not started and must not be reached through virtual disposal. + this.DisposeResources(); + throw; + } } /// @@ -1637,13 +1647,7 @@ internal static class Av1FrameEncoder public void Dispose() { this.DisposeFrames(); - this.ConversionWorkspace.Dispose(); - this.PictureBuffer.Dispose(); - this.Coefficients.Dispose(); - this.SuperblockWorkspace.Dispose(); - this.BlockWorkspace.Dispose(); - this.ObuWriter.Dispose(); - this.SymbolEncoder.Dispose(); + this.DisposeResources(); } /// @@ -1657,6 +1661,19 @@ internal static class Av1FrameEncoder ObuFrameType frameType, bool writeSequenceHeader) where TPixel : unmanaged, IPixel; + + private void DisposeResources() + { + // Construction can stop between any two allocations. Successful instances have every owner; + // failed constructors retain only the prefix completed before the allocator rejected a request. + this.ObuWriter?.Dispose(); + this.SymbolEncoder?.Dispose(); + this.BlockWorkspace?.Dispose(); + this.SuperblockWorkspace?.Dispose(); + this.Coefficients?.Dispose(); + this.PictureBuffer?.Dispose(); + this.ConversionWorkspace.Dispose(); + } } private sealed class ByteSequenceEncoder : SequenceEncoder @@ -1683,40 +1700,50 @@ internal static class Av1FrameEncoder encodeAlpha, usesHighBitDepth: false) { - Av1ColorFormat colorFormat = colorConfig.GetColorFormat(); - this.source = new( - configuration, - width, - height, - ByteSampleBitDepth, - colorFormat, - CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + try + { + Av1ColorFormat colorFormat = colorConfig.GetColorFormat(); + this.source = new( + configuration, + width, + height, + ByteSampleBitDepth, + colorFormat, + CenteredChromaSamplePosition, + CenteredChromaSamplePosition); - this.reference = new( - configuration, - width, - height, - ByteSampleBitDepth, - colorFormat, - CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + this.reference = new( + configuration, + width, + height, + ByteSampleBitDepth, + colorFormat, + CenteredChromaSamplePosition, + CenteredChromaSamplePosition); - this.reconstruction = new( - configuration, - width, - height, - ByteSampleBitDepth, - colorFormat, - CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + this.reconstruction = new( + configuration, + width, + height, + ByteSampleBitDepth, + colorFormat, + CenteredChromaSamplePosition, + CenteredChromaSamplePosition); + } + catch + { + // The common state already exists, and any preceding frame allocations also need returning. + this.Dispose(); + throw; + } } protected override void DisposeFrames() { - this.source.Dispose(); - this.reference.Dispose(); - this.reconstruction.Dispose(); + // A derived constructor can fail before all three frame owners exist. + this.reconstruction?.Dispose(); + this.reference?.Dispose(); + this.source?.Dispose(); } protected override void EncodeFrame( @@ -1793,41 +1820,51 @@ internal static class Av1FrameEncoder encodeAlpha, usesHighBitDepth: true) { - int bitDepth = colorConfig.BitDepth.GetBitCount(); - Av1ColorFormat colorFormat = colorConfig.GetColorFormat(); - this.source = new( - configuration, - width, - height, - bitDepth, - colorFormat, - CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + try + { + int bitDepth = colorConfig.BitDepth.GetBitCount(); + Av1ColorFormat colorFormat = colorConfig.GetColorFormat(); + this.source = new( + configuration, + width, + height, + bitDepth, + colorFormat, + CenteredChromaSamplePosition, + CenteredChromaSamplePosition); - this.reference = new( - configuration, - width, - height, - bitDepth, - colorFormat, - CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + this.reference = new( + configuration, + width, + height, + bitDepth, + colorFormat, + CenteredChromaSamplePosition, + CenteredChromaSamplePosition); - this.reconstruction = new( - configuration, - width, - height, - bitDepth, - colorFormat, - CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + this.reconstruction = new( + configuration, + width, + height, + bitDepth, + colorFormat, + CenteredChromaSamplePosition, + CenteredChromaSamplePosition); + } + catch + { + // The common state already exists, and any preceding frame allocations also need returning. + this.Dispose(); + throw; + } } protected override void DisposeFrames() { - this.source.Dispose(); - this.reference.Dispose(); - this.reconstruction.Dispose(); + // A derived constructor can fail before all three frame owners exist. + this.reconstruction?.Dispose(); + this.reference?.Dispose(); + this.source?.Dispose(); } protected override void EncodeFrame( diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs index 3618efcbdc..cb70eb6094 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs @@ -90,235 +90,246 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable height, disallow4x4AllFrames); - int alignedModeInfoRowCount = Av1Math.AlignPowerOf2(this.modeInfo.ModeInfoRowCount, ContextAlignmentLog2); - int lumaLeftLength = alignedModeInfoRowCount; - int lumaTopLength = this.modeInfo.ModeInfoStride; - ObuColorConfig colorConfig = sequenceHeader.ColorConfig; - int chromaLeftLength = colorConfig.IsMonochrome - ? 0 - : lumaLeftLength >> (colorConfig.SubSamplingY ? 1 : 0); - - int chromaTopLength = colorConfig.IsMonochrome - ? 0 - : lumaTopLength >> (colorConfig.SubSamplingX ? 1 : 0); - - int tileCount = frameHeader.TilesInfo.TileColumnCount * frameHeader.TilesInfo.TileRowCount; - int lumaContextLength = checked(lumaLeftLength + lumaTopLength); - int chromaContextLength = checked(chromaLeftLength + chromaTopLength); - int byteContextLengthPerTile = checked((2 * lumaContextLength) + (2 * chromaContextLength)); - int partitionContextLength = checked(tileCount * lumaContextLength); - int segmentationLength = checked(this.modeInfo.ModeInfoColumnCount * this.modeInfo.ModeInfoRowCount); - int partitionContextSize = Unsafe.SizeOf(); - int partitionStorageOffset = checked( - ((segmentationLength + partitionContextSize - 1) / partitionContextSize) * partitionContextSize); - - int partitionStorageLength = checked( - partitionContextLength * Unsafe.SizeOf()); - - int byteContextStorageOffset = checked(partitionStorageOffset + partitionStorageLength); - int byteContextStorageLength = checked(tileCount * byteContextLengthPerTile); - int byteContextStorageEnd = checked(byteContextStorageOffset + byteContextStorageLength); - int paletteLeftLength = alignedModeInfoRowCount; - int paletteTopLength = this.modeInfo.ModeInfoStride; - int paletteContextLength = checked(paletteLeftLength + paletteTopLength); - int paletteStorageOffset = allocateScreenContentState - ? Av1Math.AlignPowerOf2(byteContextStorageEnd, 1) - : byteContextStorageEnd; - - int paletteStorageLength = allocateScreenContentState - ? checked(tileCount * paletteContextLength * Unsafe.SizeOf()) - : 0; - - int paletteStorageEnd = checked(paletteStorageOffset + paletteStorageLength); - int displacementVectorLength = allocateMotionVectorState ? this.modeInfo.Allocation.Length : 0; - int displacementVectorStorageOffset = allocateMotionVectorState - ? Av1Math.AlignPowerOf2(paletteStorageEnd, 1) - : paletteStorageEnd; - - int displacementVectorStorageLength = checked( - displacementVectorLength * Unsafe.SizeOf()); - - int displacementVectorStorageEnd = checked(displacementVectorStorageOffset + displacementVectorStorageLength); - int intraBlockCopySearchStorageOffset = allocateIntraBlockCopySearch - ? Av1Math.AlignPowerOf2(displacementVectorStorageEnd, 2) - : displacementVectorStorageEnd; - - int intraBlockCopySearchStorageLength = allocateIntraBlockCopySearch - ? Av1IntraBlockCopySearchIndex.GetStorageLength(width, height) - : 0; - - int intraBlockCopySearchStorageEnd = checked( - intraBlockCopySearchStorageOffset + intraBlockCopySearchStorageLength); - - int tileStateStorageOffset = Av1Math.AlignPowerOf2(intraBlockCopySearchStorageEnd, 2); - int cdefPresetLength = tileCount * Av1Constants.CdefUnitsPerSuperblock; - int tileStateLength = cdefPresetLength + (3 * tileCount); - int tileStateStorageLength = tileStateLength * sizeof(int); - int stateStorageLength = checked(tileStateStorageOffset + tileStateStorageLength); - - // Segmentation and every tile edge share one clean picture lifetime. The partition region begins at its - // native alignment. CDEF, quantizer, and encoded-tile bounds occupy one aligned trailing integer region - // instead of allocating separate managed arrays for every picture. - this.stateStorage = configuration.MemoryAllocator.Allocate( - stateStorageLength, - AllocationOptions.Clean); - - this.stateMemory = this.stateStorage.Memory[..stateStorageLength]; - Memory stateStorage = this.stateMemory; - this.partitionContextMemory = new ByteMemoryManager( - stateStorage.Slice(partitionStorageOffset, partitionStorageLength)); - - Memory partitionStorage = this.partitionContextMemory.Memory; - Memory byteContextStorage = stateStorage.Slice(byteContextStorageOffset, byteContextStorageLength); - this.partitionContexts = new Av1NeighborArrayUnit[tileCount]; - this.lumaCoefficientContexts = new Av1NeighborArrayUnit[tileCount]; - this.blueCoefficientContexts = new Av1NeighborArrayUnit[tileCount]; - this.redCoefficientContexts = new Av1NeighborArrayUnit[tileCount]; - this.transformContexts = new Av1NeighborArrayUnit[tileCount]; - Memory paletteStorage = Memory.Empty; - if (allocateScreenContentState) + try { - // Palette entries contain 16-bit colors, so their packed typed region begins at an even byte offset. - ByteMemoryManager paletteMemory = new( - stateStorage.Slice(paletteStorageOffset, paletteStorageLength)); - - paletteStorage = paletteMemory.Memory; - this.paletteContexts = new Av1NeighborArrayUnit[tileCount]; - } - else - { - this.paletteContexts = []; - } - - Memory displacementVectors = Memory.Empty; - if (allocateMotionVectorState) - { - // Each component lies strictly inside plus or minus 16384. Two signed 16-bit fields preserve both - // inter and intra-block-copy vectors without expanding every compact mode-information entry. - ByteMemoryManager displacementVectorMemory = new( - stateStorage.Slice(displacementVectorStorageOffset, displacementVectorStorageLength)); + int alignedModeInfoRowCount = Av1Math.AlignPowerOf2(this.modeInfo.ModeInfoRowCount, ContextAlignmentLog2); + int lumaLeftLength = alignedModeInfoRowCount; + int lumaTopLength = this.modeInfo.ModeInfoStride; + ObuColorConfig colorConfig = sequenceHeader.ColorConfig; + int chromaLeftLength = colorConfig.IsMonochrome + ? 0 + : lumaLeftLength >> (colorConfig.SubSamplingY ? 1 : 0); + + int chromaTopLength = colorConfig.IsMonochrome + ? 0 + : lumaTopLength >> (colorConfig.SubSamplingX ? 1 : 0); + + int tileCount = frameHeader.TilesInfo.TileColumnCount * frameHeader.TilesInfo.TileRowCount; + int lumaContextLength = checked(lumaLeftLength + lumaTopLength); + int chromaContextLength = checked(chromaLeftLength + chromaTopLength); + int byteContextLengthPerTile = checked((2 * lumaContextLength) + (2 * chromaContextLength)); + int partitionContextLength = checked(tileCount * lumaContextLength); + int segmentationLength = checked(this.modeInfo.ModeInfoColumnCount * this.modeInfo.ModeInfoRowCount); + int partitionContextSize = Unsafe.SizeOf(); + int partitionStorageOffset = checked( + ((segmentationLength + partitionContextSize - 1) / partitionContextSize) * partitionContextSize); + + int partitionStorageLength = checked( + partitionContextLength * Unsafe.SizeOf()); + + int byteContextStorageOffset = checked(partitionStorageOffset + partitionStorageLength); + int byteContextStorageLength = checked(tileCount * byteContextLengthPerTile); + int byteContextStorageEnd = checked(byteContextStorageOffset + byteContextStorageLength); + int paletteLeftLength = alignedModeInfoRowCount; + int paletteTopLength = this.modeInfo.ModeInfoStride; + int paletteContextLength = checked(paletteLeftLength + paletteTopLength); + int paletteStorageOffset = allocateScreenContentState + ? Av1Math.AlignPowerOf2(byteContextStorageEnd, 1) + : byteContextStorageEnd; + + int paletteStorageLength = allocateScreenContentState + ? checked(tileCount * paletteContextLength * Unsafe.SizeOf()) + : 0; + + int paletteStorageEnd = checked(paletteStorageOffset + paletteStorageLength); + int displacementVectorLength = allocateMotionVectorState ? this.modeInfo.Allocation.Length : 0; + int displacementVectorStorageOffset = allocateMotionVectorState + ? Av1Math.AlignPowerOf2(paletteStorageEnd, 1) + : paletteStorageEnd; + + int displacementVectorStorageLength = checked( + displacementVectorLength * Unsafe.SizeOf()); + + int displacementVectorStorageEnd = checked(displacementVectorStorageOffset + displacementVectorStorageLength); + int intraBlockCopySearchStorageOffset = allocateIntraBlockCopySearch + ? Av1Math.AlignPowerOf2(displacementVectorStorageEnd, 2) + : displacementVectorStorageEnd; + + int intraBlockCopySearchStorageLength = allocateIntraBlockCopySearch + ? Av1IntraBlockCopySearchIndex.GetStorageLength(width, height) + : 0; + + int intraBlockCopySearchStorageEnd = checked( + intraBlockCopySearchStorageOffset + intraBlockCopySearchStorageLength); + + int tileStateStorageOffset = Av1Math.AlignPowerOf2(intraBlockCopySearchStorageEnd, 2); + int cdefPresetLength = tileCount * Av1Constants.CdefUnitsPerSuperblock; + int tileStateLength = cdefPresetLength + (3 * tileCount); + int tileStateStorageLength = tileStateLength * sizeof(int); + int stateStorageLength = checked(tileStateStorageOffset + tileStateStorageLength); + + // Segmentation and every tile edge share one clean picture lifetime. The partition region begins at its + // native alignment. CDEF, quantizer, and encoded-tile bounds occupy one aligned trailing integer region + // instead of allocating separate managed arrays for every picture. + this.stateStorage = configuration.MemoryAllocator.Allocate( + stateStorageLength, + AllocationOptions.Clean); + + this.stateMemory = this.stateStorage.Memory[..stateStorageLength]; + Memory stateStorage = this.stateMemory; + this.partitionContextMemory = new ByteMemoryManager( + stateStorage.Slice(partitionStorageOffset, partitionStorageLength)); + + Memory partitionStorage = this.partitionContextMemory.Memory; + Memory byteContextStorage = stateStorage.Slice(byteContextStorageOffset, byteContextStorageLength); + this.partitionContexts = new Av1NeighborArrayUnit[tileCount]; + this.lumaCoefficientContexts = new Av1NeighborArrayUnit[tileCount]; + this.blueCoefficientContexts = new Av1NeighborArrayUnit[tileCount]; + this.redCoefficientContexts = new Av1NeighborArrayUnit[tileCount]; + this.transformContexts = new Av1NeighborArrayUnit[tileCount]; + Memory paletteStorage = Memory.Empty; + if (allocateScreenContentState) + { + // Palette entries contain 16-bit colors, so their packed typed region begins at an even byte offset. + ByteMemoryManager paletteMemory = new( + stateStorage.Slice(paletteStorageOffset, paletteStorageLength)); - displacementVectors = displacementVectorMemory.Memory; - } + paletteStorage = paletteMemory.Memory; + this.paletteContexts = new Av1NeighborArrayUnit[tileCount]; + } + else + { + this.paletteContexts = []; + } - Av1IntraBlockCopySearchIndex intraBlockCopySearch = default; - if (allocateIntraBlockCopySearch) - { - // The search index casts its packed workspace to 32-bit links, so its non-owning region begins at - // a four-byte boundary inside the existing picture-state rent. - intraBlockCopySearch = new Av1IntraBlockCopySearchIndex( - stateStorage.Slice(intraBlockCopySearchStorageOffset, intraBlockCopySearchStorageLength), - width, - height); - } + Memory displacementVectors = Memory.Empty; + if (allocateMotionVectorState) + { + // Each component lies strictly inside plus or minus 16384. Two signed 16-bit fields preserve both + // inter and intra-block-copy vectors without expanding every compact mode-information entry. + ByteMemoryManager displacementVectorMemory = new( + stateStorage.Slice(displacementVectorStorageOffset, displacementVectorStorageLength)); - ByteMemoryManager tileStateMemory = new( - stateStorage.Slice(tileStateStorageOffset, tileStateStorageLength)); + displacementVectors = displacementVectorMemory.Memory; + } - Memory tileState = tileStateMemory.Memory; - Memory cdefPreset = tileState[..cdefPresetLength]; - Memory previousQIndex = tileState.Slice(cdefPresetLength, tileCount); - Memory tileDataOffsets = tileState.Slice(cdefPresetLength + tileCount, tileCount); - Memory tileDataLengths = tileState.Slice(cdefPresetLength + (2 * tileCount), tileCount); - cdefPreset.Span.Fill(-1); - for (int tileIndex = 0; tileIndex < tileCount; tileIndex++) - { - this.partitionContexts[tileIndex] = new Av1NeighborArrayUnit( - partitionStorage.Slice(tileIndex * lumaContextLength, lumaContextLength), - lumaLeftLength, - lumaTopLength) + Av1IntraBlockCopySearchIndex intraBlockCopySearch = default; + if (allocateIntraBlockCopySearch) { - GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 - }; + // The search index casts its packed workspace to 32-bit links, so its non-owning region begins at + // a four-byte boundary inside the existing picture-state rent. + intraBlockCopySearch = new Av1IntraBlockCopySearchIndex( + stateStorage.Slice(intraBlockCopySearchStorageOffset, intraBlockCopySearchStorageLength), + width, + height); + } - int byteContextOffset = tileIndex * byteContextLengthPerTile; - this.lumaCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit( - byteContextStorage.Slice(byteContextOffset, lumaContextLength), - lumaLeftLength, - lumaTopLength) - { - GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 - }; + ByteMemoryManager tileStateMemory = new( + stateStorage.Slice(tileStateStorageOffset, tileStateStorageLength)); - byteContextOffset += lumaContextLength; - this.blueCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit( - byteContextStorage.Slice(byteContextOffset, chromaContextLength), - chromaLeftLength, - chromaTopLength) + Memory tileState = tileStateMemory.Memory; + Memory cdefPreset = tileState[..cdefPresetLength]; + Memory previousQIndex = tileState.Slice(cdefPresetLength, tileCount); + Memory tileDataOffsets = tileState.Slice(cdefPresetLength + tileCount, tileCount); + Memory tileDataLengths = tileState.Slice(cdefPresetLength + (2 * tileCount), tileCount); + cdefPreset.Span.Fill(-1); + for (int tileIndex = 0; tileIndex < tileCount; tileIndex++) { - GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 - }; + this.partitionContexts[tileIndex] = new Av1NeighborArrayUnit( + partitionStorage.Slice(tileIndex * lumaContextLength, lumaContextLength), + lumaLeftLength, + lumaTopLength) + { + GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 + }; - byteContextOffset += chromaContextLength; - this.redCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit( - byteContextStorage.Slice(byteContextOffset, chromaContextLength), - chromaLeftLength, - chromaTopLength) - { - GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 - }; + int byteContextOffset = tileIndex * byteContextLengthPerTile; + this.lumaCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit( + byteContextStorage.Slice(byteContextOffset, lumaContextLength), + lumaLeftLength, + lumaTopLength) + { + GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 + }; - byteContextOffset += chromaContextLength; - this.transformContexts[tileIndex] = new Av1NeighborArrayUnit( - byteContextStorage.Slice(byteContextOffset, lumaContextLength), - lumaLeftLength, - lumaTopLength) - { - GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 - }; + byteContextOffset += lumaContextLength; + this.blueCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit( + byteContextStorage.Slice(byteContextOffset, chromaContextLength), + chromaLeftLength, + chromaTopLength) + { + GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 + }; - // Variable-transform contexts consult both edges without separate availability flags. The largest - // transform makes an unavailable edge compare as unsplit until a coded neighbor publishes its size. - this.transformContexts[tileIndex].Left.Fill((byte)Av1Constants.MaxTransformSize); - this.transformContexts[tileIndex].Top.Fill((byte)Av1Constants.MaxTransformSize); + byteContextOffset += chromaContextLength; + this.redCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit( + byteContextStorage.Slice(byteContextOffset, chromaContextLength), + chromaLeftLength, + chromaTopLength) + { + GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 + }; - if (allocateScreenContentState) - { - this.paletteContexts[tileIndex] = new Av1NeighborArrayUnit( - paletteStorage.Slice(tileIndex * paletteContextLength, paletteContextLength), - paletteLeftLength, - paletteTopLength) + byteContextOffset += chromaContextLength; + this.transformContexts[tileIndex] = new Av1NeighborArrayUnit( + byteContextStorage.Slice(byteContextOffset, lumaContextLength), + lumaLeftLength, + lumaTopLength) { GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 }; - } - previousQIndex.Span[tileIndex] = frameHeader.QuantizationParameters.BaseQIndex; - } + // Variable-transform contexts consult both edges without separate availability flags. The largest + // transform makes an unavailable edge compare as unsplit until a coded neighbor publishes its size. + this.transformContexts[tileIndex].Left.Fill((byte)Av1Constants.MaxTransformSize); + this.transformContexts[tileIndex].Top.Fill((byte)Av1Constants.MaxTransformSize); - this.Picture = new Av1PictureControlSet - { - PartitionContexts = this.partitionContexts, - LuminanceDcSignLevelCoefficientNeighbors = this.lumaCoefficientContexts, - CbDcSignLevelCoefficientNeighbors = this.blueCoefficientContexts, - CrDcSignLevelCoefficientNeighbors = this.redCoefficientContexts, - TransformFunctionContexts = this.transformContexts, - PaletteContexts = this.paletteContexts, - Sequence = new Av1SequenceControlSet { SequenceHeader = sequenceHeader }, - Parent = new Av1PictureParentControlSet + if (allocateScreenContentState) + { + this.paletteContexts[tileIndex] = new Av1NeighborArrayUnit( + paletteStorage.Slice(tileIndex * paletteContextLength, paletteContextLength), + paletteLeftLength, + paletteTopLength) + { + GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 + }; + } + + previousQIndex.Span[tileIndex] = frameHeader.QuantizationParameters.BaseQIndex; + } + + this.Picture = new Av1PictureControlSet { - Common = new Av1EncoderCommon + PartitionContexts = this.partitionContexts, + LuminanceDcSignLevelCoefficientNeighbors = this.lumaCoefficientContexts, + CbDcSignLevelCoefficientNeighbors = this.blueCoefficientContexts, + CrDcSignLevelCoefficientNeighbors = this.redCoefficientContexts, + TransformFunctionContexts = this.transformContexts, + PaletteContexts = this.paletteContexts, + Sequence = new Av1SequenceControlSet { SequenceHeader = sequenceHeader }, + Parent = new Av1PictureParentControlSet { - ModeInfoColumnCount = this.modeInfo.ModeInfoColumnCount, - ModeInfoRowCount = this.modeInfo.ModeInfoRowCount, - ModeInfoStride = this.modeInfo.ModeInfoStride, - FrameSize = frameHeader.FrameSize, - TilesInfo = frameHeader.TilesInfo + Common = new Av1EncoderCommon + { + ModeInfoColumnCount = this.modeInfo.ModeInfoColumnCount, + ModeInfoRowCount = this.modeInfo.ModeInfoRowCount, + ModeInfoStride = this.modeInfo.ModeInfoStride, + FrameSize = frameHeader.FrameSize, + TilesInfo = frameHeader.TilesInfo + }, + FrameHeader = frameHeader, + PreviousQIndex = previousQIndex }, - FrameHeader = frameHeader, - PreviousQIndex = previousQIndex - }, - SegmentationNeighborMap = stateStorage[..segmentationLength], - ModeInfoGrid = this.modeInfo.Grid, - ModeInfoAllocation = this.modeInfo.Allocation, - DisplacementVectors = displacementVectors, - IntraBlockCopySearch = intraBlockCopySearch, - ModeInfoStride = this.modeInfo.ModeInfoStride, - Disallow4x4AllFrames = this.modeInfo.Disallow4x4AllFrames, - CdefPreset = cdefPreset, - TileDataOffsets = tileDataOffsets, - TileDataLengths = tileDataLengths - }; + SegmentationNeighborMap = stateStorage[..segmentationLength], + ModeInfoGrid = this.modeInfo.Grid, + ModeInfoAllocation = this.modeInfo.Allocation, + DisplacementVectors = displacementVectors, + IntraBlockCopySearch = intraBlockCopySearch, + ModeInfoStride = this.modeInfo.ModeInfoStride, + Disallow4x4AllFrames = this.modeInfo.Disallow4x4AllFrames, + CdefPreset = cdefPreset, + TileDataOffsets = tileDataOffsets, + TileDataLengths = tileDataLengths + }; + } + catch + { + // The context objects only borrow these two owners. A failed constructor must release the + // completed allocations itself because the enclosing sequence never receives this picture. + this.stateStorage?.Dispose(); + this.modeInfo.Dispose(); + throw; + } } /// diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs index ee667719f3..179b7181c9 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs @@ -593,6 +593,55 @@ public class Av1EncoderFrameTests Assert.Equal(second.Size, decodedSecond.Size); } + [Theory] + [InlineData(false, EightBit)] + [InlineData(false, TenBit)] + [InlineData(false, TwelveBit)] + [InlineData(true, EightBit)] + [InlineData(true, TenBit)] + [InlineData(true, TwelveBit)] + public void SequenceEncoderConstructionFailureReturnsEveryAllocation(bool encodeAlpha, int bitDepthValue) + { + ObuColorConfig colorConfig = CreateColorConfig( + (Av1BitDepth)bitDepthValue, + encodeAlpha ? Av1ColorFormat.Yuv400 : Av1ColorFormat.Yuv420); + + Configuration configuration = Configuration.Default.Clone(); + TestMemoryAllocator successfulAllocator = new(); + successfulAllocator.EnableNonThreadSafeLogging(); + configuration.MemoryAllocator = successfulAllocator; + using (Av1FrameEncoder.SequenceEncoder encoder = encodeAlpha + ? Av1FrameEncoder.CreateAlphaSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9) + : Av1FrameEncoder.CreateColorSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9)) + { + Assert.NotEmpty(successfulAllocator.AllocationLog); + } + + Assert.Equal(successfulAllocator.AllocationLog.Count, successfulAllocator.ReturnLog.Count); + for (int failureIndex = 0; failureIndex < successfulAllocator.AllocationLog.Count; failureIndex++) + { + FailingSequenceAllocator allocator = new(failureIndex); + configuration.MemoryAllocator = allocator; + + // Fail each real allocator request, including those made inside nested constructors. A constructor + // that throws never reaches the caller's using statement, so its completed owners must unwind there. + InvalidMemoryOperationException exception = Assert.Throws(() => + { + using Av1FrameEncoder.SequenceEncoder encoder = encodeAlpha + ? Av1FrameEncoder.CreateAlphaSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9) + : Av1FrameEncoder.CreateColorSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9); + }); + + Assert.Equal("Sequence allocation failure.", exception.Message); + Assert.Equal(failureIndex, allocator.AllocationLog.Count); + Assert.All( + allocator.AllocationLog, + allocation => Assert.Single(allocator.ReturnLog, returned => returned.AllocationId == allocation.AllocationId)); + + Assert.Equal(allocator.AllocationLog.Count, allocator.ReturnLog.Count); + } + } + [Theory] [InlineData(false, EightBit, Yuv420, 384)] [InlineData(false, TwelveBit, Yuv444, 288)] @@ -1879,4 +1928,30 @@ public class Av1EncoderFrameTests } } } + + private sealed class FailingSequenceAllocator : TestMemoryAllocator + { + private readonly int failureIndex; + + /// + /// Initializes a new instance of the class. + /// + /// The zero-based allocation request that fails. + public FailingSequenceAllocator(int failureIndex) + { + this.failureIndex = failureIndex; + this.EnableNonThreadSafeLogging(); + } + + /// + protected override AllocationTrackedMemoryManager AllocateCore(int length, AllocationOptions options) + { + if (this.AllocationLog.Count == this.failureIndex) + { + throw new InvalidMemoryOperationException("Sequence allocation failure."); + } + + return base.AllocateCore(length, options); + } + } }