diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 9fe2cb1453..ca562c6f27 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -143,6 +143,54 @@ Sequence-construction ownership correction, verified subsequently on 2026-09-05: child environment keys case-insensitively, disable collection parallelism, stop on failure, and run serially. Absence of an observed dialog is not evidence that no popup occurred. +Ordinary-intra skip investigation at checkpoint `c778217a9`: + +- Managed `Av1IntraSuperblockEncoder.ModeDecision.cs:622-667,758-829,1291-1396` scans retained empty + transforms, computes their rate again, and can replace ordinary intra coefficient syntax with block skip. + `Av1TileWriter.cs:2628-2648` explicitly describes this as a departure from current libaom. +- Verified reference `av1/encoder/rdopt.c:3516-3578` assigns ordinary intra rate including + `skip_txfm_cost[skip_ctx][0]` and sets `skip_txfm = 0` before separate IBC evaluation. + Its intra-in-inter-frame path does the same at `rdopt.c:5772-5801`. + Final coding enforces this at `av1/encoder/partition_search.c:2122`. + Reference `av1/common/blockd.h:372-374` counts IBC as inter for this decision. +- This is an encoder-policy deviation, not an invalid-bitstream claim. Suppressing empty transform symbols + changes block-skip and coefficient probability adaptation for subsequent blocks even when current pixels match. + Managed `Av1TileWriter.cs:825-852,1100-1143` consumes the selected flag in symbol order and publishes neighbors. +- The regression `PreservesIntraNonSkipForAllZeroTransforms` asserts the reference policy while retaining + all zero-EOB and precomputed-versus-live syntax checks. It failed on the fixed-DC traversal before correction: + mono intra returned `Skip = true` (`intra-skip-before.trx` in the local takeover report directory). + The first affected run then exposed a second path: `Av1IntraSuperblockEncoder.cs:214-285` independently + marked zero-coefficient blocks skipped, so its exact byte comparison with the corrected live path failed. + Both paths now preserve ordinary-intra non-skip, and their exact byte comparison is retained. + These changes enforce the demonstrated reference contract; no pixel tolerance was changed. +- Automatic approval review rejected a combined production patch and deletion of the old helper-specific + entropy test. A second review rejected their removal after verification as weakened coverage. + The unused helper and its test remain unchanged; no further removal was attempted. + With that helper and test still present, the corrected production, fixed-DC, and public encoder paths pass + 258/258 serialized Release .NET 11 VSTest cases (`intra-skip-r2.trx`). Decoded mode assertions cover ordinary + intra syntax in key/inter frames at 8/10/12 bits and all three color subsampling formats; repeated-frame + tests still require skipped inter blocks. The build has zero errors and 1,009 existing test-project warnings, + with none in the changed files; Roslynk reports zero compiler errors. + Optimized native decoding of freshly regenerated streams matches 23,396 samples: maximum error 0 and zero + samples exceeding one. This is bounded same-stream evidence, not separate-encoder parity. + Roslynk now finds only the old entropy test referencing the obsolete skip helper; no production caller remains. + No benchmark was run, and the full controller, coefficient optimization, and separate-encoder gates remain open. + +Coefficient optimization and evaluation-stage investigation: + +- `av1/encoder/encodemb.c:208-224,474-563,831-887` selects quantization and coefficient optimization + from segment and evaluation policy before publishing coefficient context and reconstruction. +- `av1/encoder/rdopt_utils.h:608-709` distinguishes default, mode, and winner evaluation: transform + pruning, default transform use, skip/DC prediction, distortion domain, coefficient optimization threshold, + and transform-size search differ by stage. Changing stage invalidates cached RD results. +- `av1/encoder/txb_rdopt.c:400-560` consumes plane/block RD scaling, transform/EOB/context costs, + quantized and original coefficients, dequantization and matrices. It can lower coefficients, move EOB, + and select an empty transform; it updates coefficients, EOB, entropy context, and rate together. + Importing this primitive without those callers and their state would leave the controller deviation unresolved. +- `aom/aomcx.h:215-221`, `av1/av1_cx_iface.c:781-782,1401-1409`, and + `av1/encoder/speed_features.c:2709-2776` show usage-specific CPU settings and feature initialization. + ImageSharp's 0-10 effort scale has not yet been reconciled with these policies. No new effort mapping is assumed. + ### Required completion gates Motion-controller investigation continued after correction checkpoint `578ec34d9`: @@ -1052,7 +1100,7 @@ Encoder verification contract: - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. High-bit-depth paths widen before multiplying instead of applying the eight-bit coefficient clamp. Lossless blocks use the AV1 4x4 Walsh-Hadamard transform, exact lossless quantization and dequantization, four-by-four-only transform syntax, and non-skipped residual coding. Transform search and coefficient optimization remain. - [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma, filter-intra, palette, and intra-block copy, now perform live rate-distortion selection. Public quality mapping and effort tiers through exhaustive uniform luma mode/transform search are implemented. Effort nine adds exact recursive 8x8 and 16x16 partition rate-distortion selection, and effort ten extends it through 128x128; effort-dependent pruning and the remaining sequence searches remain. - [~] Frame effort now progressively expands the available current search: zero is DC-only, one adds every zero-angle spatial mode, two adds every legal directional adjustment, three refines the preliminary luma winner's transform type, four adds filter-intra and chroma-from-luma, and five adds adaptive palette and intra-block-copy analysis. Lower tiers do not signal unavailable sequence or frame tools, and tiers below five skip the whole-frame screen-content scan. Effort six enables `TX_MODE_SELECT` and compares the winning ordinary spatial or filter-intra luma mode as one 8x8 transform against four raster-ordered 4x4 transforms; each luma palette candidate owns that size comparison from effort six onward. Effort seven searches every legal 8x8 transform type inside every ordinary spatial candidate rather than refining only the preliminary winner. Effort eight also performs the 8x8-versus-four-4x4 comparison inside every ordinary spatial and filter-intra candidate, matching current libaom's per-candidate uniform-transform ownership. Effort nine additionally searches every legal partition at complete 8x8 and 16x16 nodes in current-libaom order, and effort ten extends that recursive search through 128x128. Prediction and residual construction run once per mode and are reused across its legal transform types. A 128x128 leaf evaluates four 64x64 luma transforms and as many as sixteen 32x32 transforms per 4:4:4 chroma plane, retaining sparse state at coefficient-area offsets. Residual emission follows AV1's bounded-region order, completing Y, U, and V for each 64x64 luma region before advancing. Every 4x4 transform searches all legal types with live coefficient contexts and reconstructed intra references. The search reuses the aligned block workspace, preserves only global improvements, and performs no per-block, per-partition, or per-transform rent. Non-skipped intra-block copy writes and costs the current-libaom unsplit variable-transform root; skipped intra-block copy emits no transform-partition symbol. Effort-dependent model and transform pruning remain. Decoder-visible production cases inspect the emitted restrictions and frame state and decode the produced streams, including real effort-nine streams selecting sub-8x8 and 8x16 rectangular blocks. The complete non-HEVC HEIF/AV1 namespace passes 9,077 of 9,077 through one foreground net11 Release VSTest run. The last independently built `aomdec`, from the then-current `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` snapshot, accepts the previously generated effort-eight and effort-ten payloads as well as the existing palette and intra-block-copy payloads. The affected encoder, partition, and workspace surface passes 139 of 139 through one foreground net11 Release VSTest run. The net11 Release build and Roslynk compiler and analyzer passes report zero errors. -- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. The corrected prepared reference edges retain the common-corner prefix and width-plus-height extent required by rectangular directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these reusable workspace-backed selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma and chroma candidate scratch is partitioned from the encoder's single aligned reusable block workspace; transform-size search uses that owner for four retained 4x4 transform states, local coefficient contexts, and the compact trial reconstruction needed to preserve the best result. No candidate path rents a buffer per block or per transform. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. Luma palette selection now evaluates dominant-color and one-dimensional K-means candidates for every legal size, snaps near-cache colors with the reference threshold and tie order, removes duplicate snapped colors, extends boundary maps from active samples, and performs complete transform rate-distortion search. Ordinary DC and filter-intra candidates pay the palette-disabled symbol whenever screen-content syntax is enabled. The exact net11 Release rebuild reports 1,992 test-project warnings and zero errors, all 58 intra-superblock cases pass, all 8,935 AVIF cases pass, and all 230 HEIF cases pass. Remaining mode decision work includes transform-size coverage for filter-intra and palette, broader joint mode/transform refinement, and effort-dependent pruning. Non-empty intra blocks deliberately remain non-skipped, matching current libaom; later inter mode selection owns its distinct skip-transform RD decision. +- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. The corrected prepared reference edges retain the common-corner prefix and width-plus-height extent required by rectangular directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these reusable workspace-backed selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma and chroma candidate scratch is partitioned from the encoder's single aligned reusable block workspace; transform-size search uses that owner for four retained 4x4 transform states, local coefficient contexts, and the compact trial reconstruction needed to preserve the best result. No candidate path rents a buffer per block or per transform. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search select DC with non-skip coefficient syntax, instead of relying on textured content to happen to select the baseline mode. Luma palette selection now evaluates dominant-color and one-dimensional K-means candidates for every legal size, snaps near-cache colors with the reference threshold and tie order, removes duplicate snapped colors, extends boundary maps from active samples, and performs complete transform rate-distortion search. Ordinary DC and filter-intra candidates pay the palette-disabled symbol whenever screen-content syntax is enabled. The exact net11 Release rebuild reports 1,992 test-project warnings and zero errors, all 58 intra-superblock cases pass, all 8,935 AVIF cases pass, and all 230 HEIF cases pass. Remaining mode decision work includes transform-size coverage for filter-intra and palette, broader joint mode/transform refinement, and effort-dependent pruning. Ordinary intra blocks now remain non-skipped even when all transforms are empty; inter and intra-block-copy mode selection own their distinct skip-transform RD decisions. - [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One picture owner packs segmentation, every tile's partition, luma, chroma, and transform edges, CDEF state, preceding quantizer, and encoded payload bounds into one clean byte allocation with typed non-owning views; together with the separately typed packed mode-information owner, the complete picture state uses two allocator rents rather than seven. Each encoded tile has independent neighbor and probability state while sharing the bounded output owner. Earlier aligned-length and balanced-return coverage exists; runtime allocation verification of the current multi-tile layout remains pending. - [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains. - [~] The superblock decision and palette-map workspace uses one reusable 40.3 KiB ImageSharp allocator owner. Its aligned 8.3 KiB decision region contains 1,024 explicitly packed 8-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree; its remaining 32 KiB contains the fixed 128x128 luma and chroma palette maps. This combines storage held separately by libaom; exact lifetime and size reconciliation remains part of the fresh allocation audit, and fewer owners alone does not establish an improvement. Palette colors have their own current-block value and are copied only to the picture edges that later blocks can reference, so enabling palette mode does not add 50 bytes to every final-block entry. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled quantizer, prediction, partition, and current-palette bytes cannot leak into the next decision pass. Roslynk reports zero compiler errors for the current one-owner refactor; runtime allocation verification remains pending. @@ -1085,7 +1133,11 @@ Encoder verification contract: - [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Paths above the speed-zero screen-content threshold continue through libaom's 256-pixel, one-pixel-step exhaustive mesh. Four adjacent byte or high-bit-depth candidates share each SIMD source load, strict row-major tie ordering is retained, and the final legal tail column remains searchable where libaom's current four-wide remainder loop omits it. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. Production mode decision now derives the same spatial displacement reference used by the writer, deduplicates hash and full-pixel finalists in search order, and evaluates every surviving vector through complete luma and chroma transform RD. This differs from libaom's preliminary-error pruning by permitting a hash and pixel finalist from the same search region to compete using final syntax and reconstruction costs. It is an unresolved controller deviation, with no established quality or performance improvement. Prediction is prepared once per plane and vector, including integer or half-sample chroma phase, then reused across every legal inter transform without an allocator rent or frame copy. The joint comparison includes the live intra-block-copy flag, discounted displacement rate, skip flag, coefficient syntax, and normalized Y/U/V distortion; an empty transform alternative can win only when its complete skip cost is strictly lower, while conventional intra and earlier vectors retain tie precedence. Winning reconstruction, coefficients, transform state, DC modes, cleared palette/filter/CfL state, and displacement are copied once into the existing retained stores. Production regressions force the path at 8 and 12 bits and force 4:2:0 horizontal half-sample chroma with an unaligned reference. The former bulk local workspace occupied 2.75 KiB for byte samples or 3.375 KiB for high-bit-depth samples. Prediction, candidate, winning reconstruction, residual, and coefficient scratch now occupy one naturally aligned 3.125 KiB extension of the existing frame-reused block-workspace owner, matching libaom's reusable macroblock-scratch lifetime without adding an allocation; only the 128-byte reference, weight, and finalist arrays remain on the stack. The net11 Release solution build reports zero errors; all 2,082 focused transform, entropy, intra-block-copy, intra-superblock, and frame-encoder cases and all 9,242 non-HEVC HEIF/AV1 cases pass through direct foreground VSTest, with tiered compilation disabled only for the full allocation-sensitive suite. Adaptive production activation is complete, and the emitted frame flag remains authoritative for the complete frame rather than being invalidated after tile coding. - [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips. - [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips. -- [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips. +- [ ] The earlier fixed-block skip checkpoint was not equivalent to libaom's ordinary-intra policy. + Its all-zero-EOB conjunction produced valid streams, but ordinary intra retains non-skip syntax in the + reference. The takeover correction above aligns both fixed-DC traversal and live mode selection. + Earlier decoder acceptance, unchanged frame hashes, and one-byte size reductions did not establish + encoder-policy parity; the earlier 8,862-case result is historical evidence only. - [x] Operation-wide allocation tracking now exercises a real 64x64 12-bit 4:4:4 frame through packed-pixel conversion, both native frame owners, picture and coefficient state, reusable block workspaces, entropy coding, OBU framing, and a non-seekable destination. It proves exactly one 60 KiB tile-output reservation from current libaom's all-intra 2.5x rule and balanced exactly-once returns for every tracked allocation before the operation completes. The focused ownership case passes 1 of 1 and the complete HEIF/AV1 namespace passes 8,863 of 8,863 direct net11 VSTest cases with zero failures or skips. - [x] HEIF box offsets are now counted from the start of the encoded file instead of reading `Stream.Position`. This preserves ISO BMFF file-relative `iloc` offsets when the destination begins at a nonzero position and permits non-seekable output. Decoder item extents and image-sequence chunk offsets now resolve from that same file origin rather than the backing stream origin. Real legacy-JPEG HEIF round trips cover non-seekable output and a prefixed destination, while current-position AV1 decode covers both a still item and a five-frame sequence. All 96 encoder/decoder cases and all 38 sequence-parser cases pass direct net11 Release VSTest; the Release build remains at the established 1,005-warning baseline with zero errors. diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs index 2986908368..7ba63c6d02 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs @@ -619,42 +619,10 @@ internal static partial class Av1IntraSuperblockEncoder block.FilterIntraMode = filterIntraMode; modeInfo.Block.TransformSize = lumaTransformSize; - // Mode decision retains one state for every uniform transform tile in the coding block. The block - // can skip coefficient syntax only when every retained transform has an empty end-of-block marker. - int lumaTransformSampleCount = lumaTransformSize.GetSize2d(); - int lumaTransformBlockCount = - (blockSize.GetWidth() * blockSize.GetHeight()) / lumaTransformSampleCount; - - int lumaStateStride = - lumaTransformSampleCount / Av1EncoderCoefficientBuffer.TransformBlockUnitCoefficientCount; - - bool lumaTransformEmpty = true; - for (int transformIndex = 0; transformIndex < lumaTransformBlockCount; transformIndex++) - { - lumaTransformEmpty &= retainedLumaStates[transformIndex * lumaStateStride].EndOfBlock == 0; - } - + // Ordinary intra keeps the block non-skipped, including when all transforms are empty. Its RD cost + // includes those transform symbols and the non-skip flag; only inter or IBC winners can replace this state. if (this.source.IsMonochrome) { - int emptyTransformRate = lumaTransformEmpty - ? this.GetEmptyTransformRate( - writer, - this.picture.LuminanceDcSignLevelCoefficientNeighbors[tileIndex], - Av1ComponentType.Luminance, - blockOrigin, - blockSize, - lumaTransformSize, - modeInfo.Block.Mode, - block.FilterIntraMode) - : 0; - - modeInfo.Block.Skip = - !this.picture.Parent.FrameHeader.CodedLossless && - lumaTransformEmpty && Av1TileWriter.ShouldSkipCoefficients( - writer, - Av1TileWriter.GetSkipContext(macroBlock), - emptyTransformRate); - bool allowIntraBlockCopy = blockSize == Av1BlockSize.Block8x8 && this.picture.Parent.FrameHeader.AllowIntraBlockCopy; @@ -662,8 +630,6 @@ internal static partial class Av1IntraSuperblockEncoder writer, macroBlock, lumaCost, - emptyTransformRate, - modeInfo.Block.Skip, allowIntraBlockCopy); if (!this.picture.Parent.FrameHeader.IsIntra) @@ -755,68 +721,6 @@ internal static partial class Av1IntraSuperblockEncoder block.PredictionUnit.ChromaFromLumaSigns = chromaFromLumaSigns; } - // Skip suppresses coefficient syntax for the entire coding block, not one plane independently. - // Preserve normal coefficient coding when any selected luma or chroma transform is nonempty. - int chromaTransformSampleCount = chromaTransformSize.GetSize2d(); - int chromaTransformBlockCount = - (chromaBlockSize.GetWidth() * chromaBlockSize.GetHeight()) / chromaTransformSampleCount; - - int chromaStateStride = - chromaTransformSampleCount / Av1EncoderCoefficientBuffer.TransformBlockUnitCoefficientCount; - - bool chromaTransformsEmpty = true; - for (int transformIndex = 0; transformIndex < chromaTransformBlockCount; transformIndex++) - { - int stateIndex = transformIndex * chromaStateStride; - chromaTransformsEmpty &= retainedBlueStates[stateIndex].EndOfBlock == 0 && - retainedRedStates[stateIndex].EndOfBlock == 0; - } - - bool allTransformsEmpty = lumaTransformEmpty && (!block.HasChroma || chromaTransformsEmpty); - - int regularEmptyTransformRate = 0; - if (allTransformsEmpty) - { - regularEmptyTransformRate = this.GetEmptyTransformRate( - writer, - this.picture.LuminanceDcSignLevelCoefficientNeighbors[tileIndex], - Av1ComponentType.Luminance, - blockOrigin, - blockSize, - lumaTransformSize, - modeInfo.Block.Mode, - block.FilterIntraMode); - - if (block.HasChroma) - { - regularEmptyTransformRate += this.GetEmptyTransformRate( - writer, - this.picture.CbDcSignLevelCoefficientNeighbors[tileIndex], - Av1ComponentType.Chroma, - chromaOrigin, - chromaBlockSize, - chromaTransformSize, - modeInfo.Block.Mode, - Av1FilterIntraMode.AllFilterIntraModes); - - regularEmptyTransformRate += this.GetEmptyTransformRate( - writer, - this.picture.CrDcSignLevelCoefficientNeighbors[tileIndex], - Av1ComponentType.Chroma, - chromaOrigin, - chromaBlockSize, - chromaTransformSize, - modeInfo.Block.Mode, - Av1FilterIntraMode.AllFilterIntraModes); - } - - modeInfo.Block.Skip = !this.picture.Parent.FrameHeader.CodedLossless && - Av1TileWriter.ShouldSkipCoefficients( - writer, - Av1TileWriter.GetSkipContext(macroBlock), - regularEmptyTransformRate); - } - bool allowColorIntraBlockCopy = blockSize == Av1BlockSize.Block8x8 && this.picture.Parent.FrameHeader.AllowIntraBlockCopy; @@ -824,8 +728,6 @@ internal static partial class Av1IntraSuperblockEncoder writer, macroBlock, lumaCost + chromaCost, - regularEmptyTransformRate, - modeInfo.Block.Skip, allowColorIntraBlockCopy); if (!this.picture.Parent.FrameHeader.IsIntra) @@ -1292,11 +1194,9 @@ internal static partial class Av1IntraSuperblockEncoder Av1SymbolEncoder writer, Av1MacroBlockD macroBlock, long modeCost, - int emptyTransformRate, - bool skip, bool allowIntraBlockCopy) { - int rateAdjustment = writer.GetSkipCost(skip, Av1TileWriter.GetSkipContext(macroBlock)); + int rateAdjustment = writer.GetSkipCost(false, Av1TileWriter.GetSkipContext(macroBlock)); if (!this.picture.Parent.FrameHeader.IsIntra) { int intraInterContext = Av1TileWriter.GetIntraInterContext(macroBlock); @@ -1308,93 +1208,9 @@ internal static partial class Av1IntraSuperblockEncoder rateAdjustment += writer.GetUseIntraBlockCopyCost(false); } - if (skip) - { - // Mode search includes empty transform symbols, while block skip suppresses them from the bitstream. - rateAdjustment -= emptyTransformRate; - } - return modeCost + Av1RateDistortion.GetCost(this.rateMultiplier, rateAdjustment, 0); } - private int GetEmptyTransformRate( - Av1SymbolEncoder writer, - Av1NeighborArrayUnit coefficientNeighbors, - Av1ComponentType componentType, - Point blockOrigin, - Av1BlockSize blockSize, - Av1TransformSize transformSize, - Av1PredictionMode lumaMode, - Av1FilterIntraMode filterIntraMode) - { - int blockWidth = blockSize.Get4x4WideCount(); - int blockHeight = blockSize.Get4x4HighCount(); - int transformWidth = transformSize.Get4x4WideCount(); - int transformHeight = transformSize.Get4x4HighCount(); - if (blockWidth == transformWidth && blockHeight == transformHeight) - { - Av1TransformBlockContext blockContext = Av1TileWriter.GetTransformBlockContexts( - componentType, - coefficientNeighbors, - blockOrigin, - blockSize, - transformSize); - - return writer.GetCoefficientCost( - transformSize, - Av1TransformType.DctDct, - lumaMode, - ReadOnlySpan.Empty, - componentType, - blockContext, - 0, - this.picture.Parent.FrameHeader.UseReducedTransformSet, - filterIntraMode, - usesInterTransformSet: false); - } - - Span contexts = this.blockWorkspace - .GetModeDecisionWorkspace() - .TransformContexts; - - Span topContexts = contexts[..blockWidth]; - Span leftContexts = contexts.Slice(blockWidth, blockHeight); - int topIndex = coefficientNeighbors.GetTopIndex(blockOrigin); - int leftIndex = coefficientNeighbors.GetLeftIndex(blockOrigin); - coefficientNeighbors.Top.Slice(topIndex, blockWidth).CopyTo(topContexts); - coefficientNeighbors.Left.Slice(leftIndex, blockHeight).CopyTo(leftContexts); - int rate = 0; - for (int blockRow = 0; blockRow < blockHeight; blockRow += transformHeight) - { - for (int blockColumn = 0; blockColumn < blockWidth; blockColumn += transformWidth) - { - Av1TransformBlockContext blockContext = Av1TileWriter.GetTransformBlockContexts( - componentType, - topContexts.Slice(blockColumn, transformWidth), - leftContexts.Slice(blockRow, transformHeight), - blockSize, - transformSize); - - rate += writer.GetCoefficientCost( - transformSize, - Av1TransformType.DctDct, - lumaMode, - ReadOnlySpan.Empty, - componentType, - blockContext, - 0, - this.picture.Parent.FrameHeader.UseReducedTransformSet, - filterIntraMode, - usesInterTransformSet: false); - - topContexts.Slice(blockColumn, transformWidth).Clear(); - leftContexts.Slice(blockRow, transformHeight).Clear(); - } - } - - return rate; - } - private Av1PredictionMode SelectLumaMode( Av1SymbolEncoder writer, Av1MacroBlockD macroBlock, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.cs index 33127ffc06..e815a89ce5 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.cs @@ -245,10 +245,8 @@ internal static partial class Av1IntraSuperblockEncoder ref lumaState); this.codedAreaLuma += LumaTransformSize.GetSize2d(); - bool skipTransform = lumaState.EndOfBlock == 0; if (this.source.IsMonochrome) { - modeInfo.Block.Skip = skipTransform; return; } @@ -279,8 +277,7 @@ internal static partial class Av1IntraSuperblockEncoder this.redCoefficients[this.codedAreaChroma..], ref redState); - // A block-level skip is valid only when every coded plane reconstructs directly from its prediction. - modeInfo.Block.Skip = skipTransform && blueState.EndOfBlock == 0 && redState.EndOfBlock == 0; + // Ordinary intra retains non-skip syntax even for empty transforms, matching live mode selection. this.codedAreaChroma += chromaTransformSize.GetSize2d(); } diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs index 179b7181c9..a645ef9de0 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs @@ -401,6 +401,17 @@ public class Av1EncoderFrameTests Assert.Equal(Width, decoded.Width); Assert.Equal(Height, decoded.Height); Assert.Equal(bitDepth, decoded.BitDepth); + Av1FrameInfo decodedFrameInfo = Assert.IsType(decoder.FrameInfo); + foreach (Av1BlockModeInfo mode in decodedFrameInfo.GetModeInfos(Point.Empty, decodedFrameInfo.GetModeInfoCount(Point.Empty))) + { + // The same ordinary-intra policy applies in key and inter frames. Read the emitted syntax, + // rather than infer the skip flag from pixel agreement between encoder and decoder. + if (mode.ReferenceFrames[0] == Av1ReferenceFrameType.Intra && !mode.UseIntraBlockCopy) + { + Assert.False(mode.Skip); + } + } + for (int planeIndex = 0; planeIndex < 3; planeIndex++) { Av1Plane plane = (Av1Plane)planeIndex; @@ -501,6 +512,14 @@ public class Av1EncoderFrameTests null, null); + Av1FrameInfo firstFrameInfo = Assert.IsType(decoder.FrameInfo); + foreach (Av1BlockModeInfo mode in firstFrameInfo.GetModeInfos(Point.Empty, firstFrameInfo.GetModeInfoCount(Point.Empty))) + { + Assert.Equal(Av1ReferenceFrameType.Intra, mode.ReferenceFrames[0]); + Assert.False(mode.UseIntraBlockCopy); + Assert.False(mode.Skip); + } + using ImageFrame decodedSecond = decoder.DecodeSequenceFrame( secondSample.ToArray(), null, @@ -515,6 +534,15 @@ public class Av1EncoderFrameTests Assert.Equal(37, frameHeader.QuantizationParameters.BaseQIndex); Assert.Equal(switchableFilters ? Av1InterpolationFilter.Switchable : Av1InterpolationFilter.Regular, frameHeader.InterpolationFilter); Assert.Equal(dualFilters, decoder.SequenceHeader.EnableDualFilter); + Av1FrameInfo secondFrameInfo = Assert.IsType(decoder.FrameInfo); + bool hasSkippedInterBlock = false; + foreach (Av1BlockModeInfo mode in secondFrameInfo.GetModeInfos(Point.Empty, secondFrameInfo.GetModeInfoCount(Point.Empty))) + { + hasSkippedInterBlock |= mode.ReferenceFrames[0] == Av1ReferenceFrameType.Last && mode.Skip; + } + + // Repeated frames still use the inter skip alternative when prediction supplies the retained samples. + Assert.True(hasSkippedInterBlock); for (int y = 0; y < Height; y++) { Assert.Equal( diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs index 93d16a8e6a..ff05a26f51 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs @@ -463,7 +463,7 @@ public class Av1IntraSuperblockEncoderTests Assert.Equal(Av1TransformSize.Size8x8, block.Block.TransformSize); Assert.Equal(Av1PredictionMode.DC, block.Block.Mode); Assert.Equal(Av1ChromaPredictionMode.DC, block.Block.UvMode); - Assert.True(block.Block.Skip); + Assert.False(block.Block.Skip); } Span lumaStates = coefficients.GetTransformBlockSpan(0, Av1Plane.Y); @@ -602,7 +602,7 @@ public class Av1IntraSuperblockEncoderTests [Theory] [InlineData(true)] [InlineData(false)] - public void MarksAllZeroTransformBlockAsSkipped(bool isMonochrome) + public void PreservesIntraNonSkipForAllZeroTransforms(bool isMonochrome) { const int Width = 8; const int Height = 8; @@ -678,7 +678,10 @@ public class Av1IntraSuperblockEncoderTests blockWorkspace); ref Av1MacroBlockModeInfo block = ref picture.GetMacroBlockModeInfo(default); - Assert.True(block.Block.Skip); + + // Ordinary intra blocks retain the non-skip flag and empty transform symbols. Libaom applies + // this policy before final coding even when skipping would reconstruct the same samples. + Assert.False(block.Block.Skip); Assert.Equal((ushort)0, coefficients.GetTransformBlockSpan(0, Av1Plane.Y)[0].EndOfBlock); if (!isMonochrome) {