From 339972b7fe92f26d37172f3ddf91120e4130bef0 Mon Sep 17 00:00:00 2001 From: James Jackson-South Date: Thu, 3 Sep 2026 13:36:47 +1000 Subject: [PATCH] Add AV1 luma transform-size search --- HEIF_IMPLEMENTATION_PLAN.md | 10 +- .../Av1/Entropy/Av1SymbolContextHelper.cs | 68 +++ .../Heif/Av1/Entropy/Av1SymbolEncoder.cs | 26 + .../Av1EncoderModeDecisionWorkspace.cs | 25 +- .../Heif/Av1/Pipeline/Av1FrameEncoder.cs | 2 +- ...rblockEncoder.ChromaPaletteModeDecision.cs | 2 + ...blockEncoder.IntraBlockCopyModeDecision.cs | 20 + .../Av1IntraSuperblockEncoder.ModeDecision.cs | 560 +++++++++++++++++- .../Av1IntraSuperblockEncoder.Operator.cs | 100 +++- ...raSuperblockEncoder.PaletteModeDecision.cs | 7 + .../Av1/Pipeline/Av1TransformBlockEncoder.cs | 291 ++++++--- .../Formats/Heif/Av1/Tiling/Av1TileWriter.cs | 98 ++- .../Heif/Av1/Av1CoefficientsEntropyTests.cs | 55 ++ .../Formats/Heif/Av1/Av1EncoderFrameTests.cs | 93 ++- .../Formats/Heif/Av1/Av1EntropyTests.cs | 9 + .../Formats/Heif/Av1/Av1SymbolContextTests.cs | 41 ++ 16 files changed, 1264 insertions(+), 143 deletions(-) diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 2871f302d9..988620acfc 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -829,9 +829,9 @@ Encoder verification contract: - [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes. Palette entropy, retained state, production syntax, exhaustive luma and paired chroma palette selection, adaptive screen-content activation, and joint intra-block-copy mode selection are complete. - [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain. -- [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma, filter-intra, palette, and intra-block copy, now perform live rate-distortion selection; quality mapping, effort tiers above the current fixed-block search ceiling, and the remaining searches are not implemented. -- [~] Frame effort now progressively expands the available current search: zero is DC-only, one adds every zero-angle spatial mode, two adds every legal directional adjustment, three adds transform refinement, four adds filter-intra and chroma-from-luma, and five adds adaptive palette and intra-block-copy analysis. Lower tiers do not signal unavailable sequence or frame tools, and tiers below five skip the whole-frame screen-content scan. Values six through ten currently share the exhaustive fixed-8x8 search ceiling and remain open until transform-size and partition searches provide additional work. Six decoder-visible production cases verify emitted flags, mode restrictions, and successful decode. The exact net11 Release test-project build reports 1,992 baseline warnings and zero errors; the clean complete HEIF/AV1 namespace run passes 9,258 of 9,258 through direct foreground VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 40 current generated payloads. Roslynk reports zero errors and no diagnostics in the touched files. -- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these stack-only selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma candidate scratch remains one 8x8 reconstruction and one 8x8 coefficient span on the stack; chroma uses one transform-sized reconstruction and coefficient span for each of U and V. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. Luma palette selection now evaluates dominant-color and one-dimensional K-means candidates for every legal size, snaps near-cache colors with the reference threshold and tie order, removes duplicate snapped colors, extends boundary maps from active samples, and performs complete transform rate-distortion search. Ordinary DC and filter-intra candidates pay the palette-disabled symbol whenever screen-content syntax is enabled. The exact net11 Release rebuild reports 1,992 test-project warnings and zero errors, all 58 intra-superblock cases pass, all 8,935 AVIF cases pass, and all 230 HEIF cases pass. Remaining mode decision work includes transform-size search, broader joint mode/transform refinement, partition search, and effort-dependent pruning. Non-empty intra blocks deliberately remain non-skipped, matching current libaom; later inter mode selection owns its distinct skip-transform RD decision. +- [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma, filter-intra, palette, and intra-block copy, now perform live rate-distortion selection; quality mapping, effort tiers above the current uniform-transform search ceiling, and the remaining searches are not implemented. +- [~] Frame effort now progressively expands the available current search: zero is DC-only, one adds every zero-angle spatial mode, two adds every legal directional adjustment, three adds transform refinement, four adds filter-intra and chroma-from-luma, and five adds adaptive palette and intra-block-copy analysis. Lower tiers do not signal unavailable sequence or frame tools, and tiers below five skip the whole-frame screen-content scan. Effort six enables `TX_MODE_SELECT` and compares the winning ordinary spatial luma mode as one 8x8 transform against four raster-ordered 4x4 transforms. Every 4x4 transform searches all legal transform types with live coefficient contexts and reconstructed intra references. The search reuses the aligned block workspace, preserves only improving candidates, and performs no per-block or per-transform rent. Non-skipped intra-block copy writes and costs the current-libaom unsplit variable-transform root; skipped intra-block copy emits no transform-partition symbol. Values seven through ten currently share the effort-six ceiling. Transform-size integration for filter-intra and palette, broader joint mode/transform refinement, partition search, and pruning remain. Ten decoder-visible production cases verify emitted flags, mode restrictions, real 4x4 selection, intra-block-copy syntax, and successful decode. The exact net11 Release test-project build remains at 1,005 baseline warnings and zero errors; the clean complete non-HEVC HEIF/AV1 namespace passes 9,294 of 9,294 through direct foreground VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts both new effort-six payloads in addition to the previously verified set; their decoded-frame MD5 values are `2dd1cbe449fe2d0471dc2c15c50acb69` for spatial 4x4 transform selection and `677435e5af39c930af1178f91c34af6a` for intra-block copy. Roslynk reports zero compiler errors and no analyzer diagnostics in the touched files. +- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these reusable workspace-backed selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma and chroma candidate scratch is partitioned from the encoder's single aligned reusable block workspace; transform-size search uses that owner for four retained 4x4 transform states, local coefficient contexts, and the compact trial reconstruction needed to preserve the best result. No candidate path rents a buffer per block or per transform. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. Luma palette selection now evaluates dominant-color and one-dimensional K-means candidates for every legal size, snaps near-cache colors with the reference threshold and tie order, removes duplicate snapped colors, extends boundary maps from active samples, and performs complete transform rate-distortion search. Ordinary DC and filter-intra candidates pay the palette-disabled symbol whenever screen-content syntax is enabled. The exact net11 Release rebuild reports 1,992 test-project warnings and zero errors, all 58 intra-superblock cases pass, all 8,935 AVIF cases pass, and all 230 HEIF cases pass. Remaining mode decision work includes transform-size coverage for filter-intra and palette, broader joint mode/transform refinement, partition search, and effort-dependent pruning. Non-empty intra blocks deliberately remain non-skipped, matching current libaom; later inter mode selection owns its distinct skip-transform RD decision. - [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One picture owner now packs segmentation plus every tile's partition, luma, chroma, and transform edges into one clean byte allocation with typed non-owning views; together with the separately typed packed mode-information owner, the complete picture state uses two allocator rents rather than seven. Exact aligned lengths, clean initialization, and balanced exactly-once returns are covered in Release. Multi-tile payload ownership and verified CDF update behavior remain. - [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains. - [~] The final-block decision workspace uses one reusable 8.3 KiB ImageSharp allocator owner. It contains 1,024 explicitly packed 8-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree, replacing separate managed arrays. Palette colors now have their own current-block value and are copied only to the picture edges that later blocks can reference, so enabling palette mode does not add 50 bytes to every final-block entry. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled quantizer, prediction, partition, and current-palette bytes cannot leak into the next decision pass. Complete mode decision still remains. @@ -848,7 +848,7 @@ Encoder verification contract: - [x] The exact net11 Release rebuild completed at the established 1,005-warning repository baseline with zero errors. The complete HEIF/AV1 namespace passes 8,838 of 8,838 direct VSTest cases with zero failures or skips. Roslynk reports zero compiler errors and no diagnostics in the five changed C# files; `git diff --check` passes and `.gitattributes` is unchanged. - [~] The same internal frame operation now produces 4:2:0, 4:2:2, and 4:4:4 payloads at 8, 10, and 12 bits. Twenty-one color cases cover constant and spatially varying input at aligned dimensions plus odd 13x11 visible dimensions for every chroma geometry. The production decoder consumes every payload, the decoded output retains non-neutral chroma, and current-main `aomdec` accepts all 29 monochrome and color outputs. After live spatial chroma mode selection, implicit chroma-transform correction, winner-mode luma-transform refinement, exhaustive chroma-from-luma alpha selection, and filter-intra search, the odd-dimension decoded-frame MD5 values are `9985f05790d2c9f5f28723ef86d5b89b` (4:2:0), `2ba2f1d0fcfef60394a5175553c7cb8b` (4:2:2), and `6aa7a2ed0dbf76ad2ec0c222585272d0` (4:4:4). This proves legal current-libaom payload syntax across native plane geometries; it does not yet prove target quality or native-plane equality with an independently encoded reference. - [x] Spatial chroma candidates now use the implicit transform derived from the selected UV mode and active transform set, matching current libaom's `intra_mode_to_tx_type` and `av1_get_tx_type` behavior. The same shared derivation is consumed by the decoder, so coefficient scan order, entropy contexts, inverse reconstruction, and encoder rate estimates cannot drift between the two paths. The previous DCT-DCT candidate transform could produce syntactically accepted streams whose non-DC chroma coefficients were interpreted under a different implicit transform. Six production mode-decision cases retain nonzero U and V coefficients and assert the selected transform state across 4:2:0, 4:2:2, and 4:4:4; fifteen exact mapping cases cover every intra mode, reduced sets, and the 32x32 DCT-only fallback. The focused contract passes 21 of 21 direct net11 VSTest cases, the complete HEIF/AV1 namespace passes 8,947 of 8,947, the exact Release rebuild remains at 1,005 warnings and zero errors, and current-main `aomdec` accepts all 29 regenerated payloads. -- [~] Luma mode selection now evaluates each of its 61 mode-and-angle candidates with the mode-derived default transform used by current libaom's fast intra path. It then refines only the winning mode across all seven transform types permitted by the 8x8 intra set in transform-enum order. This removes the fixed DCT-DCT limitation while avoiding a 61-by-7 expansion; each trial includes live transform-type and coefficient rate, reconstructed pixel-domain distortion, and the existing allocation-free stack scratch. Eighteen exact-prediction production cases prove DCT-DCT wins equal-cost ties in reference order even when the first pass used a different default, while the 72x72 textured traversal proves a non-DCT transform with nonzero coefficients reaches retained syntax. Current-main `aomdec` accepts all 29 regenerated payloads. Full partition, transform-size, and effort-dependent joint mode/transform search remain. +- [~] Luma mode selection now evaluates each of its 61 mode-and-angle candidates with the mode-derived default transform used by current libaom's fast intra path. It then refines only the winning mode across all seven transform types permitted by the 8x8 intra set in transform-enum order. This removes the fixed DCT-DCT limitation while avoiding a 61-by-7 expansion; each trial includes live transform-type and coefficient rate, reconstructed pixel-domain distortion, and the existing aligned reusable block workspace. Eighteen exact-prediction production cases prove DCT-DCT wins equal-cost ties in reference order even when the first pass used a different default, while the 72x72 textured traversal proves a non-DCT transform with nonzero coefficients reaches retained syntax. Current-main `aomdec` accepts all 29 regenerated payloads. Special-mode transform-size coverage, full partition search, and broader effort-dependent joint mode/transform search remain. - [x] Chroma-from-luma mode decision now reuses the decoder's SIMD-first 4:2:0, 4:2:2, and 4:4:4 reconstructed-luma preparation and prediction kernels for both byte and high-bit-depth encoder operators. The constant DC predictor for each chroma plane is computed once and its sample refills every alpha candidate, matching libaom's per-plane DC cache instead of rebuilding the same edge average 33 times. Each block uses 512 bytes of fixed stack scratch for the maximum 8-row predictor surface plus 792 bytes for complete U/V rate and distortion tables; no allocator owner, managed object, frame copy, or persistent buffer was added. Live probability costs exactly mirror current libaom's joint-sign ownership and conditional magnitude symbols. Nine production cases independently derive exact CfL targets from decoder-visible reconstructed luma at 8, 10, and 12 bits, and three entropy cases cover two nonzero signs plus each single-zero-plane form. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, all 8,959 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. - [x] Filter-intra mode decision now runs after ordinary luma modes in current-libaom order, evaluates all five recursive predictors, and refines each predictor across every legal 8x8 transform in transform-enum order. Strictly-better replacement preserves ordinary-mode and filter-mode tie order. Each filter prediction and its source residual are prepared once and reused across transform candidates, avoiding repeated recursive prediction while retaining SIMD-first predictor and subtraction operators. The stack cost is 192 bytes for eight-bit samples or 256 bytes for high-bit-depth samples; no allocator owner or managed buffer was added. Fifteen production cases force every filter mode at 8, 10, and 12 bits and prove retained filter syntax, zero-residual reconstruction, and the DCT-DCT equal-cost transform tie. The decoded-frame MD5 values selected by this checkpoint are `d7d68803763b95827483f14515281d3a` for the 8x8 10-bit gradient, `3f7e34d44c65d7797ad26b5cd4c35bf4` for the 8x8 12-bit gradient, and `9985f05790d2c9f5f28723ef86d5b89b`, `2ba2f1d0fcfef60394a5175553c7cb8b`, and `6aa7a2ed0dbf76ad2ec0c222585272d0` for the odd 4:2:0, 4:2:2, and 4:4:4 gradients. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, 18 focused filter-intra, predictor-reference, syntax-cost, and allocation cases pass, all 8,974 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. - [x] Empty-transform block skip now compares the complete live rate of the two decoder-identical syntax choices after luma and every coded chroma plane have been selected. Current libaom forces all-intra blocks to non-skip; this encoder retains that behavior for every non-empty block and for equal-cost empty blocks, but emits block skip when its adapted context cost is strictly lower than non-skip plus all empty-transform coefficient costs. Costing and writing share the same above-and-left skip-context calculation, and the coefficient estimator returns after the transform-block-skip symbol without reading coefficient storage. This adds no allocation, copy, or persistent state. A focused adapted-CDF regression proves both outcomes through the production decision helper, the two production all-zero fixtures still prove the default real block path, the exact net11 Release rebuild remains at 1,005 warnings and zero errors, all 8,975 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. @@ -868,6 +868,8 @@ Encoder verification contract: - [x] Operation-wide allocation tracking now exercises a real 64x64 12-bit 4:4:4 frame through packed-pixel conversion, both native frame owners, picture and coefficient state, reusable block workspaces, entropy coding, OBU framing, and a non-seekable destination. It proves exactly one 60 KiB tile-output reservation from current libaom's all-intra 2.5x rule and balanced exactly-once returns for every tracked allocation before the operation completes. The focused ownership case passes 1 of 1 and the complete HEIF/AV1 namespace passes 8,863 of 8,863 direct net11 VSTest cases with zero failures or skips. - [x] HEIF box offsets are now counted from the start of the encoded file instead of reading `Stream.Position`. This preserves ISO BMFF file-relative `iloc` offsets when the destination begins at a nonzero position and permits non-seekable output. Decoder item extents and image-sequence chunk offsets now resolve from that same file origin rather than the backing stream origin. Real legacy-JPEG HEIF round trips cover non-seekable output and a prefixed destination, while current-position AV1 decode covers both a still item and a five-frame sequence. All 96 encoder/decoder cases and all 38 sequence-parser cases pass direct net11 Release VSTest; the Release build remains at the established 1,005-warning baseline with zero errors. +- [~] Effort-six uniform luma transform selection now compares the winning ordinary spatial mode as one 8x8 transform against four raster-ordered 4x4 transforms. Each 4x4 transform searches every legal type with coefficient contexts derived from the already retained transform edges and the preceding trial blocks, while reconstructed top-right and bottom-left references follow production coding order. The strided transform operator writes each candidate directly into its 8x8 reconstruction mosaic. Prepared prediction and residual data are reused across transform trials, and the existing aligned block-workspace owner retains four final transform states, local coefficient contexts, coefficients, and compact trial reconstruction; no allocator rent, managed array, best-candidate re-transform, or full-block intermediate copy was added. Non-skipped intra-block copy under `TX_MODE_SELECT` emits and costs the unsplit variable-transform root required by current libaom, while skipped intra-block copy emits no transform-partition symbol. Uniform transform-size contexts use coding-block extents for intra-block-copy neighbors and residual-transform extents for intra neighbors. The focused contract passes 21 of 21 direct net11 Release VSTest cases, including an independently decoded stream that proves at least one real four-transform luma block. The complete non-HEVC HEIF/AV1 namespace passes 9,294 of 9,294 cases with zero failures or skips. The exact Release build remains at 1,005 baseline warnings and zero errors, Roslynk reports no touched-file diagnostics, and current-main `aomdec` accepts both generated effort-six streams with decoded-frame MD5 values `2dd1cbe449fe2d0471dc2c15c50acb69` and `677435e5af39c930af1178f91c34af6a`. Transform-size integration for filter-intra and palette, broader joint mode/transform refinement, partition search, and effort-dependent pruning remain. + ### 7. Write complete AVIF output - [~] The encoder-side AV1 codec configuration is now derived directly from the encoded sequence header and writes the fixed four-byte `av1C` record with empty `configOBUs`. The image payload retains the required sequence header, so the property introduces no sequence-header allocation, retention, or copy. Four production-header cases cover main, high, and professional profiles; 8-, 10-, and 12-bit precision; monochrome, 4:2:0, 4:2:2, and 4:4:4 sampling; exact fixed bytes; decoder reparsing; and header/property equivalence through direct net11 Release VSTest. Property-container emission and public AVIF activation remain open. diff --git a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolContextHelper.cs b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolContextHelper.cs index ac75040cfe..1b7ddd9706 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolContextHelper.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolContextHelper.cs @@ -288,6 +288,43 @@ internal static class Av1SymbolContextHelper return TransformBlockSkipContexts[(topClass * 5) + leftClass]; } + /// + /// Derives the variable-transform partition context from the current node and its adjacent transform edges. + /// + /// The transform width retained immediately above the current node. + /// The transform height retained immediately left of the current node. + /// The containing coding-block size. + /// The transform size represented by the current partition node. + /// The variable-transform partition context. + public static int GetTransformPartitionContext( + byte aboveTransformWidth, + byte leftTransformHeight, + Av1BlockSize blockSize, + Av1TransformSize transformSize) + { + if (transformSize <= Av1TransformSize.Size4x4) + { + return 0; + } + + int above = aboveTransformWidth < transformSize.GetWidth() ? 1 : 0; + int left = leftTransformHeight < transformSize.GetHeight() ? 1 : 0; + int maximumDimension = Math.Max(blockSize.GetWidth(), blockSize.GetHeight()); + Av1TransformSize maximumSquareTransform = maximumDimension switch + { + >= 64 => Av1TransformSize.Size64x64, + >= 32 => Av1TransformSize.Size32x32, + >= 16 => Av1TransformSize.Size16x16, + _ => Av1TransformSize.Size8x8 + }; + + int category = (transformSize.GetSquareUpSize() != maximumSquareTransform && + maximumSquareTransform > Av1TransformSize.Size8x8 ? 1 : 0) + + ((((int)Av1TransformSize.SquareSizes - 1) - (int)maximumSquareTransform) * 2); + + return (category * 3) + above + left; + } + /// /// Reconstructs an end-of-block coefficient position from its token and extra offset. /// @@ -668,6 +705,37 @@ internal static class Av1SymbolContextHelper } } + /// + /// Packs the magnitude class and DC sign retained by neighboring transform blocks. + /// + /// The raster-ordered quantized coefficients. + /// The transform dimensions. + /// The transform type selecting scan order. + /// The one-based final nonzero scan position. + /// The packed coefficient context, or zero for an empty transform. + public static byte GetCoefficientContext( + ReadOnlySpan coefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + ushort endOfBlock) + { + if (endOfBlock == 0) + { + return 0; + } + + ReadOnlySpan scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan; + int culLevel = 0; + for (int scanIndex = 0; scanIndex < endOfBlock; scanIndex++) + { + culLevel += Math.Abs(coefficients[scan[scanIndex]]); + } + + culLevel = Math.Min(Av1Constants.CoefficientContextMask, culLevel); + SetDcSign(ref culLevel, coefficients[0]); + return (byte)culLevel; + } + /// /// Converts a one-based end-of-block position to its token and group offset. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs index d75a5399a0..5b8781dbf6 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs @@ -117,6 +117,11 @@ internal class Av1SymbolEncoder : IDisposable /// private readonly Av1Distribution[][] transformSize; + /// + /// The tile-adaptive variable-transform partition distributions. + /// + private readonly Av1Distribution[] transformPartition; + /// /// The tile-adaptive spatial segment-identifier distributions. /// @@ -200,6 +205,7 @@ internal class Av1SymbolEncoder : IDisposable this.intraExtendedTransform = Av1DefaultDistributions.IntraExtendedTransform; this.interExtendedTransform = Av1DefaultDistributions.InterExtendedTransform; this.transformSize = Av1DefaultDistributions.TransformSize; + this.transformPartition = Av1DefaultDistributions.TransformPartition; this.segmentId = Av1DefaultDistributions.SegmentId; this.angleDelta = Av1DefaultDistributions.AngleDelta; this.skip = Av1DefaultDistributions.Skip; @@ -1208,6 +1214,26 @@ internal class Av1SymbolEncoder : IDisposable w.WriteSymbol(selectedDepth, this.transformSize[categoryDepth - 1][context]); } + /// + /// Gets the current fixed-point cost of one variable-transform partition decision. + /// + /// Indicates whether the current transform node is split. + /// The neighboring variable-transform context. + /// The rate cost in 1/512-bit units. + public int GetTransformPartitionCost(bool split, int context) + => Av1ProbabilityCost.GetSymbolCost(this.transformPartition[context], split ? 1 : 0); + + /// + /// Writes one variable-transform partition decision. + /// + /// Indicates whether the current transform node is split. + /// The neighboring variable-transform context. + public void WriteTransformPartition(bool split, int context) + { + ref Av1SymbolWriter w = ref this.writer; + w.WriteSymbol(split ? 1 : 0, this.transformPartition[context]); + } + private static int GetTransformSizeDepth( Av1BlockSize blockSize, Av1TransformSize transformSize, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderModeDecisionWorkspace.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderModeDecisionWorkspace.cs index 8e1c5b1bfc..8ec5f48f4f 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderModeDecisionWorkspace.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderModeDecisionWorkspace.cs @@ -4,6 +4,7 @@ using System.Runtime.InteropServices; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; +using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; @@ -19,6 +20,11 @@ internal readonly ref struct Av1EncoderModeDecisionWorkspace /// public const int MaximumSampleCount = 8 * 8; + /// + /// The number of 4x4 transform blocks covering one 8x8 coding block. + /// + public const int CandidateTransformBlockCount = 4; + /// /// The required workspace length in signed-integer storage elements. /// @@ -31,7 +37,11 @@ internal readonly ref struct Av1EncoderModeDecisionWorkspace private const int CandidateSampleStorageLength = 2 * MaximumSampleCount * sizeof(ushort) / sizeof(int); private const int CandidateCoefficientStorageOffset = CandidateSampleStorageOffset + CandidateSampleStorageLength; private const int CandidateCoefficientStorageLength = 2 * MaximumSampleCount; - private const int TransientStorageOffset = CandidateCoefficientStorageOffset + CandidateCoefficientStorageLength; + private const int CandidateTransformBlockStorageOffset = CandidateCoefficientStorageOffset + CandidateCoefficientStorageLength; + private const int CandidateTransformBlockStorageLength = CandidateTransformBlockCount; + private const int TransformContextStorageOffset = CandidateTransformBlockStorageOffset + CandidateTransformBlockStorageLength; + private const int TransformContextStorageLength = 1; + private const int TransientStorageOffset = TransformContextStorageOffset + TransformContextStorageLength; private const int ChromaFromLumaSampleCount = Av1ChromaFromLumaContext.BufferLine * 8; private const int ChromaFromLumaSampleStorageLength = ChromaFromLumaSampleCount * sizeof(short) / sizeof(int); private const int ChromaFromLumaBlueRateOffset = ChromaFromLumaSampleStorageLength; @@ -81,6 +91,19 @@ internal readonly ref struct Av1EncoderModeDecisionWorkspace public Av1EncoderPaletteWorkspace Palette => new(this.storage[TransientStorageOffset..]); + /// + /// Gets the transform state retained while evaluating a uniform 4x4 luma layout. + /// + public Span CandidateTransformBlocks + => MemoryMarshal.Cast( + this.storage.Slice(CandidateTransformBlockStorageOffset, CandidateTransformBlockStorageLength)); + + /// + /// Gets the two above and two left coefficient contexts used by a uniform 4x4 luma layout. + /// + public Span TransformContexts + => MemoryMarshal.AsBytes(this.storage.Slice(TransformContextStorageOffset, TransformContextStorageLength)); + /// /// Gets one reference edge including its common-corner prefix. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs index 8226b58fdd..07756edf65 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs @@ -119,7 +119,7 @@ internal static class Av1FrameEncoder ErrorResilientMode = true, RefreshFrameFlags = byte.MaxValue, DisableFrameEndUpdateCdf = true, - TransformMode = Av1TransformMode.Largest, + TransformMode = effort >= 6 ? Av1TransformMode.Select : Av1TransformMode.Largest, ModeInfoColumnCount = modeInfoColumnCount, ModeInfoRowCount = modeInfoRowCount, TilesInfo = tiles, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaPaletteModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaPaletteModeDecision.cs index 92836d0bcc..0d853ea876 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaPaletteModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaPaletteModeDecision.cs @@ -247,6 +247,7 @@ internal static partial class Av1IntraSuperblockEncoder bluePrediction, blueResidual, candidateBlueReconstruction, + transformSize.GetWidth(), candidateBlueCoefficients, transformSize, Av1TransformType.DctDct, @@ -265,6 +266,7 @@ internal static partial class Av1IntraSuperblockEncoder redPrediction, redResidual, candidateRedReconstruction, + transformSize.GetWidth(), candidateRedCoefficients, transformSize, Av1TransformType.DctDct, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.IntraBlockCopyModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.IntraBlockCopyModeDecision.cs index 1b34df4dcb..d311d3722d 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.IntraBlockCopyModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.IntraBlockCopyModeDecision.cs @@ -128,6 +128,23 @@ internal static partial class Av1IntraSuperblockEncoder BlockSize, LumaTransformSize); + int transformPartitionRate = 0; + if (this.picture.Parent.FrameHeader.TransformMode == Av1TransformMode.Select) + { + Av1NeighborArrayUnit transformContexts = this.picture.TransformFunctionContexts[tileIndex]; + int topIndex = transformContexts.GetTopIndex(blockOrigin); + int leftIndex = transformContexts.GetLeftIndex(blockOrigin); + int transformPartitionContext = Av1SymbolContextHelper.GetTransformPartitionContext( + transformContexts.Top[topIndex], + transformContexts.Left[leftIndex], + BlockSize, + LumaTransformSize); + + transformPartitionRate = writer.GetTransformPartitionCost( + false, + transformPartitionContext); + } + ObuColorConfig colorConfig = this.picture.Sequence.SequenceHeader.ColorConfig; int subsamplingX = colorConfig.SubSamplingX ? 1 : 0; int subsamplingY = colorConfig.SubSamplingY ? 1 : 0; @@ -250,6 +267,7 @@ internal static partial class Av1IntraSuperblockEncoder int candidateRate = writer.GetUseIntraBlockCopyCost(true) + displacementRate + writer.GetSkipCost(false, skipContext) + + transformPartitionRate + lumaRate + blueRate + redRate; @@ -385,6 +403,7 @@ internal static partial class Av1IntraSuperblockEncoder modeInfo.Block.Mode = Av1PredictionMode.DC; modeInfo.Block.UvMode = Av1ChromaPredictionMode.DC; + modeInfo.Block.TransformSize = LumaTransformSize; modeInfo.Block.Skip = selectedSkip; modeInfo.Block.UseIntraBlockCopy = true; block.FilterIntraMode = Av1FilterIntraMode.AllFilterIntraModes; @@ -466,6 +485,7 @@ internal static partial class Av1IntraSuperblockEncoder prediction[..sampleCount], residual[..sampleCount], transformReconstruction[..sampleCount], + transformSize.GetWidth(), transformCoefficients[..sampleCount], transformSize, transformType, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs index 12bc8c5a21..15459befb1 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs @@ -189,23 +189,31 @@ internal static partial class Av1IntraSuperblockEncoder int lumaTransformIndex = this.codedAreaLuma / Av1EncoderCoefficientBuffer.TransformBlockUnitCoefficientCount; - ref Av1EncoderTransformBlockState lumaState = ref lumaTransformBlocks[lumaTransformIndex]; + Span retainedLumaStates = lumaTransformBlocks[lumaTransformIndex..]; modeInfo.Block.Mode = this.SelectLumaMode( writer, macroBlock, blockOrigin, tileIndex, lumaCoefficients[this.codedAreaLuma..], - ref lumaState, + retainedLumaStates, ref paletteInfo, out int lumaAngleDelta, out Av1FilterIntraMode filterIntraMode, + out Av1TransformSize lumaTransformSize, out long lumaCost); block.PredictionUnit.AngleDelta[(int)Av1PlaneType.Y] = (sbyte)lumaAngleDelta; block.FilterIntraMode = filterIntraMode; + modeInfo.Block.TransformSize = lumaTransformSize; + + int lumaTransformBlockCount = LumaTransformSize.GetSize2d() / lumaTransformSize.GetSize2d(); + bool lumaTransformEmpty = true; + for (int transformIndex = 0; transformIndex < lumaTransformBlockCount; transformIndex++) + { + lumaTransformEmpty &= retainedLumaStates[transformIndex].EndOfBlock == 0; + } - bool lumaTransformEmpty = lumaState.EndOfBlock == 0; if (this.source.IsMonochrome) { int emptyTransformRate = lumaTransformEmpty @@ -215,8 +223,7 @@ internal static partial class Av1IntraSuperblockEncoder Av1ComponentType.Luminance, blockOrigin, BlockSize, - LumaTransformSize, - lumaState.TransformType, + lumaTransformSize, modeInfo.Block.Mode, block.FilterIntraMode) : 0; @@ -302,8 +309,7 @@ internal static partial class Av1IntraSuperblockEncoder Av1ComponentType.Luminance, blockOrigin, BlockSize, - LumaTransformSize, - lumaState.TransformType, + lumaTransformSize, modeInfo.Block.Mode, block.FilterIntraMode); @@ -314,7 +320,6 @@ internal static partial class Av1IntraSuperblockEncoder chromaOrigin, chromaBlockSize, chromaTransformSize, - blueState.TransformType, modeInfo.Block.Mode, Av1FilterIntraMode.AllFilterIntraModes); @@ -325,7 +330,6 @@ internal static partial class Av1IntraSuperblockEncoder chromaOrigin, chromaBlockSize, chromaTransformSize, - redState.TransformType, modeInfo.Block.Mode, Av1FilterIntraMode.AllFilterIntraModes); @@ -360,27 +364,73 @@ internal static partial class Av1IntraSuperblockEncoder Point blockOrigin, Av1BlockSize blockSize, Av1TransformSize transformSize, - Av1TransformType transformType, Av1PredictionMode lumaMode, Av1FilterIntraMode filterIntraMode) { - Av1TransformBlockContext blockContext = Av1TileWriter.GetTransformBlockContexts( - componentType, - coefficientNeighbors, - blockOrigin, - blockSize, - transformSize); + int blockWidth = blockSize.Get4x4WideCount(); + int blockHeight = blockSize.Get4x4HighCount(); + int transformWidth = transformSize.Get4x4WideCount(); + int transformHeight = transformSize.Get4x4HighCount(); + if (blockWidth == transformWidth && blockHeight == transformHeight) + { + Av1TransformBlockContext blockContext = Av1TileWriter.GetTransformBlockContexts( + componentType, + coefficientNeighbors, + blockOrigin, + blockSize, + transformSize); + + return writer.GetCoefficientCost( + transformSize, + Av1TransformType.DctDct, + lumaMode, + ReadOnlySpan.Empty, + componentType, + blockContext, + 0, + this.picture.Parent.FrameHeader.UseReducedTransformSet, + filterIntraMode); + } - return writer.GetCoefficientCost( - transformSize, - transformType, - lumaMode, - ReadOnlySpan.Empty, - componentType, - blockContext, - 0, - this.picture.Parent.FrameHeader.UseReducedTransformSet, - filterIntraMode); + Span contexts = this.blockWorkspace + .GetModeDecisionWorkspace() + .TransformContexts; + + Span topContexts = contexts[..blockWidth]; + Span leftContexts = contexts.Slice(blockWidth, blockHeight); + int topIndex = coefficientNeighbors.GetTopIndex(blockOrigin); + int leftIndex = coefficientNeighbors.GetLeftIndex(blockOrigin); + coefficientNeighbors.Top.Slice(topIndex, blockWidth).CopyTo(topContexts); + coefficientNeighbors.Left.Slice(leftIndex, blockHeight).CopyTo(leftContexts); + int rate = 0; + for (int blockRow = 0; blockRow < blockHeight; blockRow += transformHeight) + { + for (int blockColumn = 0; blockColumn < blockWidth; blockColumn += transformWidth) + { + Av1TransformBlockContext blockContext = Av1TileWriter.GetTransformBlockContexts( + componentType, + topContexts.Slice(blockColumn, transformWidth), + leftContexts.Slice(blockRow, transformHeight), + blockSize, + transformSize); + + rate += writer.GetCoefficientCost( + transformSize, + Av1TransformType.DctDct, + lumaMode, + ReadOnlySpan.Empty, + componentType, + blockContext, + 0, + this.picture.Parent.FrameHeader.UseReducedTransformSet, + filterIntraMode); + + topContexts.Slice(blockColumn, transformWidth).Clear(); + leftContexts.Slice(blockRow, transformHeight).Clear(); + } + } + + return rate; } private Av1PredictionMode SelectLumaMode( @@ -389,10 +439,11 @@ internal static partial class Av1IntraSuperblockEncoder Point blockOrigin, ushort tileIndex, Span retainedCoefficients, - ref Av1EncoderTransformBlockState retainedState, + Span retainedStates, ref Av1EncoderPaletteInfo paletteInfo, out int selectedAngleDelta, out Av1FilterIntraMode selectedFilterIntraMode, + out Av1TransformSize selectedTransformSize, out long selectedCost) { const Av1BlockSize BlockSize = Av1BlockSize.Block8x8; @@ -509,6 +560,16 @@ internal static partial class Av1IntraSuperblockEncoder BlockSize, TransformSize); + int transformSizeContext = Av1TileWriter.GetTransformSizeContext( + this.picture.TransformFunctionContexts[tileIndex], + macroBlock, + blockOrigin, + BlockSize); + + int largestTransformRate = this.picture.Parent.FrameHeader.TransformMode == Av1TransformMode.Select + ? writer.GetTransformSizeCost(BlockSize, TransformSize, transformSizeContext) + : 0; + int paletteDisabledCost = 0; if (this.picture.Parent.FrameHeader.AllowScreenContentTools) { @@ -586,6 +647,7 @@ internal static partial class Av1IntraSuperblockEncoder defaultTransformType, blockContext, paletteDisabledCost, + largestTransformRate, candidateReconstruction, candidateCoefficients, ref candidateState); @@ -600,7 +662,7 @@ internal static partial class Av1IntraSuperblockEncoder retainedCoefficients, TransformSize, candidateState, - ref retainedState); + ref retainedStates[0]); bestCost = candidateCost; bestMode = mode; @@ -635,6 +697,7 @@ internal static partial class Av1IntraSuperblockEncoder transformType, blockContext, paletteDisabledCost, + largestTransformRate, candidateReconstruction, candidateCoefficients, ref candidateState); @@ -649,7 +712,7 @@ internal static partial class Av1IntraSuperblockEncoder retainedCoefficients, TransformSize, candidateState, - ref retainedState); + ref retainedStates[0]); bestTransformCost = candidateCost; } @@ -699,6 +762,7 @@ internal static partial class Av1IntraSuperblockEncoder transformType, blockContext, paletteDisabledCost, + largestTransformRate, candidateReconstruction, candidateCoefficients, ref candidateState); @@ -713,7 +777,7 @@ internal static partial class Av1IntraSuperblockEncoder retainedCoefficients, TransformSize, candidateState, - ref retainedState); + ref retainedStates[0]); bestTransformCost = candidateCost; bestMode = Av1PredictionMode.DC; @@ -735,10 +799,11 @@ internal static partial class Av1IntraSuperblockEncoder tileIndex, transformSetType, blockContext, + largestTransformRate, candidateReconstruction, candidateCoefficients, retainedCoefficients, - ref retainedState, + ref retainedStates[0], ref bestTransformCost, ref paletteInfo)) { @@ -747,10 +812,413 @@ internal static partial class Av1IntraSuperblockEncoder selectedFilterIntraMode = Av1FilterIntraMode.AllFilterIntraModes; } + selectedTransformSize = TransformSize; + if (this.effort >= 6 && + this.picture.Parent.FrameHeader.TransformMode == Av1TransformMode.Select && + paletteInfo.PaletteSizes[0] == 0 && + selectedFilterIntraMode == Av1FilterIntraMode.AllFilterIntraModes) + { + long splitCost = this.GetSplitLumaCandidateCost( + writer, + macroBlock, + sourcePlane, + reconstructionPlane, + blockOrigin, + tileIndex, + bestMode, + selectedAngleDelta, + paletteDisabledCost, + transformSizeContext, + bestTransformCost, + candidateReconstruction, + candidateCoefficients, + workspace.CandidateTransformBlocks); + + if (splitCost < bestTransformCost) + { + CopySplitCandidate( + candidateReconstruction, + candidateCoefficients, + workspace.CandidateTransformBlocks, + reconstructionPlane, + blockOrigin, + retainedCoefficients, + retainedStates); + + bestTransformCost = splitCost; + selectedTransformSize = Av1TransformSize.Size4x4; + } + } + selectedCost = bestTransformCost; return bestMode; } + private long GetSplitLumaCandidateCost( + Av1SymbolEncoder writer, + Av1MacroBlockD macroBlock, + Buffer2DRegion sourcePlane, + Buffer2DRegion reconstructionPlane, + Point blockOrigin, + ushort tileIndex, + Av1PredictionMode mode, + int angleDelta, + int paletteDisabledCost, + int transformSizeContext, + long costLimit, + Span candidateReconstruction, + Span candidateCoefficients, + Span candidateTransformBlocks) + { + const Av1BlockSize BlockSize = Av1BlockSize.Block8x8; + const Av1TransformSize TransformSize = Av1TransformSize.Size4x4; + const int BlockWidth = 8; + const int TransformWidth = 4; + const int TransformSampleCount = TransformWidth * TransformWidth; + Av1EncoderModeDecisionWorkspace workspace = + this.blockWorkspace.GetModeDecisionWorkspace(); + + Span transformSamples = workspace.GetCandidateReconstruction(1); + Span prediction = transformSamples[..TransformSampleCount]; + Span transformReconstruction = transformSamples.Slice( + TransformSampleCount, + TransformSampleCount); + + Span transformCoefficients = workspace.GetCandidateCoefficients(1)[..TransformSampleCount]; + Span residual = workspace.FilterResidual[..TransformSampleCount]; + Span contexts = workspace.TransformContexts; + Span topContexts = contexts[..2]; + Span leftContexts = contexts[2..4]; + Av1NeighborArrayUnit coefficientNeighbors = + this.picture.LuminanceDcSignLevelCoefficientNeighbors[tileIndex]; + + int topIndex = coefficientNeighbors.GetTopIndex(blockOrigin); + int leftIndex = coefficientNeighbors.GetLeftIndex(blockOrigin); + coefficientNeighbors.Top.Slice(topIndex, 2).CopyTo(topContexts); + coefficientNeighbors.Left.Slice(leftIndex, 2).CopyTo(leftContexts); + bool useReducedTransformSet = this.picture.Parent.FrameHeader.UseReducedTransformSet; + Av1TransformSetType transformSetType = Av1SymbolContextHelper.GetExtendedTransformSetType( + TransformSize, + useReducedTransformSet); + + int rate = Av1TileWriter.GetLumaModeCost(writer, macroBlock, BlockSize, mode, angleDelta); + rate += writer.GetTransformSizeCost(BlockSize, TransformSize, transformSizeContext); + if (mode == Av1PredictionMode.DC) + { + rate += paletteDisabledCost; + if (this.picture.Sequence.SequenceHeader.EnableFilterIntra) + { + rate += writer.GetFilterIntraModeCost( + Av1FilterIntraMode.AllFilterIntraModes, + BlockSize); + } + } + + long distortion = 0; + for (int transformRow = 0; transformRow < 2; transformRow++) + { + for (int transformColumn = 0; transformColumn < 2; transformColumn++) + { + int transformIndex = (transformRow * 2) + transformColumn; + int reconstructionOffset = + (transformRow * TransformWidth * BlockWidth) + (transformColumn * TransformWidth); + + Point transformOrigin = blockOrigin + new Size( + transformColumn * TransformWidth, + transformRow * TransformWidth); + + Span aboveStorage = workspace.GetReferenceSamples(0); + Span leftStorage = workspace.GetReferenceSamples(1); + this.PrepareSplitLumaReferenceSamples( + reconstructionPlane, + blockOrigin, + macroBlock, + transformRow, + transformColumn, + candidateReconstruction, + aboveStorage, + leftStorage, + out bool hasLeft, + out bool hasAbove); + + TOperator.PrepareIntra( + this.blockWorkspace, + sourcePlane, + transformOrigin, + prediction, + aboveStorage.Slice(1, TransformWidth * 2), + leftStorage.Slice(1, TransformWidth * 2), + hasLeft, + hasAbove, + mode, + angleDelta, + residual, + TransformSize, + this.bitDepth); + + Av1TransformBlockContext blockContext = Av1TileWriter.GetTransformBlockContexts( + Av1ComponentType.Luminance, + topContexts.Slice(transformColumn, 1), + leftContexts.Slice(transformRow, 1), + BlockSize, + TransformSize); + + long bestTransformCost = long.MaxValue; + Av1TransformType bestTransformType = Av1TransformType.DctDct; + int bestTransformRate = 0; + long bestTransformDistortion = 0; + Av1EncoderTransformBlockState bestTransformState = default; + Span retainedTransformCoefficients = candidateCoefficients.Slice( + transformIndex * TransformSampleCount, + TransformSampleCount); + + for (Av1TransformType transformType = Av1TransformType.DctDct; + transformType < Av1TransformType.AllTransformTypes; + transformType++) + { + if (!transformType.IsExtendedSetUsed(transformSetType)) + { + continue; + } + + Av1EncoderTransformBlockState candidateState = default; + long candidateDistortion = TOperator.EncodePredictionCandidate( + this.blockWorkspace, + sourcePlane, + transformOrigin, + prediction, + residual, + transformReconstruction, + TransformWidth, + transformCoefficients, + TransformSize, + transformType, + Av1Plane.Y, + this.quantization.QIndex[0], + this.quantization.DeltaQDc[(int)Av1Plane.Y], + this.quantization.DeltaQAc[(int)Av1Plane.Y], + this.bitDepth, + ref candidateState); + + int candidateRate = writer.GetCoefficientCost( + TransformSize, + transformType, + mode, + transformCoefficients, + Av1ComponentType.Luminance, + blockContext, + candidateState.EndOfBlock, + useReducedTransformSet, + Av1FilterIntraMode.AllFilterIntraModes); + + long candidateCost = Av1RateDistortion.GetCost( + this.rateMultiplier, + candidateRate, + candidateDistortion); + + if (candidateCost < bestTransformCost) + { + // Preserve the improving 4x4 trial in the block mosaic. Later transform predictions + // consume that reconstruction, and copying the compact result avoids another transform. + transformCoefficients.CopyTo(retainedTransformCoefficients); + for (int row = 0; row < TransformWidth; row++) + { + transformReconstruction.Slice(row * TransformWidth, TransformWidth) + .CopyTo( + candidateReconstruction.Slice( + reconstructionOffset + (row * BlockWidth), + TransformWidth)); + } + + bestTransformCost = candidateCost; + bestTransformType = transformType; + bestTransformRate = candidateRate; + bestTransformDistortion = candidateDistortion; + bestTransformState = candidateState; + } + } + + rate += bestTransformRate; + distortion += bestTransformDistortion; + candidateTransformBlocks[transformIndex] = bestTransformState; + byte coefficientContext = Av1SymbolContextHelper.GetCoefficientContext( + retainedTransformCoefficients, + TransformSize, + bestTransformType, + bestTransformState.EndOfBlock); + + topContexts[transformColumn] = coefficientContext; + leftContexts[transformRow] = coefficientContext; + + // Every remaining transform can only add nonnegative rate and distortion. + if (Av1RateDistortion.GetCost(this.rateMultiplier, rate, distortion) >= costLimit) + { + return long.MaxValue; + } + } + } + + return Av1RateDistortion.GetCost(this.rateMultiplier, rate, distortion); + } + + private void PrepareSplitLumaReferenceSamples( + Buffer2DRegion reconstructionPlane, + Point blockOrigin, + Av1MacroBlockD macroBlock, + int transformRow, + int transformColumn, + ReadOnlySpan candidateReconstruction, + Span aboveStorage, + Span leftStorage, + out bool hasLeft, + out bool hasAbove) + { + const Av1BlockSize BlockSize = Av1BlockSize.Block8x8; + const Av1TransformSize TransformSize = Av1TransformSize.Size4x4; + const int BlockWidth = 8; + const int TransformWidth = 4; + int rowOffset = transformRow * TransformWidth; + int columnOffset = transformColumn * TransformWidth; + int modeInfoRow = blockOrigin.Y >> Av1Constants.ModeInfoSizeLog2; + int modeInfoColumn = blockOrigin.X >> Av1Constants.ModeInfoSizeLog2; + hasAbove = transformRow > 0 || macroBlock.IsUpAvailable; + hasLeft = transformColumn > 0 || macroBlock.IsLeftAvailable; + bool rightAvailable = + modeInfoColumn + transformColumn + TransformSize.Get4x4WideCount() < + macroBlock.Tile.ModeInfoColumnEnd; + + bool bottomAvailable = + modeInfoRow + transformRow + TransformSize.Get4x4HighCount() < + macroBlock.Tile.ModeInfoRowEnd; + + bool hasTopRight = Av1IntraReferenceAvailability.HasTopRight( + this.picture.Sequence.SequenceHeader.SuperblockSize, + BlockSize, + modeInfoRow, + modeInfoColumn, + hasAbove, + rightAvailable, + Av1PartitionType.None, + TransformSize, + transformRow, + transformColumn, + 0, + 0); + + bool hasBottomLeft = Av1IntraReferenceAvailability.HasBottomLeft( + this.picture.Sequence.SequenceHeader.SuperblockSize, + BlockSize, + modeInfoRow, + modeInfoColumn, + bottomAvailable, + hasLeft, + Av1PartitionType.None, + TransformSize, + transformRow, + transformColumn, + 0, + 0); + + Span above = aboveStorage.Slice(1, TransformWidth * 2); + Span left = leftStorage.Slice(1, TransformWidth * 2); + if (hasAbove) + { + if (transformRow > 0) + { + candidateReconstruction + .Slice(((rowOffset - 1) * BlockWidth) + columnOffset, TransformWidth) + .CopyTo(above); + } + else + { + reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y - 1) + .Slice(blockOrigin.X + columnOffset, TransformWidth) + .CopyTo(above); + } + } + + if (hasLeft) + { + if (transformColumn > 0) + { + for (int row = 0; row < TransformWidth; row++) + { + left[row] = candidateReconstruction[((rowOffset + row) * BlockWidth) + columnOffset - 1]; + } + } + else + { + for (int row = 0; row < TransformWidth; row++) + { + left[row] = reconstructionPlane + .DangerousGetRowSpan(blockOrigin.Y + rowOffset + row)[blockOrigin.X - 1]; + } + } + } + + int midpoint = 128 << (this.bitDepth.GetBitCount() - 8); + if (!hasAbove) + { + above[..TransformWidth].Fill(hasLeft ? left[0] : TOperator.CreateSample(midpoint - 1)); + } + + if (!hasLeft) + { + left[..TransformWidth].Fill(hasAbove ? above[0] : TOperator.CreateSample(midpoint + 1)); + } + + if (hasTopRight) + { + if (transformRow > 0) + { + candidateReconstruction + .Slice( + ((rowOffset - 1) * BlockWidth) + columnOffset + TransformWidth, + TransformWidth) + .CopyTo(above[TransformWidth..]); + } + else + { + reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y - 1) + .Slice(blockOrigin.X + columnOffset + TransformWidth, TransformWidth) + .CopyTo(above[TransformWidth..]); + } + } + else + { + above[TransformWidth..].Fill(above[TransformWidth - 1]); + } + + if (hasBottomLeft) + { + for (int row = TransformWidth; row < TransformWidth * 2; row++) + { + left[row] = reconstructionPlane + .DangerousGetRowSpan(blockOrigin.Y + rowOffset + row)[blockOrigin.X - 1]; + } + } + else + { + left[TransformWidth..].Fill(left[TransformWidth - 1]); + } + + // Only an interior transform corner belongs to decision scratch. Boundary corners continue + // to read the already reconstructed neighboring block so candidate trials remain isolated. + TSample corner = hasAbove && hasLeft + ? transformRow > 0 && transformColumn > 0 + ? candidateReconstruction[((rowOffset - 1) * BlockWidth) + columnOffset - 1] + : reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y + rowOffset - 1)[ + blockOrigin.X + columnOffset - 1] + : hasAbove + ? above[0] + : hasLeft + ? left[0] + : TOperator.CreateSample(midpoint); + + aboveStorage[0] = corner; + leftStorage[0] = corner; + } + private long GetLumaCandidateCost( Av1SymbolEncoder writer, Av1MacroBlockD macroBlock, @@ -765,6 +1233,7 @@ internal static partial class Av1IntraSuperblockEncoder Av1TransformType transformType, Av1TransformBlockContext blockContext, int paletteDisabledCost, + int transformSizeRate, Span candidateReconstruction, Span candidateCoefficients, ref Av1EncoderTransformBlockState candidateState) @@ -793,6 +1262,7 @@ internal static partial class Av1IntraSuperblockEncoder ref candidateState); int rate = Av1TileWriter.GetLumaModeCost(writer, macroBlock, BlockSize, mode, angleDelta); + rate += transformSizeRate; if (mode == Av1PredictionMode.DC) { rate += paletteDisabledCost; @@ -828,6 +1298,7 @@ internal static partial class Av1IntraSuperblockEncoder Av1TransformType transformType, Av1TransformBlockContext blockContext, int paletteDisabledCost, + int transformSizeRate, Span candidateReconstruction, Span candidateCoefficients, ref Av1EncoderTransformBlockState candidateState) @@ -841,6 +1312,7 @@ internal static partial class Av1IntraSuperblockEncoder prediction, residual, candidateReconstruction, + TransformSize.GetWidth(), candidateCoefficients, TransformSize, transformType, @@ -858,6 +1330,7 @@ internal static partial class Av1IntraSuperblockEncoder Av1PredictionMode.DC, 0); + rate += transformSizeRate; rate += paletteDisabledCost; rate += writer.GetFilterIntraModeCost(filterIntraMode, BlockSize); rate += writer.GetCoefficientCost( @@ -895,5 +1368,30 @@ internal static partial class Av1IntraSuperblockEncoder retainedState = candidateState; } + + private static void CopySplitCandidate( + ReadOnlySpan candidateReconstruction, + ReadOnlySpan candidateCoefficients, + ReadOnlySpan candidateTransformBlocks, + Buffer2DRegion reconstructionPlane, + Point blockOrigin, + Span retainedCoefficients, + Span retainedTransformBlocks) + { + const int BlockWidth = 8; + const int SampleCount = BlockWidth * BlockWidth; + candidateCoefficients[..SampleCount].CopyTo(retainedCoefficients); + candidateTransformBlocks[..Av1EncoderModeDecisionWorkspace.CandidateTransformBlockCount] + .CopyTo(retainedTransformBlocks); + + for (int row = 0; row < BlockWidth; row++) + { + candidateReconstruction.Slice(row * BlockWidth, BlockWidth) + .CopyTo( + reconstructionPlane + .DangerousGetRowSpan(blockOrigin.Y + row) + .Slice(blockOrigin.X, BlockWidth)); + } + } } } diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs index 80ddb3ca35..e3108325d0 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs @@ -191,6 +191,37 @@ internal static partial class Av1IntraSuperblockEncoder Av1BitDepth bitDepth, ref Av1EncoderTransformBlockState state); + /// + /// Builds one spatial intra prediction and its source residual for reuse across transform candidates. + /// + /// The reusable block workspace. + /// The coded source plane. + /// The transform-block origin in plane samples. + /// The contiguous prediction destination. + /// The top reference samples, with prefix storage for the shared corner. + /// The left reference samples. + /// Whether the left reference is available. + /// Whether the top reference is available. + /// The intra prediction mode. + /// The signed directional-angle adjustment. + /// The contiguous source-minus-prediction destination. + /// The prediction dimensions. + /// The coded sample bit depth. + public static abstract void PrepareIntra( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span prediction, + ReadOnlySpan above, + ReadOnlySpan left, + bool hasLeft, + bool hasAbove, + Av1PredictionMode mode, + int angleDelta, + Span residual, + Av1TransformSize transformSize, + Av1BitDepth bitDepth); + /// /// Builds one filter-intra prediction for reuse across transform candidates. /// @@ -247,7 +278,8 @@ internal static partial class Av1IntraSuperblockEncoder /// The transform-block origin in plane samples. /// The contiguous prediction samples. /// The contiguous source-minus-prediction samples. - /// The contiguous candidate reconstruction. + /// The candidate reconstruction. + /// The number of reconstruction samples between rows. /// The candidate entropy-coding coefficients. /// The transform dimensions. /// The compound transform applied to the residual. @@ -265,6 +297,7 @@ internal static partial class Av1IntraSuperblockEncoder ReadOnlySpan prediction, ReadOnlySpan residual, Span reconstruction, + int reconstructionStride, Span quantizedCoefficients, Av1TransformSize transformSize, Av1TransformType transformType, @@ -704,6 +737,36 @@ internal static partial class Av1IntraSuperblockEncoder plane, ref state); + /// + public static void PrepareIntra( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span prediction, + ReadOnlySpan above, + ReadOnlySpan left, + bool hasLeft, + bool hasAbove, + Av1PredictionMode mode, + int angleDelta, + Span residual, + Av1TransformSize transformSize, + Av1BitDepth bitDepth) + => Av1TransformBlockEncoder.PrepareIntraPrediction( + workspace, + Av1TransformBlockEncoder.GetPlaneSpan(source, blockOrigin), + source.Stride, + prediction, + transformSize.GetWidth(), + above, + left, + hasLeft, + hasAbove, + mode, + angleDelta, + residual, + transformSize); + /// public static void PrepareFilterIntra( Av1EncoderBlockWorkspace workspace, @@ -782,6 +845,7 @@ internal static partial class Av1IntraSuperblockEncoder ReadOnlySpan prediction, ReadOnlySpan residual, Span reconstruction, + int reconstructionStride, Span quantizedCoefficients, Av1TransformSize transformSize, Av1TransformType transformType, @@ -798,6 +862,7 @@ internal static partial class Av1IntraSuperblockEncoder prediction, residual, reconstruction, + reconstructionStride, quantizedCoefficients, transformSize, transformType, @@ -1213,6 +1278,37 @@ internal static partial class Av1IntraSuperblockEncoder bitDepth, ref state); + /// + public static void PrepareIntra( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span prediction, + ReadOnlySpan above, + ReadOnlySpan left, + bool hasLeft, + bool hasAbove, + Av1PredictionMode mode, + int angleDelta, + Span residual, + Av1TransformSize transformSize, + Av1BitDepth bitDepth) + => Av1TransformBlockEncoder.PrepareIntraPrediction( + workspace, + Av1TransformBlockEncoder.GetPlaneSpan(source, blockOrigin), + source.Stride, + prediction, + transformSize.GetWidth(), + above, + left, + hasLeft, + hasAbove, + mode, + angleDelta, + residual, + transformSize, + bitDepth); + /// public static void PrepareFilterIntra( Av1EncoderBlockWorkspace workspace, @@ -1299,6 +1395,7 @@ internal static partial class Av1IntraSuperblockEncoder ReadOnlySpan prediction, ReadOnlySpan residual, Span reconstruction, + int reconstructionStride, Span quantizedCoefficients, Av1TransformSize transformSize, Av1TransformType transformType, @@ -1315,6 +1412,7 @@ internal static partial class Av1IntraSuperblockEncoder prediction, residual, reconstruction, + reconstructionStride, quantizedCoefficients, transformSize, transformType, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.PaletteModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.PaletteModeDecision.cs index 6b617ea6af..937847f5c5 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.PaletteModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.PaletteModeDecision.cs @@ -28,6 +28,7 @@ internal static partial class Av1IntraSuperblockEncoder ushort tileIndex, Av1TransformSetType transformSetType, Av1TransformBlockContext blockContext, + int transformSizeRate, Span candidateReconstruction, Span candidateCoefficients, Span retainedCoefficients, @@ -138,6 +139,7 @@ internal static partial class Av1IntraSuperblockEncoder blockOrigin, transformSetType, blockContext, + transformSizeRate, samples, rows, columns, @@ -167,6 +169,7 @@ internal static partial class Av1IntraSuperblockEncoder blockOrigin, transformSetType, blockContext, + transformSizeRate, samples, rows, columns, @@ -205,6 +208,7 @@ internal static partial class Av1IntraSuperblockEncoder blockOrigin, transformSetType, blockContext, + transformSizeRate, samples, rows, columns, @@ -243,6 +247,7 @@ internal static partial class Av1IntraSuperblockEncoder Point blockOrigin, Av1TransformSetType transformSetType, Av1TransformBlockContext blockContext, + int transformSizeRate, ReadOnlySpan samples, int rows, int columns, @@ -345,6 +350,7 @@ internal static partial class Av1IntraSuperblockEncoder Av1PredictionMode.DC, 0); + rate += transformSizeRate; rate += writer.GetPaletteYModeCost(true, blockSizeContext, neighborContext); rate += writer.GetPaletteSizeCost(paletteSize, blockSizeContext, Av1PlaneType.Y); rate += Av1SymbolEncoder.GetPaletteYColorCost(colorCache, paletteColors, bitDepth); @@ -372,6 +378,7 @@ internal static partial class Av1IntraSuperblockEncoder prediction, residual, candidateReconstruction, + TransformSize.GetWidth(), candidateCoefficients, TransformSize, transformType, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs index b70ade1a11..ddcaceb4e3 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs @@ -166,7 +166,8 @@ internal static class Av1TransformBlockEncoder /// The block origin in plane samples. /// The contiguous prediction samples. /// The contiguous source-minus-prediction samples. - /// The contiguous candidate reconstruction. + /// The candidate reconstruction. + /// The number of reconstruction samples between rows. /// The candidate entropy-coding coefficients. /// The selected transform dimensions. /// The selected compound transform type. @@ -183,6 +184,7 @@ internal static class Av1TransformBlockEncoder ReadOnlySpan prediction, ReadOnlySpan residual, Span reconstruction, + int reconstructionStride, Span quantizedCoefficients, Av1TransformSize transformSize, Av1TransformType transformType, @@ -198,7 +200,12 @@ internal static class Av1TransformBlockEncoder ReadOnlySpan sourceSamples = GetPlaneSpan(source, blockOrigin); // Each transform trial mutates reconstruction and residual scratch, so restore both prepared inputs. - prediction[..sampleCount].CopyTo(reconstruction); + // Row copies preserve a larger candidate surface without materializing a second compact block. + for (int row = 0; row < height; row++) + { + prediction.Slice(row * width, width).CopyTo(reconstruction.Slice(row * reconstructionStride, width)); + } + residual[..sampleCount].CopyTo(workspace.Residual); EncodeLossy( workspace, @@ -216,7 +223,7 @@ internal static class Av1TransformBlockEncoder Av1InverseTransformer.Reconstruct8Bit( workspace.DequantizedCoefficients, reconstruction, - width, + reconstructionStride, transformSize, transformType, (int)plane, @@ -229,7 +236,7 @@ internal static class Av1TransformBlockEncoder sourceSamples, source.Stride, reconstruction, - width, + reconstructionStride, workspace.Residual, width, width, @@ -490,7 +497,8 @@ internal static class Av1TransformBlockEncoder /// The block origin in plane samples. /// The contiguous prediction samples. /// The contiguous source-minus-prediction samples. - /// The contiguous candidate reconstruction. + /// The candidate reconstruction. + /// The number of reconstruction samples between rows. /// The candidate entropy-coding coefficients. /// The selected transform dimensions. /// The selected compound transform type. @@ -508,6 +516,7 @@ internal static class Av1TransformBlockEncoder ReadOnlySpan prediction, ReadOnlySpan residual, Span reconstruction, + int reconstructionStride, Span quantizedCoefficients, Av1TransformSize transformSize, Av1TransformType transformType, @@ -524,7 +533,12 @@ internal static class Av1TransformBlockEncoder ReadOnlySpan sourceSamples = GetPlaneSpan(source, blockOrigin); // Each transform trial mutates reconstruction and residual scratch, so restore both prepared inputs. - prediction[..sampleCount].CopyTo(reconstruction); + // Row copies preserve a larger candidate surface without materializing a second compact block. + for (int row = 0; row < height; row++) + { + prediction.Slice(row * width, width).CopyTo(reconstruction.Slice(row * reconstructionStride, width)); + } + residual[..sampleCount].CopyTo(workspace.Residual); EncodeLossy( workspace, @@ -542,7 +556,7 @@ internal static class Av1TransformBlockEncoder Av1InverseTransformer.ReconstructHighBitDepth( workspace.DequantizedCoefficients, MemoryMarshal.Cast(reconstruction), - width, + reconstructionStride, transformSize, transformType, (int)plane, @@ -556,7 +570,7 @@ internal static class Av1TransformBlockEncoder sourceSamples, source.Stride, reconstruction, - width, + reconstructionStride, workspace.Residual, width, width, @@ -678,65 +692,54 @@ internal static class Av1TransformBlockEncoder } /// - /// Encodes and reconstructs one eight-bit lossy intra block. + /// Builds an eight-bit intra prediction and its compact source residual. /// - /// The reusable residual, coefficient, and transform storage. + /// The reusable prediction scratch. /// The source samples. /// The number of source samples between rows. - /// The reconstructed frame samples and prediction destination. - /// The number of reconstruction samples between rows. + /// The prediction destination. + /// The number of prediction samples between rows. /// The contiguous top reference samples, with prefix storage for the shared corner. /// The contiguous left reference samples. /// Whether the left reference is available. /// Whether the top reference is available. /// The intra prediction mode. /// The signed directional-angle adjustment. - /// The retained entropy-coding coefficients. - /// The selected transform dimensions. - /// The selected compound transform type. - /// The segment quantizer index. - /// The plane DC quantizer adjustment. - /// The plane AC quantizer adjustment. - /// The component plane containing the block. - /// The retained transform type and end-of-block syntax. - private static void EncodeIntraLossyContiguous( + /// The compact source-minus-prediction destination. + /// The prediction dimensions. + public static void PrepareIntraPrediction( Av1EncoderBlockWorkspace workspace, ReadOnlySpan source, int sourceStride, - Span reconstruction, - int reconstructionStride, + Span prediction, + int predictionStride, ReadOnlySpan above, ReadOnlySpan left, bool hasLeft, bool hasAbove, Av1PredictionMode mode, int angleDelta, - Span quantizedCoefficients, - Av1TransformSize transformSize, - Av1TransformType transformType, - int qIndex, - int dcDeltaQ, - int acDeltaQ, - Av1Plane plane, - ref Av1EncoderTransformBlockState state) + Span residual, + Av1TransformSize transformSize) { int width = transformSize.GetWidth(); int height = transformSize.GetHeight(); - // Prediction and subtraction stay in their SIMD-first operators while this method owns the required block-stage ordering. + // Prediction remains in the specialized SIMD-first kernels. This boundary only shares the prepared + // samples and residual across transform trials that differ in transform size or type. if (mode == Av1PredictionMode.DC) { - Av1DcIntraPredictor.Predict(hasLeft, hasAbove, reconstruction, reconstructionStride, above, left, width, height); + Av1DcIntraPredictor.Predict(hasLeft, hasAbove, prediction, predictionStride, above, left, width, height); } else if (mode.IsDirectional()) { - // The current encoder disables intra-edge filtering in sequence syntax. Reusing transform workspace for - // zone-three transposition keeps directional prediction allocation-free before the transform overwrites it. + // The current encoder disables intra-edge filtering in sequence syntax. Zone-three transposition + // borrows transform scratch because prediction completes before forward transformation starts. Span directionalScratch = MemoryMarshal.AsBytes(workspace.TransformWorkspace)[..(width * height)]; Av1DirectionalIntraPredictor.Predict( - reconstruction, - reconstructionStride, + prediction, + predictionStride, transformSize, above, left, @@ -748,10 +751,163 @@ internal static class Av1TransformBlockEncoder else { Av1NonDirectionalIntraPredictorBase.GetPredictor(mode) - .Predict(reconstruction, reconstructionStride, above, left, width, height); + .Predict(prediction, predictionStride, above, left, width, height); } - Av1ResidualBuilder.Subtract(source, sourceStride, reconstruction, reconstructionStride, workspace.Residual, width, width, height); + Av1ResidualBuilder.Subtract( + source, + sourceStride, + prediction, + predictionStride, + residual, + width, + width, + height); + } + + /// + /// Builds a high-bit-depth intra prediction and its compact source residual. + /// + /// The reusable prediction scratch. + /// The source samples. + /// The number of source samples between rows. + /// The prediction destination. + /// The number of prediction samples between rows. + /// The contiguous top reference samples, with prefix storage for the shared corner. + /// The contiguous left reference samples. + /// Whether the left reference is available. + /// Whether the top reference is available. + /// The intra prediction mode. + /// The signed directional-angle adjustment. + /// The compact source-minus-prediction destination. + /// The prediction dimensions. + /// The coded sample bit depth. + public static void PrepareIntraPrediction( + Av1EncoderBlockWorkspace workspace, + ReadOnlySpan source, + int sourceStride, + Span prediction, + int predictionStride, + ReadOnlySpan above, + ReadOnlySpan left, + bool hasLeft, + bool hasAbove, + Av1PredictionMode mode, + int angleDelta, + Span residual, + Av1TransformSize transformSize, + Av1BitDepth bitDepth) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + + // Valid high-bit-depth samples remain below the sign bit, so the predictor kernels can share + // the unsigned frame storage with their signed transform-domain implementation. + Span signedPrediction = MemoryMarshal.Cast(prediction); + ReadOnlySpan signedAbove = MemoryMarshal.Cast(above); + ReadOnlySpan signedLeft = MemoryMarshal.Cast(left); + if (mode == Av1PredictionMode.DC) + { + Av1DcIntraPredictor.Predict( + hasLeft, + hasAbove, + signedPrediction, + predictionStride, + signedAbove, + signedLeft, + width, + height, + bitDepth.GetBitCount()); + } + else if (mode.IsDirectional()) + { + Span directionalScratch = MemoryMarshal.Cast(workspace.TransformWorkspace)[..(width * height)]; + + Av1DirectionalIntraPredictor.Predict( + signedPrediction, + predictionStride, + transformSize, + signedAbove, + signedLeft, + false, + false, + mode.ToAngle() + (angleDelta * Av1Constants.AngleStep), + directionalScratch); + } + else + { + Av1NonDirectionalIntraPredictorBase.GetPredictor(mode) + .Predict(signedPrediction, predictionStride, signedAbove, signedLeft, width, height); + } + + Av1ResidualBuilder.Subtract( + source, + sourceStride, + prediction, + predictionStride, + residual, + width, + width, + height); + } + + /// + /// Encodes and reconstructs one eight-bit lossy intra block. + /// + /// The reusable residual, coefficient, and transform storage. + /// The source samples. + /// The number of source samples between rows. + /// The reconstructed frame samples and prediction destination. + /// The number of reconstruction samples between rows. + /// The contiguous top reference samples, with prefix storage for the shared corner. + /// The contiguous left reference samples. + /// Whether the left reference is available. + /// Whether the top reference is available. + /// The intra prediction mode. + /// The signed directional-angle adjustment. + /// The retained entropy-coding coefficients. + /// The selected transform dimensions. + /// The selected compound transform type. + /// The segment quantizer index. + /// The plane DC quantizer adjustment. + /// The plane AC quantizer adjustment. + /// The component plane containing the block. + /// The retained transform type and end-of-block syntax. + private static void EncodeIntraLossyContiguous( + Av1EncoderBlockWorkspace workspace, + ReadOnlySpan source, + int sourceStride, + Span reconstruction, + int reconstructionStride, + ReadOnlySpan above, + ReadOnlySpan left, + bool hasLeft, + bool hasAbove, + Av1PredictionMode mode, + int angleDelta, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1Plane plane, + ref Av1EncoderTransformBlockState state) + { + PrepareIntraPrediction( + workspace, + source, + sourceStride, + reconstruction, + reconstructionStride, + above, + left, + hasLeft, + hasAbove, + mode, + angleDelta, + workspace.Residual, + transformSize); EncodeLossy( workspace, @@ -825,56 +981,21 @@ internal static class Av1TransformBlockEncoder Av1BitDepth bitDepth, ref Av1EncoderTransformBlockState state) { - int width = transformSize.GetWidth(); - int height = transformSize.GetHeight(); - - // Valid high-bit-depth samples remain below the sign bit, so signed transform lanes can share the unsigned frame storage. - Span signedReconstruction = MemoryMarshal.Cast(reconstruction); - ReadOnlySpan signedAbove = MemoryMarshal.Cast(above); - ReadOnlySpan signedLeft = MemoryMarshal.Cast(left); - if (mode == Av1PredictionMode.DC) - { - Av1DcIntraPredictor.Predict( - hasLeft, - hasAbove, - signedReconstruction, - reconstructionStride, - signedAbove, - signedLeft, - width, - height, - bitDepth.GetBitCount()); - } - else if (mode.IsDirectional()) - { - Span directionalScratch = MemoryMarshal.Cast(workspace.TransformWorkspace)[..(width * height)]; - - Av1DirectionalIntraPredictor.Predict( - signedReconstruction, - reconstructionStride, - transformSize, - signedAbove, - signedLeft, - false, - false, - mode.ToAngle() + (angleDelta * Av1Constants.AngleStep), - directionalScratch); - } - else - { - Av1NonDirectionalIntraPredictorBase.GetPredictor(mode) - .Predict(signedReconstruction, reconstructionStride, signedAbove, signedLeft, width, height); - } - - Av1ResidualBuilder.Subtract( + PrepareIntraPrediction( + workspace, source, sourceStride, reconstruction, reconstructionStride, + above, + left, + hasLeft, + hasAbove, + mode, + angleDelta, workspace.Residual, - width, - width, - height); + transformSize, + bitDepth); EncodeLossy( workspace, @@ -892,7 +1013,7 @@ internal static class Av1TransformBlockEncoder // Reconstructing the quantized result makes later predictions use exactly the samples a decoder will reproduce. Av1InverseTransformer.ReconstructHighBitDepth( workspace.DequantizedCoefficients, - signedReconstruction, + MemoryMarshal.Cast(reconstruction), reconstructionStride, transformSize, transformType, diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs index 234c245797..58a5c5d4b5 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs @@ -910,6 +910,50 @@ internal partial class Av1TileWriter UpdateNeighbors(pcs, entropyCodingContext, blockOrigin, ref blk_ptr, tile_idx, blockSize); } + /// + /// Derives the uniform intra transform-size context from the current above and left edges. + /// + /// The retained transform widths and heights. + /// The reusable macroblock edge and neighbor state. + /// The block origin in samples. + /// The block size defining the maximum transform. + /// The uniform transform-size context. + public static int GetTransformSizeContext( + Av1NeighborArrayUnit transformContexts, + Av1MacroBlockD macroBlock, + Point blockOrigin, + Av1BlockSize blockSize) + { + Av1TransformSize maximumTransformSize = blockSize.GetMaximumTransformSize(); + int above = transformContexts.Top[transformContexts.GetTopIndex(blockOrigin)] >= maximumTransformSize.GetWidth() ? 1 : 0; + int left = transformContexts.Left[transformContexts.GetLeftIndex(blockOrigin)] >= maximumTransformSize.GetHeight() ? 1 : 0; + + // Inter neighbors contribute their coding-block extent, not their residual-transform extent. + if (macroBlock.IsUpAvailable) + { + ref Av1MacroBlockModeInfo aboveModeInfo = + ref macroBlock.GetRelativeModeInfo(-macroBlock.ModeInfoStride); + + if (aboveModeInfo.Block.UseIntraBlockCopy) + { + above = aboveModeInfo.Block.BlockSize.GetWidth() >= maximumTransformSize.GetWidth() ? 1 : 0; + } + } + + if (macroBlock.IsLeftAvailable) + { + ref Av1MacroBlockModeInfo leftModeInfo = ref macroBlock.GetRelativeModeInfo(-1); + if (leftModeInfo.Block.UseIntraBlockCopy) + { + left = leftModeInfo.Block.BlockSize.GetHeight() >= maximumTransformSize.GetHeight() ? 1 : 0; + } + } + + return macroBlock.IsUpAvailable + ? macroBlock.IsLeftAvailable ? above + left : above + : macroBlock.IsLeftAvailable ? left : 0; + } + /// /// Writes or derives the block transform size and publishes its edge contexts. /// @@ -931,28 +975,42 @@ internal partial class Av1TileWriter { ObuFrameHeader frameHeader = pcs.Parent.FrameHeader; bool isLossless = frameHeader.LosslessArray[macroBlockModeInfo.Block.SegmentId]; - bool writesTransformSize = !isLossless && + bool isInter = macroBlockModeInfo.Block.UseIntraBlockCopy; + bool writesUniformTransformSize = !isLossless && frameHeader.TransformMode == Av1TransformMode.Select && + !isInter && blockSize > Av1BlockSize.Block4x4; + bool writesVariableTransformSize = !isLossless && + frameHeader.TransformMode == Av1TransformMode.Select && + isInter && + !macroBlockModeInfo.Block.Skip; + Av1TransformSize transformSize = isLossless ? Av1TransformSize.Size4x4 - : writesTransformSize + : writesUniformTransformSize || writesVariableTransformSize ? macroBlockModeInfo.Block.TransformSize : blockSize.GetMaximumTransformSize(); macroBlockModeInfo.Block.TransformSize = transformSize; Av1NeighborArrayUnit transformContexts = pcs.TransformFunctionContexts[tileIndex]; - if (writesTransformSize) + if (writesUniformTransformSize) { - Av1TransformSize maximumTransformSize = blockSize.GetMaximumTransformSize(); - int above = transformContexts.Top[transformContexts.GetTopIndex(blockOrigin)] >= maximumTransformSize.GetWidth() ? 1 : 0; - int left = transformContexts.Left[transformContexts.GetLeftIndex(blockOrigin)] >= maximumTransformSize.GetHeight() ? 1 : 0; - int context = macroBlock.IsUpAvailable - ? macroBlock.IsLeftAvailable ? above + left : above - : macroBlock.IsLeftAvailable ? left : 0; - + int context = GetTransformSizeContext(transformContexts, macroBlock, blockOrigin, blockSize); writer.WriteTransformSize(blockSize, transformSize, context); } + else if (writesVariableTransformSize) + { + int topIndex = transformContexts.GetTopIndex(blockOrigin); + int leftIndex = transformContexts.GetLeftIndex(blockOrigin); + int context = Av1SymbolContextHelper.GetTransformPartitionContext( + transformContexts.Top[topIndex], + transformContexts.Left[leftIndex], + blockSize, + blockSize.GetMaximumTransformSize()); + + // Intra-block copy currently retains the maximum transform, so its variable-transform tree has one unsplit root. + writer.WriteTransformPartition(false, context); + } Size blockDimensions = new(blockSize.GetWidth(), blockSize.GetHeight()); @@ -1863,6 +1921,26 @@ internal partial class Av1TileWriter int transformBlockHeight = transformSize.Get4x4HighCount(); ReadOnlySpan topContexts = dcSignLevelCoefficientNeighborArray.Top.Slice(topIndex, transformBlockWidth); ReadOnlySpan leftContexts = dcSignLevelCoefficientNeighborArray.Left.Slice(leftIndex, transformBlockHeight); + + return GetTransformBlockContexts(plane, topContexts, leftContexts, planeBlockSize, transformSize); + } + + /// + /// Derives coefficient skip and DC-sign contexts from explicit transform-edge contexts. + /// + /// The luma or chroma component class. + /// The packed contexts immediately above the transform block. + /// The packed contexts immediately left of the transform block. + /// The containing block size on the target plane. + /// The transform size. + /// The coefficient skip and DC-sign contexts selected by both transform edges. + public static Av1TransformBlockContext GetTransformBlockContexts( + Av1ComponentType plane, + ReadOnlySpan topContexts, + ReadOnlySpan leftContexts, + Av1BlockSize planeBlockSize, + Av1TransformSize transformSize) + { int dcSign = 0; int top = 0; int left = 0; diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs index 7c41848072..047dfb39c9 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs @@ -594,6 +594,53 @@ public class Av1CoefficientsEntropyTests } } + [Fact] + public void TransformSizeContextUsesIntraBlockCopyNeighborExtents() + { + Av1PictureControlSet picture = CreateEncoderPicture(16, 16); + Point blockOrigin = new(16, 16); + Av1MacroBlockD macroBlock = new() + { + Tile = new Av1TileInfo(0, 0, picture.Parent.FrameHeader), + IsUpAvailable = true, + IsLeftAvailable = true + }; + + int modeInfoIndex = + ((blockOrigin.Y >> Av1Constants.ModeInfoSizeLog2) * picture.ModeInfoStride) + + (blockOrigin.X >> Av1Constants.ModeInfoSizeLog2); + + macroBlock.ModeInfoStride = picture.ModeInfoStride; + macroBlock.SetModeInfoGrid(picture.ModeInfoGrid, picture.ModeInfoAllocation, modeInfoIndex); + + ref Av1MacroBlockModeInfo aboveModeInfo = ref macroBlock.GetRelativeModeInfo(-macroBlock.ModeInfoStride); + aboveModeInfo.Block.BlockSize = Av1BlockSize.Block16x8; + aboveModeInfo.Block.UseIntraBlockCopy = true; + ref Av1MacroBlockModeInfo leftModeInfo = ref macroBlock.GetRelativeModeInfo(-1); + leftModeInfo.Block.BlockSize = Av1BlockSize.Block8x16; + leftModeInfo.Block.UseIntraBlockCopy = true; + + using Av1NeighborArrayUnit transforms = new( + Configuration.Default, + leftSize: 64, + topSize: 64) + { + GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2 + }; + + // Residual contexts report 8x8, but libaom derives 16x16 availability from the IBC coding blocks. + transforms.Top[transforms.GetTopIndex(blockOrigin)] = 8; + transforms.Left[transforms.GetLeftIndex(blockOrigin)] = 8; + + Assert.Equal( + 2, + Av1TileWriter.GetTransformSizeContext( + transforms, + macroBlock, + blockOrigin, + Av1BlockSize.Block16x16)); + } + [Fact] public void SelectedTransformSizeRoundTripsAndPublishesRectangularEdgeContexts() { @@ -611,6 +658,14 @@ public class Av1CoefficientsEntropyTests IsLeftAvailable = true }; + int modeInfoIndex = + ((blockOrigin.Y >> Av1Constants.ModeInfoSizeLog2) * picture.ModeInfoStride) + + (blockOrigin.X >> Av1Constants.ModeInfoSizeLog2); + + // Uniform-size context substitutes coding-block extents for inter neighbors, so mirror production mode-info setup. + macroBlock.ModeInfoStride = picture.ModeInfoStride; + macroBlock.SetModeInfoGrid(picture.ModeInfoGrid, picture.ModeInfoAllocation, modeInfoIndex); + using Av1NeighborArrayUnit transforms = new( Configuration.Default, leftSize: 64, diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs index dcbe393c7e..b15a10b909 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs @@ -7,6 +7,7 @@ using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; +using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; using SixLabors.ImageSharp.Formats.Heif.Components; using SixLabors.ImageSharp.Formats.Heif.Components.Alpha; using SixLabors.ImageSharp.Memory; @@ -488,13 +489,18 @@ public class Av1EncoderFrameTests } [Theory] - [InlineData(0, false, false)] - [InlineData(1, false, false)] - [InlineData(2, false, false)] - [InlineData(3, false, false)] - [InlineData(4, true, false)] - [InlineData(5, true, true)] - public void EncodeEffortControlsSearchFeatures(int effort, bool enableFilterIntra, bool enableScreenContentTools) + [InlineData(0, false, false, false)] + [InlineData(1, false, false, false)] + [InlineData(2, false, false, false)] + [InlineData(3, false, false, false)] + [InlineData(4, true, false, false)] + [InlineData(5, true, true, false)] + [InlineData(6, true, true, true)] + public void EncodeEffortControlsSearchFeatures( + int effort, + bool enableFilterIntra, + bool enableScreenContentTools, + bool selectTransformSize) { const int width = 16; const int height = 16; @@ -529,6 +535,9 @@ public class Av1EncoderFrameTests Assert.Equal(enableFilterIntra, sequenceHeader.EnableFilterIntra); Assert.Equal(enableScreenContentTools, frameHeader.AllowScreenContentTools); Assert.Equal(enableScreenContentTools, frameHeader.AllowIntraBlockCopy); + Assert.Equal( + selectTransformSize ? Av1TransformMode.Select : Av1TransformMode.Largest, + frameHeader.TransformMode); Assert.Equal(new Size(width, height), decoded.Size); int modeCount = 0; @@ -573,7 +582,64 @@ public class Av1EncoderFrameTests } [Fact] - public void EncodeSelectsIntraBlockCopyForRepeatedScreenContent() + public void EncodeEffortSixSelectsFourByFourLumaTransforms() + { + const int Width = 16; + const int Height = 16; + using Image source = new(Width, Height); + for (int row = 0; row < Height; row++) + { + Span pixels = source.Frames.RootFrame.PixelBuffer.DangerousGetRowSpan(row); + for (int column = 0; column < Width; column++) + { + byte value = (byte)(16 + ((((row >> 2) * 4) + (column >> 2)) * 14)); + pixels[column] = new Rgba32(value, value, value); + } + } + + ObuColorConfig colorConfig = CreateColorConfig(Av1BitDepth.EightBit, Av1ColorFormat.Yuv400); + using MemoryStream stream = new(); + _ = Av1FrameEncoder.Encode( + Configuration.Default, + source.Frames.RootFrame, + stream, + colorConfig, + qIndex: 37, + effort: 6); + + byte[] payload = stream.ToArray(); + using Av1Decoder decoder = new(Configuration.Default); + using Image decoded = decoder.Decode(payload); + Assert.NotNull(decoder.FrameHeader); + Assert.Equal(Av1TransformMode.Select, decoder.FrameHeader.TransformMode); + Assert.NotNull(decoder.FrameInfo); + bool foundSplitTransform = false; + foreach (Av1BlockModeInfo modeInfo in decoder.FrameInfo.GetSuperblock(Point.Empty).GetModeInfos()) + { + foundSplitTransform |= modeInfo.GetTransformUnitCount(Av1Plane.Y) == 4; + } + + Assert.True(foundSplitTransform); + Assert.Equal(new Size(Width, Height), decoded.Size); + + string outputDirectory = Path.Combine( + TestEnvironment.ActualOutputDirectoryFullPath, + "Formats", + "Heif", + "Av1"); + + Directory.CreateDirectory(outputDirectory); + File.WriteAllBytes( + Path.Combine(outputDirectory, "encoder-frame-16x16-8b-400-transform-size-select.obu"), + payload); + } + + [Theory] + [InlineData(5, false)] + [InlineData(6, true)] + public void EncodeSelectsIntraBlockCopyForRepeatedScreenContent( + int effort, + bool selectTransformSize) { const int Width = 328; const int Height = 16; @@ -599,7 +665,7 @@ public class Av1EncoderFrameTests stream, colorConfig, qIndex: 37, - effort: 5); + effort); byte[] payload = stream.ToArray(); using Av1Decoder decoder = new(Configuration.Default); @@ -607,6 +673,9 @@ public class Av1EncoderFrameTests Assert.NotNull(decoder.FrameHeader); Assert.True(decoder.FrameHeader.AllowScreenContentTools); Assert.True(decoder.FrameHeader.AllowIntraBlockCopy); + Assert.Equal( + selectTransformSize ? Av1TransformMode.Select : Av1TransformMode.Largest, + decoder.FrameHeader.TransformMode); Assert.NotNull(decoder.FrameInfo); Av1SuperblockInfo targetSuperblock = decoder.FrameInfo.GetSuperblock(new Point(5, 0)); bool usesIntraBlockCopy = false; @@ -625,7 +694,11 @@ public class Av1EncoderFrameTests "Av1"); Directory.CreateDirectory(outputDirectory); - File.WriteAllBytes(Path.Combine(outputDirectory, "encoder-frame-328x16-8b-444-intrabc.obu"), payload); + string fileName = effort == 5 + ? "encoder-frame-328x16-8b-444-intrabc.obu" + : "encoder-frame-328x16-8b-444-intrabc-effort-6.obu"; + + File.WriteAllBytes(Path.Combine(outputDirectory, fileName), payload); } [Fact] diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs index 88cd147d40..0bb7484115 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs @@ -134,6 +134,15 @@ public class Av1EntropyTests Av1ProbabilityCost.GetSymbolCost(transformSize, 1), encoder.GetTransformSizeCost(BlockSize, Av1TransformSize.Size4x4, SkipContext)); + Av1Distribution transformPartition = Av1DefaultDistributions.TransformPartition[SkipContext]; + Assert.Equal( + Av1ProbabilityCost.GetSymbolCost(transformPartition, 0), + encoder.GetTransformPartitionCost(false, SkipContext)); + + Assert.Equal( + Av1ProbabilityCost.GetSymbolCost(transformPartition, 1), + encoder.GetTransformPartitionCost(true, SkipContext)); + Av1Distribution transformSkip = Av1DefaultDistributions .GetTransformBlockSkip(BaseQIndex)[(int)TransformSize][SkipContext]; diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1SymbolContextTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1SymbolContextTests.cs index b21e7c5030..4d2eaaef79 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1SymbolContextTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1SymbolContextTests.cs @@ -76,6 +76,47 @@ public class Av1SymbolContextTests Assert.Equal((Av1TransformType)expectedValue, actual); } + [Theory] + [InlineData(0, 0, 0)] + [InlineData(-1, 4, 12)] + [InlineData(1, 4, 20)] + public void CoefficientContextMatchesCurrentLibaom(int dcCoefficient, ushort endOfBlock, byte expected) + { + Span coefficients = stackalloc int[16]; + coefficients.Fill(1); + coefficients[0] = dcCoefficient; + + byte actual = Av1SymbolContextHelper.GetCoefficientContext( + coefficients, + Av1TransformSize.Size4x4, + Av1TransformType.DctDct, + endOfBlock); + + Assert.Equal(expected, actual); + } + + [Theory] + [InlineData(8, 8, (int)Av1BlockSize.Block8x8, (int)Av1TransformSize.Size8x8, 18)] + [InlineData(4, 8, (int)Av1BlockSize.Block8x8, (int)Av1TransformSize.Size8x8, 19)] + [InlineData(4, 4, (int)Av1BlockSize.Block8x8, (int)Av1TransformSize.Size8x8, 20)] + [InlineData(4, 4, (int)Av1BlockSize.Block16x16, (int)Av1TransformSize.Size8x8, 17)] + [InlineData(0, 0, (int)Av1BlockSize.Block8x8, (int)Av1TransformSize.Size4x4, 0)] + public void TransformPartitionContextMatchesCurrentLibaom( + byte aboveWidth, + byte leftHeight, + int blockSizeValue, + int transformSizeValue, + int expected) + { + int actual = Av1SymbolContextHelper.GetTransformPartitionContext( + aboveWidth, + leftHeight, + (Av1BlockSize)blockSizeValue, + (Av1TransformSize)transformSizeValue); + + Assert.Equal(expected, actual); + } + public static TheoryData GetLowLevelContextEndOfBlockData() { TheoryData result = [];