From bdb366458c2de474a3d9dffaf45797a773518590 Mon Sep 17 00:00:00 2001 From: James Jackson-South Date: Thu, 3 Sep 2026 01:57:14 +1000 Subject: [PATCH] Add live AV1 chroma-from-luma search --- HEIF_IMPLEMENTATION_PLAN.md | 9 +- .../Heif/Av1/Entropy/Av1SymbolEncoder.cs | 28 ++ ...traSuperblockEncoder.ChromaModeDecision.cs | 251 ++++++++++++++- .../Av1IntraSuperblockEncoder.ModeDecision.cs | 6 +- .../Av1IntraSuperblockEncoder.Operator.cs | 217 +++++++++++++ .../Av1/Pipeline/Av1TransformBlockEncoder.cs | 199 +++++++++++- .../Av1ChromaFromLumaContext.Operations.cs | 117 +++++-- .../Av1ChromaFromLumaContext.cs | 24 +- .../ChromaFromLuma/Av1ChromaFromLumaMath.cs | 53 ++++ .../Formats/Heif/Av1/Av1EntropyTests.cs | 32 ++ .../Av1/Av1IntraSuperblockEncoderTests.cs | 291 +++++++++++++++++- 11 files changed, 1188 insertions(+), 39 deletions(-) diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 62cf022541..5e4ef1e19d 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -824,11 +824,11 @@ Encoder verification contract: - [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder operation now instantiates matching source and reconstruction owners with ordinary `using` lifetimes and converts packed pixels directly into the source owner before extension. - [~] Temporal delimiter, sequence header, frame header, combined-frame tile-group writing, and an internal reduced-still-picture frame operation now exist locally. The remaining required metadata, padding, multi-tile, option, and public encoder paths are not complete. - [~] Implement superblock and partition analysis for every permitted block size and partition. The current baseline deliberately splits every in-frame node to 8x8 blocks and records decisions in current-libaom writer preorder; block-size selection and non-split partition analysis remain. -- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma remains a separate search because it consumes reconstructed luma AC state; the remaining decision families remain. +- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax; palette, filter intra, and intra-block copy remain. - [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain. -- [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint spatial-chroma candidate sets now perform live rate-distortion selection; quality mapping, effort-dependent pruning, and the remaining searches are not implemented. -- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma is deliberately excluded until its reconstructed-luma AC search is implemented. Every candidate includes its live mode, angle, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these stack-only selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma candidate scratch remains one 8x8 reconstruction and one 8x8 coefficient span on the stack; chroma uses one transform-sized reconstruction and coefficient span for each of U and V. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, 63 focused predictor, block, mode-decision, decoder, and angle-rate cases pass, all 8,932 HEIF/AV1 namespace cases pass, and current-main `aomdec` accepts all 29 emitted 8/10/12-bit 4:0:0, 4:2:0, 4:2:2, and 4:4:4 constant or gradient payloads. Remaining mode decision work includes transform-size/type search, chroma-from-luma, palette and filter-intra search, partition search, full block-skip RD comparison, and effort-dependent pruning. +- [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma, now perform live rate-distortion selection; quality mapping, effort-dependent pruning, and the remaining searches are not implemented. +- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Every candidate includes its live mode, angle, alpha, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these stack-only selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma candidate scratch remains one 8x8 reconstruction and one 8x8 coefficient span on the stack; chroma uses one transform-sized reconstruction and coefficient span for each of U and V. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, 31 focused CfL, spatial-chroma, and entropy cases pass, all 8,959 HEIF/AV1 namespace cases pass, and current-main `aomdec` accepts all 29 emitted 8/10/12-bit 4:0:0, 4:2:0, 4:2:2, and 4:4:4 constant or gradient payloads. Remaining mode decision work includes transform-size search, broader joint mode/transform refinement, palette and filter-intra search, partition search, full block-skip RD comparison, and effort-dependent pruning. - [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One picture owner now packs segmentation plus every tile's partition, luma, chroma, and transform edges into one clean byte allocation with typed non-owning views; together with the separately typed packed mode-information owner, the complete picture state uses two allocator rents rather than seven. Exact aligned lengths, clean initialization, and balanced exactly-once returns are covered in Release. Multi-tile payload ownership and verified CDF update behavior remain. - [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains. - [~] The final-block decision workspace uses one reusable 10.3 KiB ImageSharp allocator owner. It contains 1,024 explicitly packed 10-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree, replacing separate managed arrays. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled palette, quantizer, prediction, and partition bytes cannot leak into the next decision pass. Exact allocation, size, initialization, reset, return, repeated-run, writer, entropy, and OBU coverage pass 1,957 of 1,957 direct net11 VSTest cases in Release; complete mode decision still remains. @@ -843,9 +843,10 @@ Encoder verification contract: - [~] The combined-frame writer now completes the byte-counted uncompressed frame header before starting the optional multi-tile tile-group flag, matching current libaom's separate frame-header and tile-group writers. A non-uniform two-tile round trip verifies the explicit boundaries, both tile payloads, and complete stream consumption through direct net11 VSTest in Release. - [~] The first internal frame-to-OBU operation encodes 8-, 10-, and 12-bit monochrome reduced still pictures through the production tile writer and production decoder. Coefficient context initialization now stores `min(abs(level), 127)`, matching current libaom; the previous signed clamp converted every negative transform coefficient to zero and selected invalid nonzero-map distributions. Signed dense and sparse entropy round trips, direct level-buffer saturation coverage, and eight constant/gradient frame cases pass 52 of 52 direct net11 VSTest cases in Release. Current-main `aomdec` accepts all eight emitted payloads. After winner-mode transform refinement, their decoded-frame MD5 values are `d09ea148582b9c93fa78e59426193bbc` (16x16 8-bit constant), `b83eedd5a84428f0120130253b30bdaa` (16x16 8-bit gradient), `f949f7422913e83dff07ee5e0a5087d3` (8x8 8-bit constant), `ae7233a94558978934469dcc4da764dd` (8x8 8-bit gradient), `09223b227f3abc3134d0a3ea15f70c0a` (8x8 10-bit constant), `539aab0e6e14bcaec271febfa8e25444` (8x8 10-bit gradient), `73117a8fc102e5d028f82444fc4d15ab` (8x8 12-bit constant), and `6936a2b62d7220dfb12f3763bb49965d` (8x8 12-bit gradient). This is an independently decodable baseline, not completion evidence for chroma, alpha, options, containers, or the public encoder. - [x] The exact net11 Release rebuild completed at the established 1,005-warning repository baseline with zero errors. The complete HEIF/AV1 namespace passes 8,838 of 8,838 direct VSTest cases with zero failures or skips. Roslynk reports zero compiler errors and no diagnostics in the five changed C# files; `git diff --check` passes and `.gitattributes` is unchanged. -- [~] The same internal frame operation now produces 4:2:0, 4:2:2, and 4:4:4 payloads at 8, 10, and 12 bits. Twenty-one color cases cover constant and spatially varying input at aligned dimensions plus odd 13x11 visible dimensions for every chroma geometry. The production decoder consumes every payload, the decoded output retains non-neutral chroma, and current-main `aomdec` accepts all 29 monochrome and color outputs. After live spatial chroma mode selection, implicit chroma-transform correction, and winner-mode luma-transform refinement, the odd-dimension decoded-frame MD5 values are `94ced594b3bc0fcbbfd55559a7e0088c` (4:2:0), `752777943f6e4ae7f6f879075dd311fe` (4:2:2), and `d25cc5e0f8404608530d6fec2f88915f` (4:4:4). This proves legal current-libaom payload syntax across native plane geometries; it does not yet prove target quality or native-plane equality with an independently encoded reference. +- [~] The same internal frame operation now produces 4:2:0, 4:2:2, and 4:4:4 payloads at 8, 10, and 12 bits. Twenty-one color cases cover constant and spatially varying input at aligned dimensions plus odd 13x11 visible dimensions for every chroma geometry. The production decoder consumes every payload, the decoded output retains non-neutral chroma, and current-main `aomdec` accepts all 29 monochrome and color outputs. After live spatial chroma mode selection, implicit chroma-transform correction, winner-mode luma-transform refinement, and exhaustive chroma-from-luma alpha selection, the odd-dimension decoded-frame MD5 values are `6a9cde0e29d02bcb1e6387f62fb59c88` (4:2:0), `ed8e9e52b0c0a80d6854ded5e0439538` (4:2:2), and `fcba16f73ce3690e83cafef529b11b7c` (4:4:4). This proves legal current-libaom payload syntax across native plane geometries; it does not yet prove target quality or native-plane equality with an independently encoded reference. - [x] Spatial chroma candidates now use the implicit transform derived from the selected UV mode and active transform set, matching current libaom's `intra_mode_to_tx_type` and `av1_get_tx_type` behavior. The same shared derivation is consumed by the decoder, so coefficient scan order, entropy contexts, inverse reconstruction, and encoder rate estimates cannot drift between the two paths. The previous DCT-DCT candidate transform could produce syntactically accepted streams whose non-DC chroma coefficients were interpreted under a different implicit transform. Six production mode-decision cases retain nonzero U and V coefficients and assert the selected transform state across 4:2:0, 4:2:2, and 4:4:4; fifteen exact mapping cases cover every intra mode, reduced sets, and the 32x32 DCT-only fallback. The focused contract passes 21 of 21 direct net11 VSTest cases, the complete HEIF/AV1 namespace passes 8,947 of 8,947, the exact Release rebuild remains at 1,005 warnings and zero errors, and current-main `aomdec` accepts all 29 regenerated payloads. - [~] Luma mode selection now evaluates each of its 61 mode-and-angle candidates with the mode-derived default transform used by current libaom's fast intra path. It then refines only the winning mode across all seven transform types permitted by the 8x8 intra set in transform-enum order. This removes the fixed DCT-DCT limitation while avoiding a 61-by-7 expansion; each trial includes live transform-type and coefficient rate, reconstructed pixel-domain distortion, and the existing allocation-free stack scratch. Eighteen exact-prediction production cases prove DCT-DCT wins equal-cost ties in reference order even when the first pass used a different default, while the 72x72 textured traversal proves a non-DCT transform with nonzero coefficients reaches retained syntax. Current-main `aomdec` accepts all 29 regenerated payloads. Full partition, transform-size, and effort-dependent joint mode/transform search remain. +- [x] Chroma-from-luma mode decision now reuses the decoder's SIMD-first 4:2:0, 4:2:2, and 4:4:4 reconstructed-luma preparation and prediction kernels for both byte and high-bit-depth encoder operators. The constant DC predictor for each chroma plane is computed once and its sample refills every alpha candidate, matching libaom's per-plane DC cache instead of rebuilding the same edge average 33 times. Each block uses 512 bytes of fixed stack scratch for the maximum 8-row predictor surface plus 792 bytes for complete U/V rate and distortion tables; no allocator owner, managed object, frame copy, or persistent buffer was added. Live probability costs exactly mirror current libaom's joint-sign ownership and conditional magnitude symbols. Nine production cases independently derive exact CfL targets from decoder-visible reconstructed luma at 8, 10, and 12 bits, and three entropy cases cover two nonzero signs plus each single-zero-plane form. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, all 8,959 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. - [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips. - [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips. - [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips. diff --git a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs index 8dcce60c21..f79a07dfb7 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs @@ -1114,6 +1114,34 @@ internal class Av1SymbolEncoder : IDisposable return Av1ProbabilityCost.GetSymbolCost(this.uvMode[cflAllowed][(int)lumaMode], (int)chromaMode); } + /// + /// Gets the current fixed-point cost of joint chroma-from-luma alpha syntax. + /// + /// The packed U/V alpha-magnitude indices. + /// The joint U/V sign symbol. + /// The rate cost in 1/512-bit units. + public int GetChromaFromLumaCost(int chromaFromLumaIndex, int joinedSign) + { + int cost = Av1ProbabilityCost.GetSymbolCost(this.chromaFromLumaSign, joinedSign); + int signU = Av1ChromaFromLumaMath.SignU(joinedSign); + if (signU != Av1ChromaFromLumaMath.SignZero) + { + int contextU = Av1ChromaFromLumaMath.ContextU(joinedSign); + int indexU = Av1ChromaFromLumaMath.IndexU(chromaFromLumaIndex); + cost += Av1ProbabilityCost.GetSymbolCost(this.chromaFromLumaAlpha[contextU], indexU); + } + + int signV = Av1ChromaFromLumaMath.SignV(joinedSign); + if (signV != Av1ChromaFromLumaMath.SignZero) + { + int contextV = Av1ChromaFromLumaMath.ContextV(joinedSign); + int indexV = Av1ChromaFromLumaMath.IndexV(chromaFromLumaIndex); + cost += Av1ProbabilityCost.GetSymbolCost(this.chromaFromLumaAlpha[contextV], indexV); + } + + return cost; + } + /// /// Writes a chroma intra prediction mode conditioned on the luma mode and chroma-from-luma availability. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs index 0b6c3c7af9..0150a5089e 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs @@ -4,6 +4,7 @@ using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; +using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; using SixLabors.ImageSharp.Memory; @@ -52,7 +53,9 @@ internal static partial class Av1IntraSuperblockEncoder Span retainedRedCoefficients, ref Av1EncoderTransformBlockState retainedBlueState, ref Av1EncoderTransformBlockState retainedRedState, - out int selectedAngleDelta) + out int selectedAngleDelta, + out byte selectedChromaFromLumaIndex, + out sbyte selectedChromaFromLumaSigns) { const Av1BlockSize BlockSize = Av1BlockSize.Block8x8; const int MaximumSampleCount = 8 * 8; @@ -154,6 +157,8 @@ internal static partial class Av1IntraSuperblockEncoder long bestCost = long.MaxValue; Av1ChromaPredictionMode bestMode = Av1ChromaPredictionMode.DC; selectedAngleDelta = 0; + selectedChromaFromLumaIndex = 0; + selectedChromaFromLumaSigns = 0; int baseModeCount = ChromaModeSearchOrder.Length; int deltaCount = AngleDeltaSearchOrder.Length; int directionalModeCount = (int)Av1ChromaPredictionMode.Directional67Degrees - (int)Av1ChromaPredictionMode.Vertical + 1; @@ -232,9 +237,253 @@ internal static partial class Av1IntraSuperblockEncoder } } + bool chromaFromLumaAllowed = BlockSize.AllowsChromaFromLuma( + this.picture.Parent.FrameHeader.LosslessArray[modeInfo.Block.SegmentId], + colorConfig.SubSamplingX, + colorConfig.SubSamplingY); + + if (chromaFromLumaAllowed) + { + Span lumaQ3 = stackalloc short[Av1ChromaFromLumaContext.BufferLine * 8]; + TOperator.PrepareChromaFromLuma( + this.reconstruction.GetPlane(Av1Plane.Y), + lumaOrigin, + lumaQ3, + transformSize, + colorConfig.SubSamplingX, + colorConfig.SubSamplingY); + + // Every alpha candidate uses the same constant DC predictor, so compute each plane once and + // refill the candidate block from its sample instead of rebuilding the identical edge average. + TOperator.PrepareChromaFromLumaDc( + candidateBlueReconstruction[..sampleCount], + blueAbove, + blueLeft, + hasLeft, + hasAbove, + transformSize, + this.bitDepth); + + TSample blueDc = candidateBlueReconstruction[0]; + TOperator.PrepareChromaFromLumaDc( + candidateRedReconstruction[..sampleCount], + redAbove, + redLeft, + hasLeft, + hasAbove, + transformSize, + this.bitDepth); + + TSample redDc = candidateRedReconstruction[0]; + Span blueRates = stackalloc int[Av1ChromaFromLumaMath.AlphaCandidateCount]; + Span redRates = stackalloc int[Av1ChromaFromLumaMath.AlphaCandidateCount]; + Span blueDistortions = stackalloc long[Av1ChromaFromLumaMath.AlphaCandidateCount]; + Span redDistortions = stackalloc long[Av1ChromaFromLumaMath.AlphaCandidateCount]; + + // Each plane has only 33 signed alpha values. Caching those complete transform results reduces + // the joint search from 1089 transform pairs to 66 transforms plus inexpensive rate combinations. + for (int alphaCandidateIndex = 0; alphaCandidateIndex < Av1ChromaFromLumaMath.AlphaCandidateCount; alphaCandidateIndex++) + { + int alphaQ3 = Av1ChromaFromLumaMath.CandidateIndexToAlpha(alphaCandidateIndex); + Av1EncoderTransformBlockState candidateBlueState = default; + blueDistortions[alphaCandidateIndex] = this.GetChromaFromLumaPlaneCost( + writer, + lumaMode, + Av1Plane.U, + chromaOrigin, + transformSize, + blueSource, + blueDc, + blueContext, + lumaQ3, + alphaQ3, + candidateBlueReconstruction[..sampleCount], + candidateBlueCoefficients[..sampleCount], + ref candidateBlueState, + out blueRates[alphaCandidateIndex]); + + Av1EncoderTransformBlockState candidateRedState = default; + redDistortions[alphaCandidateIndex] = this.GetChromaFromLumaPlaneCost( + writer, + lumaMode, + Av1Plane.V, + chromaOrigin, + transformSize, + redSource, + redDc, + redContext, + lumaQ3, + alphaQ3, + candidateRedReconstruction[..sampleCount], + candidateRedCoefficients[..sampleCount], + ref candidateRedState, + out redRates[alphaCandidateIndex]); + } + + int chromaFromLumaModeRate = Av1TileWriter.GetChromaModeCost( + writer, + this.picture.Parent.FrameHeader, + colorConfig, + modeInfo, + BlockSize, + lumaMode, + Av1ChromaPredictionMode.ChromaFromLuma, + 0); + + bool chromaFromLumaSelected = false; + int selectedBlueCandidateIndex = 0; + int selectedRedCandidateIndex = 0; + for (int blueCandidateIndex = 0; blueCandidateIndex < Av1ChromaFromLumaMath.AlphaCandidateCount; blueCandidateIndex++) + { + int alphaU = Av1ChromaFromLumaMath.CandidateIndexToAlpha(blueCandidateIndex); + int signU = Av1ChromaFromLumaMath.AlphaToSign(alphaU); + int indexU = Av1ChromaFromLumaMath.AlphaToMagnitudeIndex(alphaU); + for (int redCandidateIndex = 0; redCandidateIndex < Av1ChromaFromLumaMath.AlphaCandidateCount; redCandidateIndex++) + { + int alphaV = Av1ChromaFromLumaMath.CandidateIndexToAlpha(redCandidateIndex); + int signV = Av1ChromaFromLumaMath.AlphaToSign(alphaV); + if (signU == Av1ChromaFromLumaMath.SignZero && signV == Av1ChromaFromLumaMath.SignZero) + { + continue; + } + + int indexV = Av1ChromaFromLumaMath.AlphaToMagnitudeIndex(alphaV); + int jointSign = Av1ChromaFromLumaMath.JointSign(signU, signV); + int packedIndex = Av1ChromaFromLumaMath.PackIndices(indexU, indexV); + int rate = chromaFromLumaModeRate + + blueRates[blueCandidateIndex] + + redRates[redCandidateIndex] + + writer.GetChromaFromLumaCost(packedIndex, jointSign); + + long distortion = blueDistortions[blueCandidateIndex] + redDistortions[redCandidateIndex]; + long candidateCost = Av1RateDistortion.GetCost(this.rateMultiplier, rate, distortion); + bool winsSearchOrderTie = candidateCost == bestCost + && !chromaFromLumaSelected + && bestMode != Av1ChromaPredictionMode.DC; + + // CfL follows DC and precedes every other chroma mode in the reference search order. + if (candidateCost < bestCost || winsSearchOrderTie) + { + bestCost = candidateCost; + bestMode = Av1ChromaPredictionMode.ChromaFromLuma; + selectedAngleDelta = 0; + selectedBlueCandidateIndex = blueCandidateIndex; + selectedRedCandidateIndex = redCandidateIndex; + selectedChromaFromLumaIndex = (byte)packedIndex; + selectedChromaFromLumaSigns = (sbyte)jointSign; + chromaFromLumaSelected = true; + } + } + } + + if (chromaFromLumaSelected) + { + Av1EncoderTransformBlockState candidateBlueState = default; + _ = this.GetChromaFromLumaPlaneCost( + writer, + lumaMode, + Av1Plane.U, + chromaOrigin, + transformSize, + blueSource, + blueDc, + blueContext, + lumaQ3, + Av1ChromaFromLumaMath.CandidateIndexToAlpha(selectedBlueCandidateIndex), + candidateBlueReconstruction[..sampleCount], + candidateBlueCoefficients[..sampleCount], + ref candidateBlueState, + out _); + + CopyCandidate( + candidateBlueReconstruction, + candidateBlueCoefficients, + blueReconstruction, + chromaOrigin, + retainedBlueCoefficients, + transformSize, + candidateBlueState, + ref retainedBlueState); + + Av1EncoderTransformBlockState candidateRedState = default; + _ = this.GetChromaFromLumaPlaneCost( + writer, + lumaMode, + Av1Plane.V, + chromaOrigin, + transformSize, + redSource, + redDc, + redContext, + lumaQ3, + Av1ChromaFromLumaMath.CandidateIndexToAlpha(selectedRedCandidateIndex), + candidateRedReconstruction[..sampleCount], + candidateRedCoefficients[..sampleCount], + ref candidateRedState, + out _); + + CopyCandidate( + candidateRedReconstruction, + candidateRedCoefficients, + redReconstruction, + chromaOrigin, + retainedRedCoefficients, + transformSize, + candidateRedState, + ref retainedRedState); + } + } + return bestMode; } + private long GetChromaFromLumaPlaneCost( + Av1SymbolEncoder writer, + Av1PredictionMode lumaMode, + Av1Plane plane, + Point chromaOrigin, + Av1TransformSize transformSize, + Buffer2DRegion source, + TSample dc, + Av1TransformBlockContext context, + ReadOnlySpan lumaQ3, + int alphaQ3, + Span reconstruction, + Span coefficients, + ref Av1EncoderTransformBlockState state, + out int rate) + { + long distortion = TOperator.EncodeChromaFromLumaCandidate( + this.blockWorkspace, + source, + chromaOrigin, + reconstruction, + dc, + lumaQ3, + alphaQ3, + coefficients, + transformSize, + plane, + this.quantization.QIndex[0], + this.quantization.DeltaQDc[(int)plane], + this.quantization.DeltaQAc[(int)plane], + this.bitDepth, + ref state); + + rate = writer.GetCoefficientCost( + transformSize, + Av1TransformType.DctDct, + lumaMode, + coefficients, + Av1ComponentType.Chroma, + context, + state.EndOfBlock, + this.picture.Parent.FrameHeader.UseReducedTransformSet, + Av1FilterIntraMode.AllFilterIntraModes); + + return distortion; + } + private long GetChromaCandidateCost( Av1SymbolEncoder writer, Av1MacroBlockModeInfo modeInfo, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs index f94959545b..1bcf0b2b14 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs @@ -237,9 +237,13 @@ internal static partial class Av1IntraSuperblockEncoder redCoefficients[this.codedAreaChroma..], ref blueState, ref redState, - out int chromaAngleDelta); + out int chromaAngleDelta, + out byte chromaFromLumaIndex, + out sbyte chromaFromLumaSigns); block.PredictionUnit.AngleDelta[(int)Av1PlaneType.Uv] = (sbyte)chromaAngleDelta; + block.PredictionUnit.ChromaFromLumaIndex = chromaFromLumaIndex; + block.PredictionUnit.ChromaFromLumaSigns = chromaFromLumaSigns; // A block-level skip suppresses every coefficient symbol, so all coded planes must be empty. modeInfo.Block.Skip = skipTransform && blueState.EndOfBlock == 0 && redState.EndOfBlock == 0; diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs index 34349731c8..9d59e76599 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs @@ -3,6 +3,7 @@ using System.Runtime.InteropServices; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; +using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; using SixLabors.ImageSharp.Memory; @@ -36,6 +37,42 @@ internal static partial class Av1IntraSuperblockEncoder /// The converted sample. public static abstract TSample CreateSample(int value); + /// + /// Builds the zero-mean Q3 luma surface shared by chroma-from-luma candidates. + /// + /// The coded reconstructed luma plane. + /// The luma block origin in plane samples. + /// The fixed-stride Q3 predictor workspace. + /// The chroma transform dimensions. + /// Whether luma is subsampled horizontally for chroma. + /// Whether luma is subsampled vertically for chroma. + public static abstract void PrepareChromaFromLuma( + Buffer2DRegion reconstruction, + Point blockOrigin, + Span lumaQ3, + Av1TransformSize transformSize, + bool subsamplingX, + bool subsamplingY); + + /// + /// Computes the DC predictor shared by every chroma-from-luma alpha candidate. + /// + /// The contiguous candidate reconstruction. + /// The top reference samples. + /// The left reference samples. + /// Whether the left reference is available. + /// Whether the top reference is available. + /// The chroma transform dimensions. + /// The coded sample bit depth. + public static abstract void PrepareChromaFromLumaDc( + Span reconstruction, + ReadOnlySpan above, + ReadOnlySpan left, + bool hasLeft, + bool hasAbove, + Av1TransformSize transformSize, + Av1BitDepth bitDepth); + /// /// Encodes and reconstructs one DC intra transform block. /// @@ -116,6 +153,42 @@ internal static partial class Av1IntraSuperblockEncoder int acDeltaQ, Av1BitDepth bitDepth, ref Av1EncoderTransformBlockState state); + + /// + /// Encodes one chroma-from-luma candidate into contiguous decision scratch. + /// + /// The reusable block workspace. + /// The coded source plane. + /// The transform-block origin in plane samples. + /// The contiguous candidate reconstruction. + /// The cached DC predictor sample shared by every alpha. + /// The zero-mean reconstructed-luma predictor surface. + /// The signed chroma-from-luma multiplier. + /// The candidate entropy-coding coefficients. + /// The transform dimensions. + /// The component plane containing the block. + /// The effective segment quantizer index. + /// The plane DC quantizer adjustment. + /// The plane AC quantizer adjustment. + /// The coded sample bit depth. + /// The candidate transform state. + /// The normalized pixel-domain distortion in AV1 transform units. + public static abstract long EncodeChromaFromLumaCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span reconstruction, + TSample dc, + ReadOnlySpan lumaQ3, + int alphaQ3, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1Plane plane, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth, + ref Av1EncoderTransformBlockState state); } /// @@ -130,6 +203,44 @@ internal static partial class Av1IntraSuperblockEncoder /// public static byte CreateSample(int value) => (byte)value; + /// + public static void PrepareChromaFromLuma( + Buffer2DRegion reconstruction, + Point blockOrigin, + Span lumaQ3, + Av1TransformSize transformSize, + bool subsamplingX, + bool subsamplingY) + => Av1ChromaFromLumaContext.PrepareBlock( + Av1TransformBlockEncoder.GetPlaneSpan(reconstruction, blockOrigin), + reconstruction.Stride, + lumaQ3, + transformSize, + subsamplingX, + subsamplingY); + + /// + public static void PrepareChromaFromLumaDc( + Span reconstruction, + ReadOnlySpan above, + ReadOnlySpan left, + bool hasLeft, + bool hasAbove, + Av1TransformSize transformSize, + Av1BitDepth bitDepth) + { + int width = transformSize.GetWidth(); + Av1DcIntraPredictor.Predict( + hasLeft, + hasAbove, + reconstruction, + width, + above, + left, + width, + transformSize.GetHeight()); + } + /// public static void Encode( Av1EncoderBlockWorkspace workspace, @@ -206,6 +317,39 @@ internal static partial class Av1IntraSuperblockEncoder acDeltaQ, plane, ref state); + + /// + public static long EncodeChromaFromLumaCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span reconstruction, + byte dc, + ReadOnlySpan lumaQ3, + int alphaQ3, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1Plane plane, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth, + ref Av1EncoderTransformBlockState state) + => Av1TransformBlockEncoder.EncodeChromaFromLumaLossyCandidate( + workspace, + source, + blockOrigin, + reconstruction, + dc, + lumaQ3, + alphaQ3, + quantizedCoefficients, + transformSize, + qIndex, + dcDeltaQ, + acDeltaQ, + plane, + ref state); } /// @@ -220,6 +364,45 @@ internal static partial class Av1IntraSuperblockEncoder /// public static ushort CreateSample(int value) => (ushort)value; + /// + public static void PrepareChromaFromLuma( + Buffer2DRegion reconstruction, + Point blockOrigin, + Span lumaQ3, + Av1TransformSize transformSize, + bool subsamplingX, + bool subsamplingY) + => Av1ChromaFromLumaContext.PrepareBlock( + MemoryMarshal.Cast(Av1TransformBlockEncoder.GetPlaneSpan(reconstruction, blockOrigin)), + reconstruction.Stride, + lumaQ3, + transformSize, + subsamplingX, + subsamplingY); + + /// + public static void PrepareChromaFromLumaDc( + Span reconstruction, + ReadOnlySpan above, + ReadOnlySpan left, + bool hasLeft, + bool hasAbove, + Av1TransformSize transformSize, + Av1BitDepth bitDepth) + { + int width = transformSize.GetWidth(); + Av1DcIntraPredictor.Predict( + hasLeft, + hasAbove, + MemoryMarshal.Cast(reconstruction), + width, + MemoryMarshal.Cast(above), + MemoryMarshal.Cast(left), + width, + transformSize.GetHeight(), + bitDepth.GetBitCount()); + } + /// public static void Encode( Av1EncoderBlockWorkspace workspace, @@ -298,5 +481,39 @@ internal static partial class Av1IntraSuperblockEncoder plane, bitDepth, ref state); + + /// + public static long EncodeChromaFromLumaCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span reconstruction, + ushort dc, + ReadOnlySpan lumaQ3, + int alphaQ3, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1Plane plane, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth, + ref Av1EncoderTransformBlockState state) + => Av1TransformBlockEncoder.EncodeChromaFromLumaLossyCandidate( + workspace, + source, + blockOrigin, + reconstruction, + dc, + lumaQ3, + alphaQ3, + quantizedCoefficients, + transformSize, + qIndex, + dcDeltaQ, + acDeltaQ, + plane, + bitDepth, + ref state); } } diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs index f80a849b6e..466accc03c 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs @@ -4,6 +4,7 @@ using System.Runtime.InteropServices; using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; +using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; using SixLabors.ImageSharp.Memory; @@ -157,6 +158,96 @@ internal static class Av1TransformBlockEncoder return Av1ResidualBuilder.SumSquares(workspace.Residual[..transformSize.GetSize2d()]) << 4; } + /// + /// Encodes one eight-bit chroma-from-luma candidate into contiguous decision scratch. + /// + /// The reusable residual, coefficient, and transform storage. + /// The coded source plane. + /// The block origin in plane samples. + /// The contiguous candidate reconstruction. + /// The cached DC predictor sample shared by every alpha. + /// The zero-mean reconstructed-luma predictor surface. + /// The signed chroma-from-luma multiplier. + /// The candidate entropy-coding coefficients. + /// The selected chroma transform dimensions. + /// The segment quantizer index. + /// The plane DC quantizer adjustment. + /// The plane AC quantizer adjustment. + /// The component plane containing the block. + /// The candidate transform type and end-of-block syntax. + /// The normalized pixel-domain distortion in AV1 transform units. + public static long EncodeChromaFromLumaLossyCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span reconstruction, + byte dc, + ReadOnlySpan lumaQ3, + int alphaQ3, + Span quantizedCoefficients, + Av1TransformSize transformSize, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1Plane plane, + ref Av1EncoderTransformBlockState state) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + ReadOnlySpan sourceSamples = GetPlaneSpan(source, blockOrigin); + reconstruction[..transformSize.GetSize2d()].Fill(dc); + + // CfL adds its scaled reconstructed-luma AC contribution to the cached DC predictor before residual coding. + Av1ChromaFromLumaPredictor.Predict(lumaQ3, reconstruction, width, alphaQ3, width, height); + Av1ResidualBuilder.Subtract( + sourceSamples, + source.Stride, + reconstruction, + width, + workspace.Residual, + width, + width, + height); + + EncodeLossy( + workspace, + quantizedCoefficients, + transformSize, + Av1TransformType.DctDct, + qIndex, + dcDeltaQ, + acDeltaQ, + Av1BitDepth.EightBit, + ref state); + + if (state.EndOfBlock > 0) + { + Av1InverseTransformer.Reconstruct8Bit( + workspace.DequantizedCoefficients, + reconstruction, + width, + transformSize, + Av1TransformType.DctDct, + (int)plane, + state.EndOfBlock, + false, + workspace.TransformWorkspace); + } + + // Final distortion is measured against the samples a decoder reconstructs, not the unquantized predictor. + Av1ResidualBuilder.Subtract( + sourceSamples, + source.Stride, + reconstruction, + width, + workspace.Residual, + width, + width, + height); + + return Av1ResidualBuilder.SumSquares(workspace.Residual[..transformSize.GetSize2d()]) << 4; + } + /// /// Encodes and reconstructs one high-bit-depth lossy DC intra block in contiguous encoder planes. /// @@ -311,6 +402,112 @@ internal static class Av1TransformBlockEncoder return normalizedDistortion << 4; } + /// + /// Encodes one high-bit-depth chroma-from-luma candidate into contiguous decision scratch. + /// + /// The reusable residual, coefficient, and transform storage. + /// The coded source plane. + /// The block origin in plane samples. + /// The contiguous candidate reconstruction. + /// The cached DC predictor sample shared by every alpha. + /// The zero-mean reconstructed-luma predictor surface. + /// The signed chroma-from-luma multiplier. + /// The candidate entropy-coding coefficients. + /// The selected chroma transform dimensions. + /// The segment quantizer index. + /// The plane DC quantizer adjustment. + /// The plane AC quantizer adjustment. + /// The component plane containing the block. + /// The coded sample bit depth. + /// The candidate transform type and end-of-block syntax. + /// The normalized pixel-domain distortion in AV1 transform units. + public static long EncodeChromaFromLumaLossyCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span reconstruction, + ushort dc, + ReadOnlySpan lumaQ3, + int alphaQ3, + Span quantizedCoefficients, + Av1TransformSize transformSize, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1Plane plane, + Av1BitDepth bitDepth, + ref Av1EncoderTransformBlockState state) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + ReadOnlySpan sourceSamples = GetPlaneSpan(source, blockOrigin); + Span signedReconstruction = MemoryMarshal.Cast(reconstruction); + reconstruction[..transformSize.GetSize2d()].Fill(dc); + + Av1ChromaFromLumaPredictor.Predict( + lumaQ3, + signedReconstruction, + width, + alphaQ3, + bitDepth.GetBitCount(), + width, + height); + + Av1ResidualBuilder.Subtract( + sourceSamples, + source.Stride, + reconstruction, + width, + workspace.Residual, + width, + width, + height); + + EncodeLossy( + workspace, + quantizedCoefficients, + transformSize, + Av1TransformType.DctDct, + qIndex, + dcDeltaQ, + acDeltaQ, + bitDepth, + ref state); + + if (state.EndOfBlock > 0) + { + Av1InverseTransformer.ReconstructHighBitDepth( + workspace.DequantizedCoefficients, + signedReconstruction, + width, + transformSize, + Av1TransformType.DctDct, + (int)plane, + state.EndOfBlock, + false, + bitDepth, + workspace.TransformWorkspace); + } + + Av1ResidualBuilder.Subtract( + sourceSamples, + source.Stride, + reconstruction, + width, + workspace.Residual, + width, + width, + height); + + long distortion = Av1ResidualBuilder.SumSquares(workspace.Residual[..transformSize.GetSize2d()]); + int shift = (bitDepth.GetBitCount() - 8) * 2; + long normalizedDistortion = shift == 0 + ? distortion + : (distortion + (1L << (shift - 1))) >> shift; + + return normalizedDistortion << 4; + } + /// /// Encodes and reconstructs one eight-bit lossy intra block. /// @@ -591,7 +788,7 @@ internal static class Av1TransformBlockEncoder state.TransformType = transformType; } - private static Span GetPlaneSpan(Buffer2DRegion plane, Point blockOrigin) + public static Span GetPlaneSpan(Buffer2DRegion plane, Point blockOrigin) where TSample : unmanaged { int offset = diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs index cbcc5fcb79..4e22a9bba0 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs @@ -18,6 +18,52 @@ namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; /// internal partial class Av1ChromaFromLumaContext { + /// + /// Subsamples one reconstructed eight-bit luma block and removes its rounded Q3 mean. + /// + /// The reconstructed luma samples. + /// The distance, in samples, between input rows. + /// The fixed-stride Q3 predictor workspace. + /// The chroma transform dimensions. + /// Whether two horizontal luma samples map to each chroma sample. + /// Whether two vertical luma samples map to each chroma sample. + public static void PrepareBlock( + ReadOnlySpan input, + int inputStride, + Span output, + Av1TransformSize transformSize, + bool subsamplingX, + bool subsamplingY) + { + int lumaWidth = transformSize.GetWidth() << (subsamplingX ? 1 : 0); + int lumaHeight = transformSize.GetHeight() << (subsamplingY ? 1 : 0); + StoreSamples(input, inputStride, 0, lumaWidth, lumaHeight, output, subsamplingX, subsamplingY); + SubtractAverage(output, transformSize); + } + + /// + /// Subsamples one reconstructed high-bit-depth luma block and removes its rounded Q3 mean. + /// + /// The reconstructed luma samples. + /// The distance, in samples, between input rows. + /// The fixed-stride Q3 predictor workspace. + /// The chroma transform dimensions. + /// Whether two horizontal luma samples map to each chroma sample. + /// Whether two vertical luma samples map to each chroma sample. + public static void PrepareBlock( + ReadOnlySpan input, + int inputStride, + Span output, + Av1TransformSize transformSize, + bool subsamplingX, + bool subsamplingY) + { + int lumaWidth = transformSize.GetWidth() << (subsamplingX ? 1 : 0); + int lumaHeight = transformSize.GetHeight() << (subsamplingY ? 1 : 0); + StoreSamples(input, inputStride, 0, lumaWidth, lumaHeight, output, subsamplingX, subsamplingY); + SubtractAverage(output, transformSize); + } + /// /// Stores 8-bit reconstructed luma samples in the Q3 predictor surface. /// @@ -26,12 +72,23 @@ internal partial class Av1ChromaFromLumaContext /// The first destination sample in the fixed-stride predictor buffer. /// The luma width in samples. /// The luma height in samples. - private void StoreSamples(ReadOnlySpan input, int inputStride, int outputOffset, int width, int height) + /// The fixed-stride Q3 predictor workspace. + /// Whether horizontal luma pairs are subsampled. + /// Whether vertical luma pairs are subsampled. + private static void StoreSamples( + ReadOnlySpan input, + int inputStride, + int outputOffset, + int width, + int height, + Span output, + bool subsamplingX, + bool subsamplingY) { ref byte inputBase = ref MemoryMarshal.GetReference(input); - ref short outputBase = ref MemoryMarshal.GetReference(this.Q3Buffer); + ref short outputBase = ref MemoryMarshal.GetReference(output); - if (!this.subX) + if (!subsamplingX) { // One luma sample maps directly to one chroma sample, so multiplying by eight converts it to Q3. for (int row = 0; row < height; row++) @@ -94,13 +151,13 @@ internal partial class Av1ChromaFromLumaContext Vector256 ones256 = Vector256.Create((sbyte)1); Vector128 ones128 = Vector128.Create((sbyte)1); - int rowStep = this.subY ? 2 : 1; - int outputShift = this.subY ? 1 : 2; + int rowStep = subsamplingY ? 2 : 1; + int outputShift = subsamplingY ? 1 : 2; for (int row = 0; row < height; row += rowStep) { ref byte inputRow = ref Unsafe.Add(ref inputBase, row * inputStride); - ref byte nextInputRow = ref Unsafe.Add(ref inputRow, this.subY ? inputStride : 0); - ref short outputRow = ref Unsafe.Add(ref outputBase, outputOffset + ((row >> (this.subY ? 1 : 0)) * BufferLine)); + ref byte nextInputRow = ref Unsafe.Add(ref inputRow, subsamplingY ? inputStride : 0); + ref short outputRow = ref Unsafe.Add(ref outputBase, outputOffset + ((row >> (subsamplingY ? 1 : 0)) * BufferLine)); int column = 0; if (Avx2.IsSupported) @@ -109,7 +166,7 @@ internal partial class Av1ChromaFromLumaContext for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 sum = Avx2.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref inputRow, (nuint)column), ones256); - if (this.subY) + if (subsamplingY) { sum += Avx2.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref nextInputRow, (nuint)column), ones256); } @@ -124,7 +181,7 @@ internal partial class Av1ChromaFromLumaContext for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 sum = PairSum(Vector128.LoadUnsafe(ref inputRow, (nuint)column), ones128); - if (this.subY) + if (subsamplingY) { sum += PairSum(Vector128.LoadUnsafe(ref nextInputRow, (nuint)column), ones128); } @@ -139,7 +196,7 @@ internal partial class Av1ChromaFromLumaContext ? Unsafe.ReadUnaligned(ref Unsafe.Add(ref inputRow, column)) : Unsafe.ReadUnaligned(ref Unsafe.Add(ref inputRow, column)); Vector128 sum = PairSum(Vector128.CreateScalarUnsafe(packed).AsByte(), ones128); - if (this.subY) + if (subsamplingY) { packed = remaining == 4 ? Unsafe.ReadUnaligned(ref Unsafe.Add(ref nextInputRow, column)) @@ -164,7 +221,7 @@ internal partial class Av1ChromaFromLumaContext for (; column < width; column += 2) { int sum = Unsafe.Add(ref inputRow, column) + Unsafe.Add(ref inputRow, column + 1); - if (this.subY) + if (subsamplingY) { sum += Unsafe.Add(ref nextInputRow, column) + Unsafe.Add(ref nextInputRow, column + 1); } @@ -182,12 +239,23 @@ internal partial class Av1ChromaFromLumaContext /// The first destination sample in the fixed-stride predictor buffer. /// The luma width in samples. /// The luma height in samples. - private void StoreSamples(ReadOnlySpan input, int inputStride, int outputOffset, int width, int height) + /// The fixed-stride Q3 predictor workspace. + /// Whether horizontal luma pairs are subsampled. + /// Whether vertical luma pairs are subsampled. + private static void StoreSamples( + ReadOnlySpan input, + int inputStride, + int outputOffset, + int width, + int height, + Span output, + bool subsamplingX, + bool subsamplingY) { ref short inputBase = ref MemoryMarshal.GetReference(input); - ref short outputBase = ref MemoryMarshal.GetReference(this.Q3Buffer); + ref short outputBase = ref MemoryMarshal.GetReference(output); - if (!this.subX) + if (!subsamplingX) { for (int row = 0; row < height; row++) { @@ -232,13 +300,13 @@ internal partial class Av1ChromaFromLumaContext Vector256 ones256 = Vector256.Create((short)1); Vector128 ones128 = Vector128.Create((short)1); - int rowStep = this.subY ? 2 : 1; - int outputShift = this.subY ? 1 : 2; + int rowStep = subsamplingY ? 2 : 1; + int outputShift = subsamplingY ? 1 : 2; for (int row = 0; row < height; row += rowStep) { ref short inputRow = ref Unsafe.Add(ref inputBase, row * inputStride); - ref short nextInputRow = ref Unsafe.Add(ref inputRow, this.subY ? inputStride : 0); - ref short outputRow = ref Unsafe.Add(ref outputBase, outputOffset + ((row >> (this.subY ? 1 : 0)) * BufferLine)); + ref short nextInputRow = ref Unsafe.Add(ref inputRow, subsamplingY ? inputStride : 0); + ref short outputRow = ref Unsafe.Add(ref outputBase, outputOffset + ((row >> (subsamplingY ? 1 : 0)) * BufferLine)); int column = 0; if (Vector256.IsHardwareAccelerated) @@ -247,7 +315,7 @@ internal partial class Av1ChromaFromLumaContext for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 sum = Vector256_.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref inputRow, (nuint)column), ones256); - if (this.subY) + if (subsamplingY) { sum += Vector256_.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref nextInputRow, (nuint)column), ones256); } @@ -262,7 +330,7 @@ internal partial class Av1ChromaFromLumaContext for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 sum = Vector128_.MultiplyAddAdjacent(Vector128.LoadUnsafe(ref inputRow, (nuint)column), ones128); - if (this.subY) + if (subsamplingY) { sum += Vector128_.MultiplyAddAdjacent(Vector128.LoadUnsafe(ref nextInputRow, (nuint)column), ones128); } @@ -275,7 +343,7 @@ internal partial class Av1ChromaFromLumaContext Vector128 samples = Vector128.CreateScalarUnsafe( Unsafe.ReadUnaligned(ref Unsafe.As(ref Unsafe.Add(ref inputRow, column)))).AsInt16(); Vector128 sum = Vector128_.MultiplyAddAdjacent(samples, ones128); - if (this.subY) + if (subsamplingY) { samples = Vector128.CreateScalarUnsafe( Unsafe.ReadUnaligned(ref Unsafe.As(ref Unsafe.Add(ref nextInputRow, column)))).AsInt16(); @@ -291,7 +359,7 @@ internal partial class Av1ChromaFromLumaContext for (; column < width; column += 2) { int sum = Unsafe.Add(ref inputRow, column) + Unsafe.Add(ref inputRow, column + 1); - if (this.subY) + if (subsamplingY) { sum += Unsafe.Add(ref nextInputRow, column) + Unsafe.Add(ref nextInputRow, column + 1); } @@ -304,8 +372,9 @@ internal partial class Av1ChromaFromLumaContext /// /// Subtracts the rounded Q3 average from each predictor sample, leaving the AC contribution used by CfL. /// + /// The fixed-stride Q3 predictor workspace. /// The populated predictor dimensions. - private void SubtractAverage(Av1TransformSize transformSize) + private static void SubtractAverage(Span buffer, Av1TransformSize transformSize) { int width = transformSize.GetWidth(); int height = transformSize.GetHeight(); @@ -313,7 +382,7 @@ internal partial class Av1ChromaFromLumaContext // Transform dimensions are powers of two, so division by the sample count is an exact right shift. Half // the sample count is accumulated first to implement the normative nearest-integer rounding. int sumQ3 = (width * height) >> 1; - ref short bufferBase = ref MemoryMarshal.GetReference(this.Q3Buffer); + ref short bufferBase = ref MemoryMarshal.GetReference(buffer); if (Vector256.IsHardwareAccelerated && width >= Vector256.Count) { diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.cs index d3641d6006..39e3de8803 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.cs @@ -16,7 +16,7 @@ internal sealed partial class Av1ChromaFromLumaContext /// /// The fixed row stride and maximum dimension, in chroma samples, of the luma predictor buffer. /// - private const int BufferLine = 32; + public const int BufferLine = 32; /// /// The number of samples in the fixed-stride chroma-from-luma workspace. @@ -145,11 +145,27 @@ internal sealed partial class Av1ChromaFromLumaContext // here keeps sample conversion out of the row kernels and lets the JIT specialize both storage layouts. if (typeof(T) == typeof(byte)) { - this.StoreSamples(MemoryMarshal.Cast(input), inputStride, outputOffset, width, height); + StoreSamples( + MemoryMarshal.Cast(input), + inputStride, + outputOffset, + width, + height, + this.Q3Buffer, + this.subX, + this.subY); } else { - this.StoreSamples(MemoryMarshal.Cast(input), inputStride, outputOffset, width, height); + StoreSamples( + MemoryMarshal.Cast(input), + inputStride, + outputOffset, + width, + height, + this.Q3Buffer, + this.subX, + this.subY); } } @@ -161,7 +177,7 @@ internal sealed partial class Av1ChromaFromLumaContext { Guard.IsFalse(this.AreParametersComputed, nameof(this.AreParametersComputed), "Do not call cfl_compute_parameters multiple time on the same values."); this.Pad(transformSize.GetWidth(), transformSize.GetHeight()); - this.SubtractAverage(transformSize); + SubtractAverage(this.Q3Buffer, transformSize); this.AreParametersComputed = true; } diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaMath.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaMath.cs index 36106563f9..1c74307f47 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaMath.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaMath.cs @@ -18,6 +18,21 @@ internal static class Av1ChromaFromLumaMath /// private const int AlphabetSizeLog2 = 4; + /// + /// The number of nonzero alpha magnitudes represented by each plane's alphabet. + /// + public const int AlphaMagnitudeCount = 1 << AlphabetSizeLog2; + + /// + /// The number of signed alpha candidates including zero. + /// + public const int AlphaCandidateCount = (AlphaMagnitudeCount * 2) + 1; + + /// + /// The candidate index representing a zero alpha. + /// + public const int AlphaZeroIndex = AlphaMagnitudeCount; + /// /// The alpha sign value representing a zero multiplier. /// @@ -74,4 +89,42 @@ internal static class Av1ChromaFromLumaMath /// The coded joint U/V sign symbol. /// The V-plane alpha entropy context. public static int ContextV(int jointSign) => (SignV(jointSign) * Signs) + SignU(jointSign) - Signs; + + /// + /// Converts a signed-candidate index to its alpha value in Q3 units. + /// + /// The candidate index in negative-to-positive order. + /// The signed Q3 alpha value. + public static int CandidateIndexToAlpha(int candidateIndex) => candidateIndex - AlphaZeroIndex; + + /// + /// Converts a signed Q3 alpha value to its coded sign state. + /// + /// The signed alpha value. + /// The zero, negative, or positive sign state. + public static int AlphaToSign(int alphaQ3) + => alphaQ3 == 0 ? SignZero : alphaQ3 < 0 ? SignNegative : SignPositive; + + /// + /// Converts a nonzero signed Q3 alpha value to its coded magnitude index. + /// + /// The signed alpha value. + /// The zero-based magnitude index, or zero for a zero alpha. + public static int AlphaToMagnitudeIndex(int alphaQ3) => alphaQ3 == 0 ? 0 : Math.Abs(alphaQ3) - 1; + + /// + /// Combines the U and V sign states into the coded joint symbol. + /// + /// The U-plane sign state. + /// The V-plane sign state. + /// The joint sign symbol. + public static int JointSign(int signU, int signV) => (signU * Signs) + signV - 1; + + /// + /// Packs the U and V alpha-magnitude indices into the coded byte. + /// + /// The U-plane magnitude index. + /// The V-plane magnitude index. + /// The packed magnitude indices. + public static int PackIndices(int indexU, int indexV) => (indexU << AlphabetSizeLog2) + indexV; } diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs index bb65992cce..19d1634e03 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs @@ -7,6 +7,7 @@ using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; using SixLabors.ImageSharp.Formats.Heif.Av1.Motion; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; +using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; using SixLabors.ImageSharp.Memory; @@ -141,6 +142,37 @@ public class Av1EntropyTests encoder.GetTransformBlockSkipCost(true, TransformSize, SkipContext)); } + [Theory] + [InlineData(-16, 16)] + [InlineData(0, 8)] + [InlineData(-4, 0)] + public void ChromaFromLumaCostMatchesCurrentDistributions(int alphaU, int alphaV) + { + int signU = Av1ChromaFromLumaMath.AlphaToSign(alphaU); + int signV = Av1ChromaFromLumaMath.AlphaToSign(alphaV); + int jointSign = Av1ChromaFromLumaMath.JointSign(signU, signV); + int indexU = Av1ChromaFromLumaMath.AlphaToMagnitudeIndex(alphaU); + int indexV = Av1ChromaFromLumaMath.AlphaToMagnitudeIndex(alphaV); + int packedIndex = Av1ChromaFromLumaMath.PackIndices(indexU, indexV); + int expected = Av1ProbabilityCost.GetSymbolCost(Av1DefaultDistributions.ChromaFromLumaSign, jointSign); + if (signU != Av1ChromaFromLumaMath.SignZero) + { + expected += Av1ProbabilityCost.GetSymbolCost( + Av1DefaultDistributions.ChromaFromLumaAlpha[Av1ChromaFromLumaMath.ContextU(jointSign)], + indexU); + } + + if (signV != Av1ChromaFromLumaMath.SignZero) + { + expected += Av1ProbabilityCost.GetSymbolCost( + Av1DefaultDistributions.ChromaFromLumaAlpha[Av1ChromaFromLumaMath.ContextV(jointSign)], + indexV); + } + + using Av1SymbolEncoder encoder = new(Configuration.Default, 64, BaseQIndex, updateCdf: false); + Assert.Equal(expected, encoder.GetChromaFromLumaCost(packedIndex, jointSign)); + } + /// /// Verifies that live luma rate accounting includes the selected signed directional adjustment. /// diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs index 98ea4a2158..bf1c415563 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs @@ -2,11 +2,13 @@ // Licensed under the Six Labors Split License. using System.Buffers; +using System.Numerics; using SixLabors.ImageSharp.Formats.Heif.Av1; using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; +using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; using SixLabors.ImageSharp.Memory; @@ -297,12 +299,12 @@ public class Av1IntraSuperblockEncoderTests 1, 1); - FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.Y), 128); + FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.Y), (byte)128); ClearPlane(reconstruction.Luma); if (!isMonochrome) { - FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.U), 128); - FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.V), 128); + FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.U), (byte)128); + FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.V), (byte)128); ClearPlane(Assert.IsType>(reconstruction.ChromaBlue)); ClearPlane(Assert.IsType>(reconstruction.ChromaRed)); } @@ -921,6 +923,277 @@ public class Av1IntraSuperblockEncoderTests Assert.NotEqual(0, tileWriter.GetTileData(0).Length); } + [Theory] + [InlineData((int)Av1ColorFormat.Yuv420)] + [InlineData((int)Av1ColorFormat.Yuv422)] + [InlineData((int)Av1ColorFormat.Yuv444)] + public void ProductionTileSelectsChromaFromReconstructedLuma(int colorFormatValue) + => VerifyProductionTileSelectsChromaFromReconstructedLuma( + colorFormatValue, + 8, + static (source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new( + Configuration.Default, + source, + reconstruction, + picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialSize: 512)); + + [Theory] + [InlineData((int)Av1ColorFormat.Yuv420, 10)] + [InlineData((int)Av1ColorFormat.Yuv420, 12)] + [InlineData((int)Av1ColorFormat.Yuv422, 10)] + [InlineData((int)Av1ColorFormat.Yuv422, 12)] + [InlineData((int)Av1ColorFormat.Yuv444, 10)] + [InlineData((int)Av1ColorFormat.Yuv444, 12)] + public void ProductionTileSelectsChromaFromReconstructedLumaHighBitDepth( + int colorFormatValue, + int bitDepth) + => VerifyProductionTileSelectsChromaFromReconstructedLuma( + colorFormatValue, + bitDepth, + static (source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new( + Configuration.Default, + source, + reconstruction, + picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialSize: 512)); + + private static void VerifyProductionTileSelectsChromaFromReconstructedLuma( + int colorFormatValue, + int bitDepth, + TileWriterFactory createWriter) + where TSample : unmanaged, IBinaryInteger + { + const int Width = 16; + const int Height = 16; + const int QIndex = 1; + const int AlphaU = 16; + const int AlphaV = -16; + Av1ColorFormat colorFormat = (Av1ColorFormat)colorFormatValue; + bool subsamplingX = colorFormat is Av1ColorFormat.Yuv420 or Av1ColorFormat.Yuv422; + bool subsamplingY = colorFormat == Av1ColorFormat.Yuv420; + int chromaSubsamplingX = subsamplingX ? 1 : 0; + int chromaSubsamplingY = subsamplingY ? 1 : 0; + int sampleScale = 1 << (bitDepth - 8); + int midpoint = 1 << (bitDepth - 1); + int maxSample = (1 << bitDepth) - 1; + Av1TransformSize transformSize = Av1BlockSize.Block8x8.GetMaxUvTransformSize( + subsamplingX, + subsamplingY); + + ObuColorConfig colorConfig = new() + { + IsMonochrome = false, + SubSamplingX = subsamplingX, + SubSamplingY = subsamplingY, + BitDepth = (Av1BitDepth)((bitDepth - 8) / 2) + }; + + using Av1EncoderFrameBuffer pilotSource = new( + Configuration.Default, + Width, + Height, + bitDepth, + colorFormat, + chromaSubsamplingX, + chromaSubsamplingY); + + using Av1EncoderFrameBuffer pilotReconstruction = new( + Configuration.Default, + Width, + Height, + bitDepth, + colorFormat, + chromaSubsamplingX, + chromaSubsamplingY); + + Buffer2DRegion pilotLuma = pilotSource.Frame.CodedView.GetPlane(Av1Plane.Y); + for (int y = 0; y < pilotLuma.Height; y++) + { + Span row = pilotLuma.DangerousGetRowSpan(y); + for (int x = 0; x < row.Length; x++) + { + row[x] = TSample.CreateChecked( + (96 + (((x * 29) + (y * 47) + (((x ^ y) & 1) * 53)) & 63)) * sampleScale); + } + } + + FillPlane(pilotSource.Frame.CodedView.GetPlane(Av1Plane.U), TSample.CreateChecked(midpoint)); + FillPlane(pilotSource.Frame.CodedView.GetPlane(Av1Plane.V), TSample.CreateChecked(midpoint)); + ClearPlane(pilotReconstruction.Luma); + ClearPlane(Assert.IsType>(pilotReconstruction.ChromaBlue)); + ClearPlane(Assert.IsType>(pilotReconstruction.ChromaRed)); + using Av1EncoderModeInfoBuffer pilotModeInfo = new(Configuration.Default, Width, Height, disallow4x4AllFrames: true); + Av1PictureControlSet pilotTemplate = CreatePicture(pilotModeInfo, colorConfig, use128x128Superblock: false, QIndex); + using Av1EncoderPictureBuffer pilotPicture = new( + Configuration.Default, + pilotTemplate.Sequence.SequenceHeader, + pilotTemplate.Parent.FrameHeader, + Width, + Height); + + using Av1EncoderCoefficientBuffer pilotCoefficients = new( + Configuration.Default, + pilotTemplate.Sequence.SequenceHeader, + Width, + Height); + + using Av1EncoderSuperblockWorkspace pilotSuperblockWorkspace = new(Configuration.Default); + using Av1EncoderBlockWorkspace pilotBlockWorkspace = new(Configuration.Default); + using Av1IntraTileWriter pilotWriter = createWriter( + pilotSource.Frame, + pilotReconstruction.Frame, + pilotPicture.Picture, + pilotCoefficients, + pilotSuperblockWorkspace, + pilotBlockWorkspace); + + Buffer2DRegion reconstructedLuma = pilotReconstruction.Frame.CodedView.GetPlane(Av1Plane.Y); + int chromaWidth = transformSize.GetWidth(); + int chromaHeight = transformSize.GetHeight(); + int sampleCount = transformSize.GetSize2d(); + int lumaScaleShift = 3 - chromaSubsamplingX - chromaSubsamplingY; + Span lumaQ3 = stackalloc short[64]; + int sumQ3 = sampleCount >> 1; + for (int row = 0; row < chromaHeight; row++) + { + for (int column = 0; column < chromaWidth; column++) + { + int lumaSum = 0; + int lumaX = 8 + (column << chromaSubsamplingX); + int lumaY = 8 + (row << chromaSubsamplingY); + for (int offsetY = 0; offsetY <= chromaSubsamplingY; offsetY++) + { + ReadOnlySpan lumaRow = reconstructedLuma.DangerousGetRowSpan(lumaY + offsetY); + for (int offsetX = 0; offsetX <= chromaSubsamplingX; offsetX++) + { + lumaSum += int.CreateChecked(lumaRow[lumaX + offsetX]); + } + } + + short sampleQ3 = (short)(lumaSum << lumaScaleShift); + lumaQ3[(row * chromaWidth) + column] = sampleQ3; + sumQ3 += sampleQ3; + } + } + + int averageQ3 = sumQ3 >> (transformSize.GetBlockWidthLog2() + transformSize.GetBlockHeightLog2()); + using Av1EncoderFrameBuffer source = new( + Configuration.Default, + Width, + Height, + bitDepth, + colorFormat, + chromaSubsamplingX, + chromaSubsamplingY); + + using Av1EncoderFrameBuffer reconstruction = new( + Configuration.Default, + Width, + Height, + bitDepth, + colorFormat, + chromaSubsamplingX, + chromaSubsamplingY); + + for (int y = 0; y < pilotLuma.Height; y++) + { + pilotLuma.DangerousGetRowSpan(y).CopyTo(source.Frame.CodedView.GetPlane(Av1Plane.Y).DangerousGetRowSpan(y)); + } + + Buffer2DRegion blue = source.Frame.CodedView.GetPlane(Av1Plane.U); + Buffer2DRegion red = source.Frame.CodedView.GetPlane(Av1Plane.V); + FillPlane(blue, TSample.CreateChecked(midpoint)); + FillPlane(red, TSample.CreateChecked(midpoint)); + for (int row = 0; row < chromaHeight; row++) + { + Span blueRow = blue.DangerousGetRowSpan(chromaHeight + row); + Span redRow = red.DangerousGetRowSpan(chromaHeight + row); + for (int column = 0; column < chromaWidth; column++) + { + int acQ3 = lumaQ3[(row * chromaWidth) + column] - averageQ3; + int blueProduct = AlphaU * acQ3; + int redProduct = AlphaV * acQ3; + int blueAdjustment = (blueProduct + 32 + (blueProduct >> 31)) >> 6; + int redAdjustment = (redProduct + 32 + (redProduct >> 31)) >> 6; + blueRow[chromaWidth + column] = TSample.CreateChecked(Math.Clamp(midpoint + blueAdjustment, 0, maxSample)); + redRow[chromaWidth + column] = TSample.CreateChecked(Math.Clamp(midpoint + redAdjustment, 0, maxSample)); + } + } + + ClearPlane(reconstruction.Luma); + ClearPlane(Assert.IsType>(reconstruction.ChromaBlue)); + ClearPlane(Assert.IsType>(reconstruction.ChromaRed)); + using Av1EncoderModeInfoBuffer modeInfo = new(Configuration.Default, Width, Height, disallow4x4AllFrames: true); + Av1PictureControlSet pictureTemplate = CreatePicture(modeInfo, colorConfig, use128x128Superblock: false, QIndex); + using Av1EncoderPictureBuffer picture = new( + Configuration.Default, + pictureTemplate.Sequence.SequenceHeader, + pictureTemplate.Parent.FrameHeader, + Width, + Height); + + using Av1EncoderCoefficientBuffer coefficients = new( + Configuration.Default, + pictureTemplate.Sequence.SequenceHeader, + Width, + Height); + + using Av1EncoderSuperblockWorkspace superblockWorkspace = new(Configuration.Default); + using Av1EncoderBlockWorkspace blockWorkspace = new(Configuration.Default); + using Av1IntraTileWriter tileWriter = createWriter( + source.Frame, + reconstruction.Frame, + picture.Picture, + coefficients, + superblockWorkspace, + blockWorkspace); + + Buffer2DRegion actualLuma = reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y); + for (int y = 0; y < reconstructedLuma.Height; y++) + { + Assert.Equal(reconstructedLuma.DangerousGetRowSpan(y), actualLuma.DangerousGetRowSpan(y)); + } + + ref Av1MacroBlockModeInfo targetBlock = ref picture.Picture.GetMacroBlockModeInfo(new Point(2, 2)); + Assert.Equal(Av1ChromaPredictionMode.ChromaFromLuma, targetBlock.Block.UvMode); + Assert.Equal( + Av1ChromaFromLumaMath.JointSign( + Av1ChromaFromLumaMath.SignPositive, + Av1ChromaFromLumaMath.SignNegative), + superblockWorkspace.FinalBlocks[3].PredictionUnit.ChromaFromLumaSigns); + + Assert.Equal( + Av1ChromaFromLumaMath.PackIndices( + Av1ChromaFromLumaMath.AlphaToMagnitudeIndex(AlphaU), + Av1ChromaFromLumaMath.AlphaToMagnitudeIndex(AlphaV)), + superblockWorkspace.FinalBlocks[3].PredictionUnit.ChromaFromLumaIndex); + + int targetTransformIndex = (3 * sampleCount) / + Av1EncoderCoefficientBuffer.TransformBlockUnitCoefficientCount; + + Av1EncoderTransformBlockState blueState = + coefficients.GetTransformBlockSpan(0, Av1Plane.U)[targetTransformIndex]; + + Av1EncoderTransformBlockState redState = + coefficients.GetTransformBlockSpan(0, Av1Plane.V)[targetTransformIndex]; + + Assert.Equal((ushort)0, blueState.EndOfBlock); + Assert.Equal((ushort)0, redState.EndOfBlock); + Assert.Equal(Av1TransformType.DctDct, blueState.TransformType); + Assert.Equal(Av1TransformType.DctDct, redState.TransformType); + Assert.NotEqual(0, pilotWriter.GetTileData(0).Length); + Assert.NotEqual(0, tileWriter.GetTileData(0).Length); + } + [Fact] public void ProductionDirectionalModesConsumeAvailableExtendedEdges() { @@ -1313,7 +1586,8 @@ public class Av1IntraSuperblockEncoderTests } } - private static void FillPlane(Buffer2DRegion plane, byte value) + private static void FillPlane(Buffer2DRegion plane, TSample value) + where TSample : unmanaged { for (int y = 0; y < plane.Height; y++) { @@ -1321,6 +1595,15 @@ public class Av1IntraSuperblockEncoderTests } } + private delegate Av1IntraTileWriter TileWriterFactory( + Av1EncoderFrame source, + Av1EncoderFrame reconstruction, + Av1PictureControlSet picture, + Av1EncoderCoefficientBuffer coefficients, + Av1EncoderSuperblockWorkspace superblockWorkspace, + Av1EncoderBlockWorkspace blockWorkspace) + where TSample : unmanaged; + private static void ClearPlane(Buffer2D plane) where TSample : unmanaged {