From 9aa485e1949dd737526b92a9880872c7cecbbb3d Mon Sep 17 00:00:00 2001 From: James Jackson-South Date: Thu, 3 Sep 2026 04:53:31 +1000 Subject: [PATCH] Add live AV1 luma palette selection --- HEIF_IMPLEMENTATION_PLAN.md | 5 +- .../Av1IntraSuperblockEncoder.ModeDecision.cs | 51 +++ .../Av1IntraSuperblockEncoder.Operator.cs | 151 +++++++ ...raSuperblockEncoder.PaletteModeDecision.cs | 416 ++++++++++++++++++ .../Tiling/Av1EncoderSuperblockWorkspace.cs | 2 +- .../Formats/Heif/Av1/Tiling/Av1TileWriter.cs | 48 +- .../Av1/Av1IntraSuperblockEncoderTests.cs | 178 ++++++++ 7 files changed, 833 insertions(+), 18 deletions(-) create mode 100644 src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.PaletteModeDecision.cs diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 96ba858e54..e00557367f 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -825,11 +825,11 @@ Encoder verification contract: - [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder operation now instantiates matching source and reconstruction owners with ordinary `using` lifetimes and converts packed pixels directly into the source owner before extension. - [~] Temporal delimiter, sequence header, frame header, combined-frame tile-group writing, and an internal reduced-still-picture frame operation now exist locally. The remaining required metadata, padding, multi-tile, option, and public encoder paths are not complete. - [~] Implement superblock and partition analysis for every permitted block size and partition. The current baseline deliberately splits every in-frame node to 8x8 blocks and records decisions in current-libaom writer preorder; block-size selection and non-split partition analysis remain. -- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes. Palette entropy, retained state, and production syntax are complete; palette candidate generation and intra-block-copy mode decision remain. +- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes. Palette entropy, retained state, production syntax, and exhaustive luma palette selection are complete; chroma palette and intra-block-copy mode decisions remain. - [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain. - [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma and filter-intra, now perform live rate-distortion selection; quality mapping, effort-dependent pruning, and the remaining searches are not implemented. -- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these stack-only selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma candidate scratch remains one 8x8 reconstruction and one 8x8 coefficient span on the stack; chroma uses one transform-sized reconstruction and coefficient span for each of U and V. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, 18 focused filter-intra, predictor-reference, syntax-cost, and allocation cases pass, all 8,974 HEIF/AV1 namespace cases pass, and current-main `aomdec` accepts all 29 emitted 8/10/12-bit 4:0:0, 4:2:0, 4:2:2, and 4:4:4 constant or gradient payloads. Remaining mode decision work includes transform-size search, broader joint mode/transform refinement, palette search, partition search, and effort-dependent pruning. Non-empty intra blocks deliberately remain non-skipped, matching current libaom; later inter mode selection owns its distinct skip-transform RD decision. +- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these stack-only selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma candidate scratch remains one 8x8 reconstruction and one 8x8 coefficient span on the stack; chroma uses one transform-sized reconstruction and coefficient span for each of U and V. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. Luma palette selection now evaluates dominant-color and one-dimensional K-means candidates for every legal size, snaps near-cache colors with the reference threshold and tie order, removes duplicate snapped colors, extends boundary maps from active samples, and performs complete transform rate-distortion search. Ordinary DC and filter-intra candidates pay the palette-disabled symbol whenever screen-content syntax is enabled. The exact net11 Release rebuild reports 1,992 test-project warnings and zero errors, all 57 intra-superblock cases pass, all 8,931 AVIF cases pass, and all 230 HEIF cases pass. Remaining mode decision work includes transform-size search, broader joint mode/transform refinement, chroma palette search, partition search, and effort-dependent pruning. Non-empty intra blocks deliberately remain non-skipped, matching current libaom; later inter mode selection owns its distinct skip-transform RD decision. - [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One picture owner now packs segmentation plus every tile's partition, luma, chroma, and transform edges into one clean byte allocation with typed non-owning views; together with the separately typed packed mode-information owner, the complete picture state uses two allocator rents rather than seven. Exact aligned lengths, clean initialization, and balanced exactly-once returns are covered in Release. Multi-tile payload ownership and verified CDF update behavior remain. - [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains. - [~] The final-block decision workspace uses one reusable 8.3 KiB ImageSharp allocator owner. It contains 1,024 explicitly packed 8-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree, replacing separate managed arrays. Palette colors now have their own current-block value and are copied only to the picture edges that later blocks can reference, so enabling palette mode does not add 50 bytes to every final-block entry. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled quantizer, prediction, partition, and current-palette bytes cannot leak into the next decision pass. Complete mode decision still remains. @@ -855,6 +855,7 @@ Encoder verification contract: - [~] Palette color-index map coding now shares the exact current-libaom neighbor weights, stable color ordering, five context classes, first-index uniform code, and diagonal wavefront between encoder costing, encoder writing, and decoder parsing. The decoder's stack-allocated context scores are explicitly cleared before accumulation, removing an invalid dependency on uninitialized stack contents. Costing and writing use a closed generic operation while the shared driver owns traversal and context derivation, so the semantic operations remain independent of map layout and tail handling. The path adds no retained state or per-call managed allocation. Twelve focused map, exact palette decode, padding, trailing-bit, and allocation cases pass; all 1,941 entropy cases and all 8,991 HEIF/AV1 cases pass direct net11 Release VSTest. The exact Release rebuild remains at 1,005 warnings and zero errors. Production payloads remain unchanged because palette selection is still disabled; retained colors, neighbor caches, index-map storage, candidate generation, and production palette mode decision remain incomplete. - [~] Retained encoder palette state and production palette writing now mirror current libaom's 50-byte palette-mode contents, separate luma and shared-chroma sizes, three eight-color planes, above-and-left sorted cache, 64-sample above-cache boundary, mode contexts, palette colors, color-index maps, and syntax order. The current block keeps one inline value in the reusable superblock workspace; only the 4x4-granularity top and left picture edges retain copies for later blocks. For a 3840x2160 tile these edges occupy about 73.4 KiB instead of about 6.2 MiB for a 50-byte palette value attached to every 8x8 mode allocation. Luma and chroma index maps share one lazily allocated 32 KiB owner containing two 128x128 maps, so the palette-disabled production path retains no map owner. The compact final-block workspace falls from about 10.3 KiB to about 8.3 KiB. The writer caps map traversal to the coded plane count, writes maps before transform syntax, and publishes palette edges only after the current block has consumed preceding contexts. Eight focused size, alignment, ownership, cache-boundary, round-trip, map-consumption, and edge-publication cases pass; all 114 palette cases, all 1,942 entropy cases, and all 8,996 HEIF/AV1 cases pass direct foreground net11 Release VSTest. The exact net11 Release rebuild reports 1,050 solution warnings and zero errors; Roslynk reports zero compiler errors and no diagnostics in the touched files. The current-main reference is `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727`. Production payloads remain unchanged because palette candidate generation is still disabled; that live rate-distortion search is the next checkpoint. - [~] Luma palette clustering now follows current libaom's one-dimensional search primitive exactly: equal-interval midpoint initialization, first-color tie order, rounded centroid means, deterministic empty-cluster replacement, the 50-iteration limit, and retention of the preceding state when distortion increases. Nearest-color assignment improves on libaom's AVX2 implementation by dispatching Vector512, Vector256, Vector128, then scalar through ImageSharp's shared vector-count helpers. The primitive uses only bounded stack scratch and introduces no allocator rent, managed array, or per-row copy. Three independent tests cover exact centroid convergence, initialization order, 12-bit nearest-color distortion, destination bounds, and every hardware-intrinsic tier. The complete AVIF set passes 8,930 of 8,930 cases and the HEIF set passes 230 of 230 cases through direct foreground net11 Release VSTest. The exact net11 Release rebuild reports 1,050 solution warnings and zero errors, and Roslynk reports zero compiler errors. Candidate enumeration, palette-cache snapping, transform RD selection, and production activation remain in the open luma-palette checkpoint. +- [~] Live luma palette selection now follows current libaom's dominant-color and one-dimensional K-means candidate families, cache-bias threshold, sorted duplicate removal, active-edge map extension, and strict winner tie order. It improves on speed-configured libaom by evaluating both candidate families at every legal 2-through-8 size without early pruning, then exhaustively evaluates every legal transform using the existing SIMD prediction, residual, transform, quantization, and reconstruction operators. Candidate storage remains bounded stack memory; the reusable 32 KiB map owner is allocated only when an eligible block enters palette search. A production tile test proves that full 8x8 and clipped 5x3 blocks at 8 and 12 bits select exact colors and indices, extend the visible edges through coded padding, reconstruct every sample without coefficients, and emit a nonempty tile. The complete 57-case intra-superblock set, 8,931-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains gated until chroma palette mode and its rate accounting are complete. - [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips. - [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips. - [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips. diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs index 24f86b4159..949ab13b39 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs @@ -193,6 +193,7 @@ internal static partial class Av1IntraSuperblockEncoder tileIndex, lumaCoefficients[this.codedAreaLuma..], ref lumaState, + ref paletteInfo, out int lumaAngleDelta, out Av1FilterIntraMode filterIntraMode); @@ -348,6 +349,7 @@ internal static partial class Av1IntraSuperblockEncoder ushort tileIndex, Span retainedCoefficients, ref Av1EncoderTransformBlockState retainedState, + ref Av1EncoderPaletteInfo paletteInfo, out int selectedAngleDelta, out Av1FilterIntraMode selectedFilterIntraMode) { @@ -463,6 +465,22 @@ internal static partial class Av1IntraSuperblockEncoder BlockSize, TransformSize); + int paletteDisabledCost = 0; + if (this.picture.Parent.FrameHeader.AllowScreenContentTools) + { + Av1NeighborArrayUnit paletteContexts = this.picture.PaletteContexts[tileIndex]; + int blockSizeContext = Av1TileWriter.GetPaletteBlockSizeContext(BlockSize); + int neighborContext = Av1TileWriter.GetPaletteYModeContext( + paletteContexts, + macroBlock, + blockOrigin); + + paletteDisabledCost = writer.GetPaletteYModeCost( + false, + blockSizeContext, + neighborContext); + } + Span candidateReconstruction = stackalloc TSample[SampleCount]; Span candidateCoefficients = stackalloc int[SampleCount]; long bestCost = long.MaxValue; @@ -515,6 +533,7 @@ internal static partial class Av1IntraSuperblockEncoder angleDelta, defaultTransformType, blockContext, + paletteDisabledCost, candidateReconstruction, candidateCoefficients, ref candidateState); @@ -563,6 +582,7 @@ internal static partial class Av1IntraSuperblockEncoder selectedAngleDelta, transformType, blockContext, + paletteDisabledCost, candidateReconstruction, candidateCoefficients, ref candidateState); @@ -626,6 +646,7 @@ internal static partial class Av1IntraSuperblockEncoder filterIntraMode, transformType, blockContext, + paletteDisabledCost, candidateReconstruction, candidateCoefficients, ref candidateState); @@ -651,6 +672,28 @@ internal static partial class Av1IntraSuperblockEncoder } } + if (this.picture.Parent.FrameHeader.AllowScreenContentTools && + this.SelectLumaPalette( + writer, + macroBlock, + sourcePlane, + reconstructionPlane, + blockOrigin, + tileIndex, + transformSetType, + blockContext, + candidateReconstruction, + candidateCoefficients, + retainedCoefficients, + ref retainedState, + ref bestTransformCost, + ref paletteInfo)) + { + bestMode = Av1PredictionMode.DC; + selectedAngleDelta = 0; + selectedFilterIntraMode = Av1FilterIntraMode.AllFilterIntraModes; + } + return bestMode; } @@ -667,6 +710,7 @@ internal static partial class Av1IntraSuperblockEncoder int angleDelta, Av1TransformType transformType, Av1TransformBlockContext blockContext, + int paletteDisabledCost, Span candidateReconstruction, Span candidateCoefficients, ref Av1EncoderTransformBlockState candidateState) @@ -695,6 +739,11 @@ internal static partial class Av1IntraSuperblockEncoder ref candidateState); int rate = Av1TileWriter.GetLumaModeCost(writer, macroBlock, BlockSize, mode, angleDelta); + if (mode == Av1PredictionMode.DC) + { + rate += paletteDisabledCost; + } + if (mode == Av1PredictionMode.DC && this.picture.Sequence.SequenceHeader.EnableFilterIntra) { rate += writer.GetFilterIntraModeCost(Av1FilterIntraMode.AllFilterIntraModes, BlockSize); @@ -724,6 +773,7 @@ internal static partial class Av1IntraSuperblockEncoder Av1FilterIntraMode filterIntraMode, Av1TransformType transformType, Av1TransformBlockContext blockContext, + int paletteDisabledCost, Span candidateReconstruction, Span candidateCoefficients, ref Av1EncoderTransformBlockState candidateState) @@ -754,6 +804,7 @@ internal static partial class Av1IntraSuperblockEncoder Av1PredictionMode.DC, 0); + rate += paletteDisabledCost; rate += writer.GetFilterIntraModeCost(filterIntraMode, BlockSize); rate += writer.GetCoefficientCost( TransformSize, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs index ebb78069c8..6e327ef80b 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs @@ -2,6 +2,7 @@ // Licensed under the Six Labors Split License. using System.Runtime.InteropServices; +using System.Runtime.Intrinsics; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; @@ -37,6 +38,40 @@ internal static partial class Av1IntraSuperblockEncoder /// The converted sample. public static abstract TSample CreateSample(int value); + /// + /// Copies active palette-search samples into contiguous signed storage. + /// + /// The coded source plane. + /// The block origin in plane samples. + /// The active row count. + /// The active column count. + /// The contiguous sample destination. + public static abstract void CopyPaletteSamples( + Buffer2DRegion source, + Point blockOrigin, + int rows, + int columns, + Span samples); + + /// + /// Builds palette prediction and the matching source residual for transform search. + /// + /// The coded source plane. + /// The block origin in plane samples. + /// The palette colors in index order. + /// The complete padded color-index map. + /// The contiguous prediction destination. + /// The contiguous source-minus-prediction destination. + /// The prediction dimensions. + public static abstract void PreparePalette( + Buffer2DRegion source, + Point blockOrigin, + ReadOnlySpan paletteColors, + Buffer2DRegion colorIndexMap, + Span prediction, + Span residual, + Av1TransformSize transformSize); + /// /// Builds the zero-mean Q3 luma surface shared by chroma-from-luma candidates. /// @@ -264,6 +299,71 @@ internal static partial class Av1IntraSuperblockEncoder /// public static byte CreateSample(int value) => (byte)value; + /// + public static void CopyPaletteSamples( + Buffer2DRegion source, + Point blockOrigin, + int rows, + int columns, + Span samples) + { + int sampleOffset = 0; + for (int row = 0; row < rows; row++) + { + ReadOnlySpan sourceRow = source + .DangerousGetRowSpan(blockOrigin.Y + row) + .Slice(blockOrigin.X, columns); + + // A complete row widens in one vector; clipped edge rows retain scalar bounds. + if (columns == 8 && Vector128.IsHardwareAccelerated) + { + ulong packed = MemoryMarshal.Read(sourceRow); + Vector128.WidenLower(Vector128.CreateScalarUnsafe(packed).AsByte()) + .AsInt16() + .CopyTo(samples[sampleOffset..]); + + sampleOffset += columns; + continue; + } + + for (int column = 0; column < sourceRow.Length; column++) + { + samples[sampleOffset++] = sourceRow[column]; + } + } + } + + /// + public static void PreparePalette( + Buffer2DRegion source, + Point blockOrigin, + ReadOnlySpan paletteColors, + Buffer2DRegion colorIndexMap, + Span prediction, + Span residual, + Av1TransformSize transformSize) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + Av1PalettePredictor.Predict( + paletteColors, + colorIndexMap, + prediction, + width, + width, + height); + + Av1ResidualBuilder.Subtract( + Av1TransformBlockEncoder.GetPlaneSpan(source, blockOrigin), + source.Stride, + prediction, + width, + residual, + width, + width, + height); + } + /// public static void PrepareChromaFromLuma( Buffer2DRegion reconstruction, @@ -493,6 +593,57 @@ internal static partial class Av1IntraSuperblockEncoder /// public static ushort CreateSample(int value) => (ushort)value; + /// + public static void CopyPaletteSamples( + Buffer2DRegion source, + Point blockOrigin, + int rows, + int columns, + Span samples) + { + int sampleOffset = 0; + for (int row = 0; row < rows; row++) + { + ReadOnlySpan sourceRow = source + .DangerousGetRowSpan(blockOrigin.Y + row) + .Slice(blockOrigin.X, columns); + + MemoryMarshal.Cast(sourceRow).CopyTo(samples[sampleOffset..]); + sampleOffset += sourceRow.Length; + } + } + + /// + public static void PreparePalette( + Buffer2DRegion source, + Point blockOrigin, + ReadOnlySpan paletteColors, + Buffer2DRegion colorIndexMap, + Span prediction, + Span residual, + Av1TransformSize transformSize) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + Av1PalettePredictor.Predict( + paletteColors, + colorIndexMap, + MemoryMarshal.Cast(prediction), + width, + width, + height); + + Av1ResidualBuilder.Subtract( + Av1TransformBlockEncoder.GetPlaneSpan(source, blockOrigin), + source.Stride, + prediction, + width, + residual, + width, + width, + height); + } + /// public static void PrepareChromaFromLuma( Buffer2DRegion reconstruction, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.PaletteModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.PaletteModeDecision.cs new file mode 100644 index 0000000000..f99bb81968 --- /dev/null +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.PaletteModeDecision.cs @@ -0,0 +1,416 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; +using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; +using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; +using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; +using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; +using SixLabors.ImageSharp.Memory; + +namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; + +/// +/// Provides luma palette mode decisions for intra encoding. +/// +internal static partial class Av1IntraSuperblockEncoder +{ + internal partial struct ModeDecision + where TSample : unmanaged + where TOperator : struct, IBlockEncodingOperator + { + private bool SelectLumaPalette( + Av1SymbolEncoder writer, + Av1MacroBlockD macroBlock, + Buffer2DRegion sourcePlane, + Buffer2DRegion reconstructionPlane, + Point blockOrigin, + ushort tileIndex, + Av1TransformSetType transformSetType, + Av1TransformBlockContext blockContext, + Span candidateReconstruction, + Span candidateCoefficients, + Span retainedCoefficients, + ref Av1EncoderTransformBlockState retainedState, + ref long bestCost, + ref Av1EncoderPaletteInfo paletteInfo) + { + const Av1BlockSize BlockSize = Av1BlockSize.Block8x8; + const int BlockLength = 8; + const int SampleCapacity = BlockLength * BlockLength; + ObuFrameSize frameSize = this.picture.Parent.FrameHeader.FrameSize; + int rows = Math.Min(BlockLength, frameSize.FrameHeight - blockOrigin.Y); + int columns = Math.Min(BlockLength, frameSize.FrameWidth - blockOrigin.X); + int sampleCount = rows * columns; + Span samples = stackalloc short[SampleCapacity]; + samples = samples[..sampleCount]; + TOperator.CopyPaletteSamples(sourcePlane, blockOrigin, rows, columns, samples); + + Span uniqueColors = stackalloc short[SampleCapacity]; + Span colorCounts = stackalloc int[SampleCapacity]; + int uniqueColorCount = 0; + short minimum = samples[0]; + short maximum = samples[0]; + foreach (short sample in samples) + { + int colorIndex = uniqueColors[..uniqueColorCount].IndexOf(sample); + if (colorIndex >= 0) + { + colorCounts[colorIndex]++; + } + else + { + uniqueColors[uniqueColorCount] = sample; + colorCounts[uniqueColorCount] = 1; + uniqueColorCount++; + } + + minimum = Math.Min(minimum, sample); + maximum = Math.Max(maximum, sample); + } + + if (uniqueColorCount < 2) + { + return false; + } + + int maximumPaletteSize = Math.Min(uniqueColorCount, Av1Constants.PaletteMaxSize); + Span dominantOrder = stackalloc byte[SampleCapacity]; + for (int index = 0; index < uniqueColorCount; index++) + { + dominantOrder[index] = (byte)index; + } + + // Count order chooses the colors that explain most samples first; sample value resolves equal counts. + for (int index = 1; index < uniqueColorCount; index++) + { + byte current = dominantOrder[index]; + int destination = index; + while (destination > 0) + { + byte preceding = dominantOrder[destination - 1]; + bool precedes = colorCounts[current] > colorCounts[preceding] || + (colorCounts[current] == colorCounts[preceding] && uniqueColors[current] < uniqueColors[preceding]); + + if (!precedes) + { + break; + } + + dominantOrder[destination] = preceding; + destination--; + } + + dominantOrder[destination] = current; + } + + Av1NeighborArrayUnit paletteContexts = this.picture.PaletteContexts[tileIndex]; + int blockSizeContext = Av1TileWriter.GetPaletteBlockSizeContext(BlockSize); + int neighborContext = Av1TileWriter.GetPaletteYModeContext(paletteContexts, macroBlock, blockOrigin); + Span colorCache = stackalloc ushort[2 * Av1Constants.PaletteMaxSize]; + int colorCacheSize = Av1TileWriter.GetPaletteCache( + paletteContexts, + macroBlock, + blockOrigin, + Av1Plane.Y, + colorCache); + + Buffer2DRegion colorIndexMap = this.superblock.Workspace + .GetPaletteMaps() + .GetMap(Av1PlaneType.Y, BlockLength, BlockLength); + + Span retainedColorIndexMap = stackalloc byte[SampleCapacity]; + Span centroids = stackalloc short[Av1Constants.PaletteMaxSize]; + bool paletteSelected = false; + + // Exhaustive ascending size search avoids the reference encoder's speed-dependent pruning. + for (int paletteSize = 2; paletteSize <= maximumPaletteSize; paletteSize++) + { + for (int index = 0; index < paletteSize; index++) + { + centroids[index] = uniqueColors[dominantOrder[index]]; + } + + this.EvaluateLumaPaletteCandidate( + writer, + macroBlock, + blockOrigin, + transformSetType, + blockContext, + samples, + rows, + columns, + colorCache[..colorCacheSize], + blockSizeContext, + neighborContext, + centroids[..paletteSize], + colorIndexMap, + candidateReconstruction, + candidateCoefficients, + retainedCoefficients, + retainedColorIndexMap, + reconstructionPlane, + ref retainedState, + ref bestCost, + ref paletteInfo, + ref paletteSelected); + } + + if (uniqueColorCount == 2) + { + centroids[0] = minimum; + centroids[1] = maximum; + this.EvaluateLumaPaletteCandidate( + writer, + macroBlock, + blockOrigin, + transformSetType, + blockContext, + samples, + rows, + columns, + colorCache[..colorCacheSize], + blockSizeContext, + neighborContext, + centroids[..2], + colorIndexMap, + candidateReconstruction, + candidateCoefficients, + retainedCoefficients, + retainedColorIndexMap, + reconstructionPlane, + ref retainedState, + ref bestCost, + ref paletteInfo, + ref paletteSelected); + } + else + { + Span clusterIndices = stackalloc byte[SampleCapacity]; + clusterIndices = clusterIndices[..sampleCount]; + for (int paletteSize = 2; paletteSize <= maximumPaletteSize; paletteSize++) + { + Span candidateCentroids = centroids[..paletteSize]; + Av1PaletteKMeans.InitializeCentroids(minimum, maximum, candidateCentroids); + Av1PaletteKMeans.Cluster(samples, candidateCentroids, clusterIndices); + this.EvaluateLumaPaletteCandidate( + writer, + macroBlock, + blockOrigin, + transformSetType, + blockContext, + samples, + rows, + columns, + colorCache[..colorCacheSize], + blockSizeContext, + neighborContext, + candidateCentroids, + colorIndexMap, + candidateReconstruction, + candidateCoefficients, + retainedCoefficients, + retainedColorIndexMap, + reconstructionPlane, + ref retainedState, + ref bestCost, + ref paletteInfo, + ref paletteSelected); + } + } + + if (paletteSelected) + { + for (int row = 0; row < BlockLength; row++) + { + retainedColorIndexMap.Slice(row * BlockLength, BlockLength) + .CopyTo(colorIndexMap.DangerousGetRowSpan(row)); + } + } + + return paletteSelected; + } + + private void EvaluateLumaPaletteCandidate( + Av1SymbolEncoder writer, + Av1MacroBlockD macroBlock, + Point blockOrigin, + Av1TransformSetType transformSetType, + Av1TransformBlockContext blockContext, + ReadOnlySpan samples, + int rows, + int columns, + ReadOnlySpan colorCache, + int blockSizeContext, + int neighborContext, + Span centroids, + Buffer2DRegion colorIndexMap, + Span candidateReconstruction, + Span candidateCoefficients, + Span retainedCoefficients, + Span retainedColorIndexMap, + Buffer2DRegion reconstructionPlane, + ref Av1EncoderTransformBlockState retainedState, + ref long bestCost, + ref Av1EncoderPaletteInfo paletteInfo, + ref bool paletteSelected) + { + const Av1BlockSize BlockSize = Av1BlockSize.Block8x8; + const Av1TransformSize TransformSize = Av1TransformSize.Size8x8; + const int BlockLength = 8; + const int SampleCount = BlockLength * BlockLength; + int bitDepth = this.bitDepth.GetBitCount(); + int cacheThreshold = 4 << (bitDepth - 8); + for (int colorIndex = 0; colorIndex < centroids.Length && !colorCache.IsEmpty; colorIndex++) + { + int minimumDifference = Math.Abs(centroids[colorIndex] - colorCache[0]); + int nearestCacheIndex = 0; + for (int cacheIndex = 1; cacheIndex < colorCache.Length; cacheIndex++) + { + int difference = Math.Abs(centroids[colorIndex] - colorCache[cacheIndex]); + if (difference < minimumDifference) + { + minimumDifference = difference; + nearestCacheIndex = cacheIndex; + } + } + + if (minimumDifference <= cacheThreshold) + { + centroids[colorIndex] = (short)colorCache[nearestCacheIndex]; + } + } + + centroids.Sort(); + int paletteSize = 1; + for (int colorIndex = 1; colorIndex < centroids.Length; colorIndex++) + { + if (centroids[colorIndex] != centroids[colorIndex - 1]) + { + centroids[paletteSize++] = centroids[colorIndex]; + } + } + + if (paletteSize < 2) + { + return; + } + + ReadOnlySpan paletteCentroids = centroids[..paletteSize]; + Span paletteColors = stackalloc ushort[Av1Constants.PaletteMaxSize]; + paletteColors = paletteColors[..paletteSize]; + for (int colorIndex = 0; colorIndex < paletteSize; colorIndex++) + { + paletteColors[colorIndex] = (ushort)paletteCentroids[colorIndex]; + } + + Span colorIndices = stackalloc byte[SampleCount]; + Av1PaletteKMeans.AssignIndices(samples, paletteCentroids, colorIndices); + for (int row = 0; row < rows; row++) + { + Span mapRow = colorIndexMap.DangerousGetRowSpan(row)[..BlockLength]; + colorIndices.Slice(row * columns, columns).CopyTo(mapRow); + mapRow[columns..].Fill(mapRow[columns - 1]); + } + + // Padding repeats the last active edge so transform prediction matches coded-frame edge extension. + for (int row = rows; row < BlockLength; row++) + { + colorIndexMap.DangerousGetRowSpan(rows - 1)[..BlockLength] + .CopyTo(colorIndexMap.DangerousGetRowSpan(row)); + } + + Span prediction = stackalloc TSample[SampleCount]; + Span residual = stackalloc short[SampleCount]; + TOperator.PreparePalette( + this.source.GetPlane(Av1Plane.Y), + blockOrigin, + paletteColors, + colorIndexMap, + prediction, + residual, + TransformSize); + + int rate = Av1TileWriter.GetLumaModeCost( + writer, + macroBlock, + BlockSize, + Av1PredictionMode.DC, + 0); + + rate += writer.GetPaletteYModeCost(true, blockSizeContext, neighborContext); + rate += writer.GetPaletteSizeCost(paletteSize, blockSizeContext, Av1PlaneType.Y); + rate += Av1SymbolEncoder.GetPaletteYColorCost(colorCache, paletteColors, bitDepth); + rate += writer.GetPaletteColorMapCost( + paletteSize, + Av1PlaneType.Y, + rows, + columns, + colorIndexMap); + + for (Av1TransformType transformType = Av1TransformType.DctDct; + transformType < Av1TransformType.AllTransformTypes; + transformType++) + { + if (!transformType.IsExtendedSetUsed(transformSetType)) + { + continue; + } + + Av1EncoderTransformBlockState candidateState = default; + long distortion = TOperator.EncodePredictionCandidate( + this.blockWorkspace, + this.source.GetPlane(Av1Plane.Y), + blockOrigin, + prediction, + residual, + candidateReconstruction, + candidateCoefficients, + TransformSize, + transformType, + Av1Plane.Y, + this.quantization.QIndex[0], + this.quantization.DeltaQDc[(int)Av1Plane.Y], + this.quantization.DeltaQAc[(int)Av1Plane.Y], + this.bitDepth, + ref candidateState); + + int candidateRate = rate + writer.GetCoefficientCost( + TransformSize, + transformType, + Av1PredictionMode.DC, + candidateCoefficients, + Av1ComponentType.Luminance, + blockContext, + candidateState.EndOfBlock, + this.picture.Parent.FrameHeader.UseReducedTransformSet, + Av1FilterIntraMode.AllFilterIntraModes); + + long candidateCost = Av1RateDistortion.GetCost(this.rateMultiplier, candidateRate, distortion); + if (candidateCost < bestCost) + { + CopyCandidate( + candidateReconstruction, + candidateCoefficients, + reconstructionPlane, + blockOrigin, + retainedCoefficients, + TransformSize, + candidateState, + ref retainedState); + + for (int row = 0; row < BlockLength; row++) + { + colorIndexMap.DangerousGetRowSpan(row)[..BlockLength] + .CopyTo(retainedColorIndexMap[(row * BlockLength)..]); + } + + paletteInfo.PaletteSizes[0] = (byte)paletteSize; + paletteInfo.SetColors(Av1Plane.Y, paletteColors); + bestCost = candidateCost; + paletteSelected = true; + } + } + } + } +} diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderSuperblockWorkspace.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderSuperblockWorkspace.cs index 387d213023..f92caf43b0 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderSuperblockWorkspace.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderSuperblockWorkspace.cs @@ -59,7 +59,7 @@ internal sealed class Av1EncoderSuperblockWorkspace : IDisposable public ref Av1EncoderPaletteInfo PaletteInfo => ref this.paletteInfo; /// - /// Gets the reusable palette maps, allocating their shared owner only after a block selects palette mode. + /// Gets the reusable palette maps, allocating their shared owner only after a block enters palette search. /// /// The reusable luma and chroma palette maps. public Av1EncoderPaletteMapBuffer GetPaletteMaps() diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs index 5bad7f31b7..9ac43dc134 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs @@ -1157,24 +1157,12 @@ internal partial class Av1TileWriter int tileIndex, bool hasChroma) { - int blockSizeContext = Av1Math.Log2(blockSize.GetWidth() * blockSize.GetHeight()) - 6; + int blockSizeContext = GetPaletteBlockSizeContext(blockSize); Av1NeighborArrayUnit paletteContexts = pcs.PaletteContexts[tileIndex]; int yPaletteSize = paletteInfo.PaletteSizes[0]; if (macroBlockModeInfo.Block.Mode == Av1PredictionMode.DC) { - int neighborContext = 0; - if (macroBlock.IsUpAvailable && - paletteContexts.Top[paletteContexts.GetTopIndex(blockOrigin)].PaletteSizes[0] != 0) - { - neighborContext++; - } - - if (macroBlock.IsLeftAvailable && - paletteContexts.Left[paletteContexts.GetLeftIndex(blockOrigin)].PaletteSizes[0] != 0) - { - neighborContext++; - } - + int neighborContext = GetPaletteYModeContext(paletteContexts, macroBlock, blockOrigin); writer.WritePaletteYMode(yPaletteSize != 0, blockSizeContext, neighborContext); if (yPaletteSize != 0) { @@ -1223,7 +1211,7 @@ internal partial class Av1TileWriter /// /// Builds the sorted palette-color cache from the available above and left encoder edges. /// - private static int GetPaletteCache( + internal static int GetPaletteCache( Av1NeighborArrayUnit paletteContexts, Av1MacroBlockD macroBlock, Point blockOrigin, @@ -1247,6 +1235,36 @@ internal partial class Av1TileWriter return Av1PaletteCache.Merge(aboveColors, leftColors, cache); } + /// + /// Gets the palette probability context derived from the logarithmic block area. + /// + internal static int GetPaletteBlockSizeContext(Av1BlockSize blockSize) + => Av1Math.Log2(blockSize.GetWidth() * blockSize.GetHeight()) - 6; + + /// + /// Counts the available above and left luma neighbors that selected palette mode. + /// + internal static int GetPaletteYModeContext( + Av1NeighborArrayUnit paletteContexts, + Av1MacroBlockD macroBlock, + Point blockOrigin) + { + int neighborContext = 0; + if (macroBlock.IsUpAvailable && + paletteContexts.Top[paletteContexts.GetTopIndex(blockOrigin)].PaletteSizes[0] != 0) + { + neighborContext++; + } + + if (macroBlock.IsLeftAvailable && + paletteContexts.Left[paletteContexts.GetLeftIndex(blockOrigin)].PaletteSizes[0] != 0) + { + neighborContext++; + } + + return neighborContext; + } + /// /// Determines whether filter-intra syntax is available for a block mode. /// diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs index cd01d1e6dc..b5fb8ff788 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs @@ -716,6 +716,70 @@ public class Av1IntraSuperblockEncoderTests Assert.False(payloads[0].SequenceEqual(payloads[1])); } + [Fact] + public void ProductionTileSelectsExactLumaPaletteAtFullAndClippedSizes() + { + AssertProductionTileSelectsExactLumaPalette( + Av1BitDepth.EightBit, + 8, + 8, + 8, + (byte)32, + (byte)224, + 32, + 224, + static (source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new Av1IntraTileWriter( + Configuration.Default, + source, + reconstruction, + picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialSize: 256)); + + AssertProductionTileSelectsExactLumaPalette( + Av1BitDepth.TwelveBit, + 12, + 8, + 8, + (ushort)512, + (ushort)3584, + 512, + 3584, + static (source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new Av1IntraTileWriter( + Configuration.Default, + source, + reconstruction, + picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialSize: 256)); + + AssertProductionTileSelectsExactLumaPalette( + Av1BitDepth.EightBit, + 8, + 5, + 3, + (byte)48, + (byte)208, + 48, + 208, + static (source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new Av1IntraTileWriter( + Configuration.Default, + source, + reconstruction, + picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialSize: 256)); + } + [Theory] [InlineData((int)Av1PredictionMode.Vertical, 0)] [InlineData((int)Av1PredictionMode.Horizontal, 0)] @@ -1835,6 +1899,120 @@ public class Av1IntraSuperblockEncoderTests }; } + private static void AssertProductionTileSelectsExactLumaPalette( + Av1BitDepth bitDepth, + int bitDepthValue, + int width, + int height, + TSample lowerColor, + TSample upperColor, + ushort expectedLowerColor, + ushort expectedUpperColor, + TileWriterFactory createTileWriter) + where TSample : unmanaged + { + const int QIndex = 37; + ObuColorConfig colorConfig = new() + { + IsMonochrome = true, + SubSamplingX = true, + SubSamplingY = true, + BitDepth = bitDepth + }; + + using Av1EncoderFrameBuffer source = new( + Configuration.Default, + width, + height, + bitDepthValue, + Av1ColorFormat.Yuv400, + 0, + 0); + + using Av1EncoderFrameBuffer reconstruction = new( + Configuration.Default, + width, + height, + bitDepthValue, + Av1ColorFormat.Yuv400, + 0, + 0); + + Buffer2DRegion sourcePlane = source.Frame.CodedView.GetPlane(Av1Plane.Y); + for (int row = 0; row < sourcePlane.Height; row++) + { + int visibleRow = Math.Min(row, height - 1); + sourcePlane.DangerousGetRowSpan(row).Fill(visibleRow < height / 2 ? lowerColor : upperColor); + } + + ClearPlane(reconstruction.Luma); + using Av1EncoderModeInfoBuffer modeInfo = new( + Configuration.Default, + width, + height, + disallow4x4AllFrames: true); + + Av1PictureControlSet pictureTemplate = CreatePicture( + modeInfo, + colorConfig, + use128x128Superblock: false, + QIndex); + + pictureTemplate.Parent.FrameHeader.AllowScreenContentTools = true; + pictureTemplate.Parent.FrameHeader.FrameSize.FrameWidth = width; + pictureTemplate.Parent.FrameHeader.FrameSize.FrameHeight = height; + using Av1EncoderPictureBuffer picture = new( + Configuration.Default, + pictureTemplate.Sequence.SequenceHeader, + pictureTemplate.Parent.FrameHeader, + width, + height); + + using Av1EncoderCoefficientBuffer coefficients = new( + Configuration.Default, + pictureTemplate.Sequence.SequenceHeader, + width, + height); + + using Av1EncoderSuperblockWorkspace superblockWorkspace = new(Configuration.Default); + using Av1EncoderBlockWorkspace blockWorkspace = new(Configuration.Default); + using Av1IntraTileWriter tileWriter = createTileWriter( + source.Frame, + reconstruction.Frame, + picture.Picture, + coefficients, + superblockWorkspace, + blockWorkspace); + + ref Av1MacroBlockModeInfo mode = ref picture.Picture.GetMacroBlockModeInfo(default); + Assert.Equal(Av1PredictionMode.DC, mode.Block.Mode); + Assert.Equal(Av1FilterIntraMode.AllFilterIntraModes, superblockWorkspace.FinalBlocks[0].FilterIntraMode); + Assert.Equal((ushort)0, coefficients.GetTransformBlockSpan(0, Av1Plane.Y)[0].EndOfBlock); + Assert.Equal(2, superblockWorkspace.PaletteInfo.PaletteSizes[0]); + Assert.Equal( + [expectedLowerColor, expectedUpperColor], + superblockWorkspace.PaletteInfo.GetColors(Av1Plane.Y).ToArray()); + + Buffer2DRegion colorIndexMap = superblockWorkspace + .GetPaletteMaps() + .GetMap(Av1PlaneType.Y, 8, 8); + + Buffer2DRegion reconstructionPlane = reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y); + for (int row = 0; row < reconstructionPlane.Height; row++) + { + int visibleRow = Math.Min(row, height - 1); + byte expectedIndex = (byte)(visibleRow < height / 2 ? 0 : 1); + foreach (byte index in colorIndexMap.DangerousGetRowSpan(row)) + { + Assert.Equal(expectedIndex, index); + } + + Assert.True(sourcePlane.DangerousGetRowSpan(row).SequenceEqual(reconstructionPlane.DangerousGetRowSpan(row))); + } + + Assert.NotEqual(0, tileWriter.GetTileData(0).Length); + } + private static void FillChromaModeSelectionPlane( Buffer2DRegion plane, Av1TransformSize transformSize,