From 777f63ba83beb2ac08d3b87d54e9e14475457b84 Mon Sep 17 00:00:00 2001 From: James Jackson-South Date: Wed, 2 Sep 2026 20:14:57 +1000 Subject: [PATCH] Encode monochrome AV1 frame OBUs --- HEIF_IMPLEMENTATION_PLAN.md | 12 +- .../Heif/Av1/Pipeline/Av1FrameEncoder.cs | 260 ++++++++++++++++++ .../Formats/Heif/Av1/Tiling/Av1LevelBuffer.cs | 5 +- .../Heif/Av1/Av1CoefficientsEntropyTests.cs | 77 +++++- .../Formats/Heif/Av1/Av1EncoderFrameTests.cs | 79 ++++++ .../Formats/Heif/Av1/Av1LevelBufferTests.cs | 15 + 6 files changed, 433 insertions(+), 15 deletions(-) diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 7ddd1061e0..c14810f488 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -16,8 +16,8 @@ This plan is the authoritative delivery checklist. A source file, unit test, bui Reference checkout evidence on 2026-08-31: -- `D:\GitHub\AOMediaCodec\aom` is attached to `main`, clean, and aligned with `origin/main` after a fresh fetch. -- Both `HEAD` and `origin/main` resolved to `441c439b9916474cac15d2822af47a9ad70674a8`. This records the tree audited on that date; it is not a pin and must not prevent later work from updating to the then-current `main`. +- `D:\GitHub\AOMediaCodec\aom` is clean. Its checked-out `HEAD` is `441c439b9916474cac15d2822af47a9ad70674a8`, while the refreshed `origin/main` is `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727`. +- Current encoder verification uses the exact `origin/main` tree exported to `D:\GitHub\ynse01\aom-main-a40` and built in `D:\GitHub\ynse01\aom-main-a40-build`. The resulting `aomdec` identifies itself as version 3.15.0. This records the tree audited on that date; it is not a pin and must not prevent later work from updating to the then-current `main`. ## Status notation @@ -821,8 +821,8 @@ Encoder verification contract: - [~] SIMD-first RGB-to-native-plane conversion now feeds eight-bit and high-bit-depth bordered AV1 source frames directly, preserving ImageSharp's arbitrary packed-pixel input contract without an intermediate full-frame native-plane copy. - [~] Forward transform families, transform workspace, and an allocation-free DC intra block boundary exist locally. For eight-bit and high-bit-depth samples, the composed boundary now follows current libaom's encoder order: predict into the reconstruction plane, subtract prediction from source, transform, quantize into separate qcoeff and dqcoeff storage, retain EOB and transform type, and inverse-transform only when EOB is nonzero so later blocks consume decoder-identical references. Prediction and subtraction retain their SIMD-first operators, independent source and reconstruction strides are preserved, and no frame-sized or per-block buffer is introduced. The block boundary consumes the real bordered encoder-plane regions and indexes their one-segment owner directly; this preserves physical row strides without a row copy and avoids the per-call enumerator allocation exposed by the initial array-only test. One reusable 61 KiB allocator owner supplies tightly packed residual, aligned transform-coefficient, dequantized-coefficient, and transform scratch spans across transform blocks; quantized coefficients write directly to the retained frame coefficient owner instead of being duplicated. A fixed 8x8 DC-intra superblock baseline now traverses the same recursive preorder and frame-edge pruning as the tile writer, gathers left references into that reusable block workspace, writes luma and chroma coefficient-owner slices in the writer's exact consumption order, and updates the caller-owned reconstruction planes for subsequent predictions. Stage-by-stage scalar-oracle, physical-border, retained-syntax, superblock-to-writer synchronization, high-bit-depth precision, and steady-state zero-allocation coverage passes 8 of 8 through direct net11 VSTest in Release. This is a legal fixed baseline, not complete partition or mode analysis. - [~] A production single-tile all-intra writer now walks raster superblocks, analyzes each immediately before entropy coding, reuses one decision workspace and one block workspace, and retains decoder-identical reconstructed references across the tile. Its closed byte and high-bit-depth operators feed the existing superblock boundary without runtime sample-type checks. A byte-exact test compares this composed path with an explicit superblock-then-tile-writer oracle, so producer and writer traversal or coefficient-area drift cannot pass unnoticed. A separate clipped 2x2-superblock regression proves global raster indexing by requiring all four coefficient segments and the bottom-right reconstruction to be populated. Multi-tile ownership and the complete frame/OBU operation remain. -- [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder boundary converts packed pixels directly into the source owner before extension; the containing encode operation still needs to instantiate matching source and reconstruction owners with ordinary `using` lifetimes. -- [~] Temporal delimiter, sequence header, frame header, and combined-frame tile-group writing exist locally. The remaining required metadata, padding, and encoder-wide syntax paths are not complete. +- [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder operation now instantiates matching source and reconstruction owners with ordinary `using` lifetimes and converts packed pixels directly into the source owner before extension. +- [~] Temporal delimiter, sequence header, frame header, combined-frame tile-group writing, and an internal reduced-still-picture frame operation now exist locally. The remaining required metadata, padding, multi-tile, option, and public encoder paths are not complete. - [~] Implement superblock and partition analysis for every permitted block size and partition. The current baseline deliberately splits every in-frame node to 8x8 blocks and records decisions in current-libaom writer preorder; block-size selection and non-split partition analysis remain. - [ ] Implement intra mode search, chroma mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. - [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. @@ -834,12 +834,14 @@ Encoder verification contract: - [~] Finalized transform coefficients and packed EOB/type state now use raster-ordered, per-superblock plane segments matching current libaom's coefficient-pool geometry. One ImageSharp allocator owner replaces libaom's separate coefficient, EOB, and entropy-context allocations while preserving the full 1024 luma and 256-per-chroma 4x4 state capacity of a 128x128 4:2:0 superblock. The fixed 8x8 DC-intra traversal populates the owner's quantized coefficient and state slices while updating the caller-owned reconstruction plane directly, and a real tile-writer integration check proves that both sides consume identical luma and chroma areas. Complete mode decision still remains. - [~] Tile partition writing now follows current libaom's recursive `write_modes_sb` preorder traversal and `update_ext_partition_context` edge updates directly. Bottom-edge blocks use the horizontal-alike partition CDF and right-edge blocks use the vertical-alike CDF; byte-exact regressions cover both paths after the previous calls were found reversed. Lossless chroma-from-luma availability now uses the subsampled plane block size shared with the decoder instead of the lossy 32x32 limit, preserving the correct UV-mode alphabet for each segment. The obsolete SVT-derived global geometry catalog and its unimplemented lookup are removed; transform geometry is derived in libaom's bounded 64x64 residual order, fixed intra transform-size symbols use the reference depth and neighbor contexts, and each derived transform size is persisted to the frame-owned mode information before the entropy snapshot and coefficient traversal consume it. Frame-edge and segmentation syntax use mode-information units, and 128x128 CDEF units use libaom's 0-to-3 indexing and first-block strength ownership. The focused transform-state regression passes 3 of 3 direct net11 VSTest cases in Release. Writer, entropy, and OBU coverage passes 1,957 of 1,957 direct net11 VSTest cases in Release, with 20 of 20 focused encoder and decoder chroma-from-luma cases. Partition and mode analysis still need to populate these retained decisions; variable inter-transform syntax remains part of later inter-frame support. - [ ] Implement legal deblocking, CDEF, restoration, super-resolution, and film-grain signaling decisions. -- [~] The coefficient symbol encoder now reuses tile-lifetime level and context workspaces instead of allocating per transform, defers both coefficient rents until the first nonzero transform block, and disposes all tile scratch independently from the detached encoded bytes. Its range coder matches current libaom's 64-bit coding window, bulk big-endian byte flush, and backward carry propagation while using one byte of allocator scratch per estimated output byte instead of the former 16-bit pre-carry storage. The single-tile production path finalizes in that existing allocation and transfers its owner plus the used byte length, removing the former second rent and full-tile copy; exact-length test callers retain the original overload. The ownership regression proves that writer disposal cannot return transferred storage and that the caller returns the original allocation exactly once. The operation boundary still needs to connect this payload to the OBU and AVIF container writers before activation. +- [~] The coefficient symbol encoder now reuses tile-lifetime level and context workspaces instead of allocating per transform, defers both coefficient rents until the first nonzero transform block, and disposes all tile scratch independently from the detached encoded bytes. Its range coder matches current libaom's 64-bit coding window, bulk big-endian byte flush, and backward carry propagation while using one byte of allocator scratch per estimated output byte instead of the former 16-bit pre-carry storage. The single-tile production path finalizes in that existing allocation and transfers its owner plus the used byte length, removing the former second rent and full-tile copy; exact-length test callers retain the original overload. The ownership regression proves that writer disposal cannot return transferred storage and that the caller returns the original allocation exactly once. The internal frame operation now passes that payload directly to the OBU writer; AVIF container integration and public activation remain. - [~] The planar conversion, DC intra prediction, residual construction, forward transform, and forward quantizer use descending SIMD dispatch: Vector512, Vector256, Vector128, then scalar. Residual construction matches current libaom's exact source-minus-prediction arithmetic for 8-bit and high-bit-depth planes, preserves independent row strides and unaligned starts, and writes directly into caller-owned signed-short storage without allocation. The composed block path delegates arithmetic to those closed operators and adds no allocation. Apply the same rule to every later hot-path family. - [~] Residual tests verify misaligned planes, independent source, prediction, and destination strides, SIMD remainders, untouched padding, 8-bit, 10-bit, and 12-bit precision, every operator width independently of host acceleration, the scalar fallback, and zero per-transform allocations. - [~] The unused coefficient-shape transform facade and its unimplemented N2, N4, and DC-only branches are removed. Finalized block encoding now follows the complete-transform path that current libaom uses before fast quantization; later rate-distortion search may add proven coefficient optimization without exposing inactive runtime throws. - [~] Forward-quantizer FeatureTestRunner and zero-allocation tests compare every hardware tier with an independent scan-order scalar oracle shaped from current-main libaom. Both passed direct net11 VSTest in Release. - [~] The combined-frame writer now completes the byte-counted uncompressed frame header before starting the optional multi-tile tile-group flag, matching current libaom's separate frame-header and tile-group writers. A non-uniform two-tile round trip verifies the explicit boundaries, both tile payloads, and complete stream consumption through direct net11 VSTest in Release. +- [~] The first internal frame-to-OBU operation encodes 8-, 10-, and 12-bit monochrome reduced still pictures through the production tile writer and production decoder. Coefficient context initialization now stores `min(abs(level), 127)`, matching current libaom; the previous signed clamp converted every negative transform coefficient to zero and selected invalid nonzero-map distributions. Signed dense and sparse entropy round trips, direct level-buffer saturation coverage, and eight constant/gradient frame cases pass 52 of 52 direct net11 VSTest cases in Release. Current-main `aomdec` accepts all eight emitted payloads. Their decoded-frame MD5 values are `d09ea148582b9c93fa78e59426193bbc` (16x16 8-bit constant), `14e7d5ee5f70ad21972b88700338f32b` (16x16 8-bit gradient), `f949f7422913e83dff07ee5e0a5087d3` (8x8 8-bit constant), `ae7233a94558978934469dcc4da764dd` (8x8 8-bit gradient), `09223b227f3abc3134d0a3ea15f70c0a` (8x8 10-bit constant), `539aab0e6e14bcaec271febfa8e25444` (8x8 10-bit gradient), `73117a8fc102e5d028f82444fc4d15ab` (8x8 12-bit constant), and `0a7c7e058d8f56e6f3685c8eb8c3ece0` (8x8 12-bit gradient). This is an independently decodable baseline, not completion evidence for chroma, alpha, options, containers, or the public encoder. +- [x] The exact net11 Release rebuild completed at the established 1,005-warning repository baseline with zero errors. The complete HEIF/AV1 namespace passes 8,838 of 8,838 direct VSTest cases with zero failures or skips. Roslynk reports zero compiler errors and no diagnostics in the five changed C# files; `git diff --check` passes and `.gitattributes` is unchanged. ### 7. Write complete AVIF output diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs index 2f769c70a0..b816b05f66 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs @@ -3,6 +3,9 @@ using SixLabors.ImageSharp.Formats.Heif.Av1.Color; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; +using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; +using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; +using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; using SixLabors.ImageSharp.Formats.Heif.Components; using SixLabors.ImageSharp.PixelFormats; @@ -13,6 +16,118 @@ namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; /// internal static class Av1FrameEncoder { + /// + /// Encodes one reduced-still-picture AV1 frame into a low-overhead OBU stream. + /// + /// The packed source pixel type. + /// The configuration providing every operation-scoped allocation. + /// The packed source frame. + /// The destination receiving the complete AV1 item payload. + /// The resolved native color and precision configuration. + /// The frame quantizer index. + /// The sequence header describing the encoded payload. + public static ObuSequenceHeader Encode( + Configuration configuration, + ImageFrame image, + Stream stream, + ObuColorConfig colorConfig, + int qIndex) + where TPixel : unmanaged, IPixel + { + int width = image.Width; + int height = image.Height; + Av1ColorFormat colorFormat = colorConfig.GetColorFormat(); + ObuSequenceProfile sequenceProfile = colorConfig.BitDepth == Av1BitDepth.TwelveBit || + colorFormat == Av1ColorFormat.Yuv422 + ? ObuSequenceProfile.Professional + : colorFormat == Av1ColorFormat.Yuv444 + ? ObuSequenceProfile.High + : ObuSequenceProfile.Main; + + ObuSequenceHeader sequenceHeader = new() + { + IsStillPicture = true, + IsReducedStillPictureHeader = true, + SequenceProfile = sequenceProfile, + OperatingPoint = [new ObuOperatingPoint { SequenceLevelIndex = 31 }], + FrameWidthBits = width > 1 ? Av1Math.MostSignificantBit((uint)(width - 1)) + 1 : 1, + FrameHeightBits = height > 1 ? Av1Math.MostSignificantBit((uint)(height - 1)) + 1 : 1, + MaxFrameWidth = width, + MaxFrameHeight = height, + Use128x128Superblock = false, + ForceScreenContentTools = 2, + ForceIntegerMotionVector = 2, + EnableFilterIntra = false, + EnableIntraEdgeFilter = false, + EnableSuperResolution = false, + EnableCdef = false, + EnableRestoration = false, + ColorConfig = colorConfig + }; + + int modeInfoColumnCount = 2 * ((width + 7) >> 3); + int modeInfoRowCount = 2 * ((height + 7) >> 3); + ObuTileGroupHeader tiles = new() + { + HasUniformTileSpacing = true, + TileColumnCount = 1, + TileRowCount = 1, + TileSizeBytes = 4 + }; + + tiles.TileColumnStartModeInfo[1] = modeInfoColumnCount; + tiles.TileRowStartModeInfo[1] = modeInfoRowCount; + ObuFrameHeader frameHeader = new() + { + FrameType = ObuFrameType.KeyFrame, + ShowFrame = true, + ErrorResilientMode = true, + RefreshFrameFlags = byte.MaxValue, + DisableFrameEndUpdateCdf = true, + TransformMode = Av1TransformMode.Largest, + ModeInfoColumnCount = modeInfoColumnCount, + ModeInfoRowCount = modeInfoRowCount, + TilesInfo = tiles, + FrameSize = new ObuFrameSize + { + FrameWidth = width, + FrameHeight = height, + SuperResolutionDenominator = Av1Constants.ScaleNumerator, + SuperResolutionUpscaledWidth = width, + RenderWidth = width, + RenderHeight = height + } + }; + + frameHeader.QuantizationParameters.BaseQIndex = qIndex; + Av1QuantizationLookup.UpdateFrameQuantizationState(frameHeader); + + // Libaom reserves 2.5 times the 32-sample-aligned native input for an all-intra output packet. + // Counting the active planes directly retains that headroom without charging monochrome for unused chroma. + int alignedWidth = Av1Math.AlignPowerOf2(width, 5); + int alignedHeight = Av1Math.AlignPowerOf2(height, 5); + int subsamplingX = colorConfig.SubSamplingX ? 1 : 0; + int subsamplingY = colorConfig.SubSamplingY ? 1 : 0; + long sampleCount = (long)alignedWidth * alignedHeight; + if (!colorConfig.IsMonochrome) + { + sampleCount += 2L * (alignedWidth >> subsamplingX) * (alignedHeight >> subsamplingY); + } + + int sampleSize = colorConfig.BitDepth == Av1BitDepth.EightBit ? 1 : 2; + int initialTileSize = checked((int)Math.Max(8192L, (sampleCount * sampleSize * 5) / 2)); + if (colorConfig.BitDepth == Av1BitDepth.EightBit) + { + EncodeByte(configuration, image, stream, sequenceHeader, frameHeader, colorFormat, initialTileSize); + } + else + { + EncodeHighBitDepth(configuration, image, stream, sequenceHeader, frameHeader, colorFormat, initialTileSize); + } + + return sequenceHeader; + } + /// /// Converts packed pixels directly into an eight-bit bordered AV1 source frame. /// @@ -45,6 +160,151 @@ internal static class Av1FrameEncoder where TPixel : unmanaged, IPixel => PrepareSource(configuration, image, source, colorConfig); + private static void EncodeByte( + Configuration configuration, + ImageFrame image, + Stream stream, + ObuSequenceHeader sequenceHeader, + ObuFrameHeader frameHeader, + Av1ColorFormat colorFormat, + int initialTileSize) + where TPixel : unmanaged, IPixel + { + using Av1EncoderFrameBuffer source = new( + configuration, + image.Width, + image.Height, + 8, + colorFormat, + chromaPositionX: 1, + chromaPositionY: 1); + + using Av1EncoderFrameBuffer reconstruction = new( + configuration, + image.Width, + image.Height, + 8, + colorFormat, + chromaPositionX: 1, + chromaPositionY: 1); + + Encode(configuration, image, stream, sequenceHeader, frameHeader, source, reconstruction, initialTileSize); + } + + private static void EncodeHighBitDepth( + Configuration configuration, + ImageFrame image, + Stream stream, + ObuSequenceHeader sequenceHeader, + ObuFrameHeader frameHeader, + Av1ColorFormat colorFormat, + int initialTileSize) + where TPixel : unmanaged, IPixel + { + int bitDepth = sequenceHeader.ColorConfig.BitDepth.GetBitCount(); + using Av1EncoderFrameBuffer source = new( + configuration, + image.Width, + image.Height, + bitDepth, + colorFormat, + chromaPositionX: 1, + chromaPositionY: 1); + + using Av1EncoderFrameBuffer reconstruction = new( + configuration, + image.Width, + image.Height, + bitDepth, + colorFormat, + chromaPositionX: 1, + chromaPositionY: 1); + + Encode(configuration, image, stream, sequenceHeader, frameHeader, source, reconstruction, initialTileSize); + } + + private static void Encode( + Configuration configuration, + ImageFrame image, + Stream stream, + ObuSequenceHeader sequenceHeader, + ObuFrameHeader frameHeader, + Av1EncoderFrameBuffer source, + Av1EncoderFrameBuffer reconstruction, + int initialTileSize) + where TPixel : unmanaged, IPixel + { + PrepareSource(configuration, image, source.Frame, sequenceHeader.ColorConfig); + using Av1EncoderPictureBuffer picture = new( + configuration, + sequenceHeader, + frameHeader, + image.Width, + image.Height); + + using Av1EncoderCoefficientBuffer coefficients = new( + configuration, + sequenceHeader, + image.Width, + image.Height); + + using Av1EncoderSuperblockWorkspace superblockWorkspace = new(configuration); + using Av1EncoderBlockWorkspace blockWorkspace = new(configuration); + using Av1IntraTileWriter tileWriter = new( + configuration, + source.Frame, + reconstruction.Frame, + picture.Picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialTileSize); + + ObuWriter writer = new(); + writer.WriteAll(configuration, stream, sequenceHeader, frameHeader, tileWriter); + } + + private static void Encode( + Configuration configuration, + ImageFrame image, + Stream stream, + ObuSequenceHeader sequenceHeader, + ObuFrameHeader frameHeader, + Av1EncoderFrameBuffer source, + Av1EncoderFrameBuffer reconstruction, + int initialTileSize) + where TPixel : unmanaged, IPixel + { + PrepareSource(configuration, image, source.Frame, sequenceHeader.ColorConfig); + using Av1EncoderPictureBuffer picture = new( + configuration, + sequenceHeader, + frameHeader, + image.Width, + image.Height); + + using Av1EncoderCoefficientBuffer coefficients = new( + configuration, + sequenceHeader, + image.Width, + image.Height); + + using Av1EncoderSuperblockWorkspace superblockWorkspace = new(configuration); + using Av1EncoderBlockWorkspace blockWorkspace = new(configuration); + using Av1IntraTileWriter tileWriter = new( + configuration, + source.Frame, + reconstruction.Frame, + picture.Picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialTileSize); + + ObuWriter writer = new(); + writer.WriteAll(configuration, stream, sequenceHeader, frameHeader, tileWriter); + } + /// /// Converts packed pixels into native component planes and initializes every coded and physical edge sample. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1LevelBuffer.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1LevelBuffer.cs index cec5f849ca..f9abba8727 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1LevelBuffer.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1LevelBuffer.cs @@ -76,8 +76,9 @@ internal sealed class Av1LevelBuffer : IDisposable ref int sourceRef = ref coefficientBuffer[y * this.Size.Width]; for (int x = 0; x < this.Size.Width; x++) { - // Entropy contexts use a saturated byte-level summary rather than the full coefficient magnitude. - destRef = (byte)Av1Math.Clamp(sourceRef, 0, byte.MaxValue); + // Entropy contexts use the absolute level, saturated to the signed-byte range used by the + // normative nonzero-map context calculation. + destRef = (byte)Math.Min(Math.Abs(sourceRef), sbyte.MaxValue); destRef = ref Unsafe.Add(ref destRef, 1); sourceRef = ref Unsafe.Add(ref sourceRef, 1); } diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs index 188260fd8b..29d3dd2ec8 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs @@ -992,6 +992,12 @@ public class Av1CoefficientsEntropyTests Configuration configuration = Configuration.Default; using Av1SymbolEncoder encoder = new(configuration, 100 / 8, BaseQIndex); Span coefficientsBuffer = [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16]; + ReadOnlySpan scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan; + for (int scanIndex = endOfBlock; scanIndex < scan.Length; scanIndex++) + { + coefficientsBuffer[scan[scanIndex]] = 0; + } + Span actuals = new int[16 + 1]; // Act @@ -1023,6 +1029,8 @@ public class Av1CoefficientsEntropyTests levels, actuals); + decoder.ValidateTrailingBits(); + // Assert Assert.Equal(endOfBlock, actuals[0]); } @@ -1039,7 +1047,7 @@ public class Av1CoefficientsEntropyTests Av1TransformType transformType = (Av1TransformType)txType; Av1PredictionMode intraDirection = Av1PredictionMode.DC; Av1FilterIntraMode filterIntraMode = Av1FilterIntraMode.DC; - RoundTripCoefficientsCore(endOfBlock, componentType, blockSize, transformSize, transformType, intraDirection, filterIntraMode); + RoundTripCoefficientsCore(endOfBlock, componentType, blockSize, transformSize, transformType, intraDirection, filterIntraMode, true, false); } [Theory] @@ -1054,10 +1062,38 @@ public class Av1CoefficientsEntropyTests Av1TransformType transformType = (Av1TransformType)txType; Av1PredictionMode intraDirection = Av1PredictionMode.DC; Av1FilterIntraMode filterIntraMode = Av1FilterIntraMode.DC; - RoundTripCoefficientsCore(endOfBlock, componentType, blockSize, transformSize, transformType, intraDirection, filterIntraMode); + RoundTripCoefficientsCore(endOfBlock, componentType, blockSize, transformSize, transformType, intraDirection, filterIntraMode, true, false); + } + + [Theory] + [InlineData(1)] + [InlineData(2)] + [InlineData(3)] + [InlineData(17)] + [InlineData(33)] + [InlineData(63)] + [InlineData(64)] + public void RoundTripCoefficientsYSize8x8(ushort endOfBlock) + { + const Av1ComponentType componentType = Av1ComponentType.Luminance; + const Av1BlockSize blockSize = Av1BlockSize.Block8x8; + const Av1TransformSize transformSize = Av1TransformSize.Size8x8; + const Av1TransformType transformType = Av1TransformType.DctDct; + const Av1PredictionMode intraDirection = Av1PredictionMode.DC; + const Av1FilterIntraMode filterIntraMode = Av1FilterIntraMode.DC; + RoundTripCoefficientsCore(endOfBlock, componentType, blockSize, transformSize, transformType, intraDirection, filterIntraMode, false, true); } - private static void RoundTripCoefficientsCore(ushort endOfBlock, Av1ComponentType componentType, Av1BlockSize blockSize, Av1TransformSize transformSize, Av1TransformType transformType, Av1PredictionMode intraDirection, Av1FilterIntraMode filterIntraMode) + private static void RoundTripCoefficientsCore( + ushort endOfBlock, + Av1ComponentType componentType, + Av1BlockSize blockSize, + Av1TransformSize transformSize, + Av1TransformType transformType, + Av1PredictionMode intraDirection, + Av1FilterIntraMode filterIntraMode, + bool useReducedTransformSet, + bool useSparseCoefficients) { Av1BlockModeInfo modeInfo = new(blockSize, new Point(0, 0)); Av1TransformInfo transformInfo = new(transformSize, 0, 0); @@ -1066,11 +1102,35 @@ public class Av1CoefficientsEntropyTests Av1TransformBlockContext transformBlockContext = default; Configuration configuration = Configuration.Default; using Av1SymbolEncoder encoder = new(configuration, 100 / 8, BaseQIndex); - Span coefficientsBuffer = Enumerable.Range(0, blockSize.GetHeight() * blockSize.GetWidth()).ToArray(); - Span actuals = new int[16 + 1]; + int coefficientCount = blockSize.GetHeight() * blockSize.GetWidth(); + ReadOnlySpan scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan; + Span coefficientsBuffer = new int[coefficientCount]; + + for (int scanIndex = 0; scanIndex < endOfBlock; scanIndex++) + { + if (!useSparseCoefficients || scanIndex == endOfBlock - 1 || scanIndex % 4 == 0) + { + int level = scanIndex + 1; + + // Signed levels prove encoder context derivation uses magnitude; sparse cases also cover zero-map runs. + coefficientsBuffer[scan[scanIndex]] = (scanIndex & 1) == 0 ? -level : level; + } + } + + Span actuals = new int[coefficientCount + 1]; // Act - encoder.WriteCoefficients(transformSize, transformType, intraDirection, coefficientsBuffer, componentType, transformBlockContext, endOfBlock, true, filterIntraMode); + encoder.WriteCoefficients( + transformSize, + transformType, + intraDirection, + coefficientsBuffer, + componentType, + transformBlockContext, + endOfBlock, + useReducedTransformSet, + filterIntraMode); + using IMemoryOwner encoded = encoder.Exit(); Av1SymbolDecoder decoder = new(Configuration.Default, encoded.GetSpan(), BaseQIndex); @@ -1089,7 +1149,7 @@ public class Av1CoefficientsEntropyTests transformBlockContext, transformSize, false, - true, + useReducedTransformSet, transformType, ref transformInfo, 0, @@ -1097,9 +1157,10 @@ public class Av1CoefficientsEntropyTests levels, actuals); + decoder.ValidateTrailingBits(); + // Assert Assert.Equal(endOfBlock, actuals[0]); - ReadOnlySpan scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan; // The parser retains quantized levels in entropy scan order; inverse quantization maps them back to raster positions. for (int coefficientIndex = 0; coefficientIndex < endOfBlock; coefficientIndex++) diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs index 0915842cc3..a66fbba1dd 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs @@ -12,6 +12,85 @@ namespace SixLabors.ImageSharp.Tests.Formats.Heif.Av1; public class Av1EncoderFrameTests { + [Theory] + [InlineData(8, 8, false, 0)] + [InlineData(8, 8, true, 0)] + [InlineData(16, 16, false, 0)] + [InlineData(16, 16, true, 0)] + [InlineData(8, 8, false, 1)] + [InlineData(8, 8, true, 1)] + [InlineData(8, 8, false, 2)] + [InlineData(8, 8, true, 2)] + public void EncodeWritesReducedStillPictureConsumedByProductionDecoder(int width, int height, bool hasGradient, int bitDepthValue) + { + Av1BitDepth bitDepth = (Av1BitDepth)bitDepthValue; + using Image source = new(width, height); + for (int y = 0; y < height; y++) + { + Span row = source.Frames.RootFrame.PixelBuffer.DangerousGetRowSpan(y); + for (int x = 0; x < width; x++) + { + byte value = hasGradient ? (byte)((x * 13) + (y * 17)) : (byte)128; + row[x] = new Rgba32(value, value, value); + } + } + + ObuColorConfig colorConfig = CreateMonochromeColorConfig(bitDepth); + using MemoryStream stream = new(); + ObuSequenceHeader encodedHeader = Av1FrameEncoder.Encode( + Configuration.Default, + source.Frames.RootFrame, + stream, + colorConfig, + qIndex: 37); + + byte[] payload = stream.ToArray(); + string outputDirectory = Path.Combine( + TestEnvironment.ActualOutputDirectoryFullPath, + "Formats", + "Heif", + "Av1"); + + Directory.CreateDirectory(outputDirectory); + string contentName = hasGradient ? "gradient" : "constant"; + int bitCount = bitDepth.GetBitCount(); + string fileName = $"encoder-frame-{width}x{height}-{bitCount}b-400-{contentName}.obu"; + File.WriteAllBytes(Path.Combine(outputDirectory, fileName), payload); + + using Av1Decoder decoder = new(Configuration.Default); + using Image decoded = decoder.Decode(payload); + + Assert.Equal(width, decoded.Width); + Assert.Equal(height, decoded.Height); + ObuSequenceProfile expectedProfile = bitDepth == Av1BitDepth.TwelveBit + ? ObuSequenceProfile.Professional + : ObuSequenceProfile.Main; + + Assert.Equal(expectedProfile, encodedHeader.SequenceProfile); + Assert.True(encodedHeader.IsReducedStillPictureHeader); + + Rgba32 first = decoded[0, 0]; + Assert.Equal(first.R, first.G); + Assert.Equal(first.R, first.B); + Assert.Equal(byte.MaxValue, first.A); + if (hasGradient) + { + Assert.True(first.R < decoded[width - 1, height - 1].R); + } + else + { + Assert.InRange(first.R, 120, 136); + + for (int y = 0; y < height; y++) + { + foreach (Rgba32 pixel in decoded.Frames.RootFrame.PixelBuffer.DangerousGetRowSpan(y)) + { + Assert.Equal(first, pixel); + } + } + } + } + [Fact] public void PrepareSourceConvertsRgba32DirectlyIntoBorderedEightBitPlane() { diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1LevelBufferTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1LevelBufferTests.cs index 937d1bc67c..5b0556510b 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1LevelBufferTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1LevelBufferTests.cs @@ -60,6 +60,21 @@ public class Av1LevelBufferTests } } + [Fact] + public void InitializeStoresAbsoluteSaturatedLevels() + { + // Arrange + using Av1LevelBuffer levels = new(Configuration.Default, new Size(2, 2)); + Span coefficients = [-300, -1, 1, 300]; + + // Act + levels.Initialize(coefficients); + + // Assert + Assert.Equal([127, 1], levels.GetRow(0)[..2].ToArray()); + Assert.Equal([1, 127], levels.GetRow(1)[..2].ToArray()); + } + [Theory] [InlineData(4, 4)] [InlineData(8, 4)]