From ba360c800afc77b033c455be9133514dee5b1660 Mon Sep 17 00:00:00 2001 From: James Jackson-South Date: Thu, 3 Sep 2026 09:31:47 +1000 Subject: [PATCH] Add AV1 auxiliary alpha encoding --- HEIF_IMPLEMENTATION_PLAN.md | 1 + .../Heif/Av1/Pipeline/Av1FrameEncoder.cs | 119 +++++++++++---- .../Alpha/HeifPlanarAlphaEncoder.cs | 139 ++++++++++++++++++ .../Formats/Heif/Av1/Av1EncoderFrameTests.cs | 104 +++++++++++++ 4 files changed, 334 insertions(+), 29 deletions(-) create mode 100644 src/ImageSharp/Formats/Heif/Components/Alpha/HeifPlanarAlphaEncoder.cs diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 82eac8a281..fe12ce7ee4 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -820,6 +820,7 @@ Encoder verification contract: ### 6. Build the complete AV1 frame encoder - [~] SIMD-first RGB-to-native-plane conversion now feeds eight-bit and high-bit-depth bordered AV1 source frames directly, preserving ImageSharp's arbitrary packed-pixel input contract without an intermediate full-frame native-plane copy. +- [~] Auxiliary-alpha encoding now follows the same packed-pixel conversion boundary without scanning pixel contents or cloning the image. Source alpha is converted through ImageSharp's 16-bit pixel contract, deinterleaved with descending Vector512, Vector256, Vector128, and scalar traversal through the shared vector-count helpers, then scaled and rounded once by the existing native-sample writer directly into the final bordered monochrome source frame. One operation-wide allocator owner provides the packed and planar row views; there is no frame-sized alpha staging allocation or second owner. Exact 12-bit precision, physical border extension, the single 12-bytes-per-pixel row rent, and balanced return pass through the production converter. The complete 47-case frame-encoder set passes direct foreground net11 Release VSTest, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts the generated 8-, 10-, and 12-bit monochrome payloads. AVIF auxiliary item properties, references, and public activation remain open. - [~] Forward transform families, transform workspace, and an allocation-free DC intra block boundary exist locally. For eight-bit and high-bit-depth samples, the composed boundary now follows current libaom's encoder order: predict into the reconstruction plane, subtract prediction from source, transform, quantize into separate qcoeff and dqcoeff storage, retain EOB and transform type, and inverse-transform only when EOB is nonzero so later blocks consume decoder-identical references. Prediction and subtraction retain their SIMD-first operators, independent source and reconstruction strides are preserved, and no frame-sized or per-block buffer is introduced. The block boundary consumes the real bordered encoder-plane regions and indexes their one-segment owner directly; this preserves physical row strides without a row copy and avoids the per-call enumerator allocation exposed by the initial array-only test. One reusable 61 KiB allocator owner supplies tightly packed residual, aligned transform-coefficient, dequantized-coefficient, and transform scratch spans across transform blocks; quantized coefficients write directly to the retained frame coefficient owner instead of being duplicated. A fixed 8x8 DC-intra superblock baseline now traverses the same recursive preorder and frame-edge pruning as the tile writer, gathers left references into that reusable block workspace, writes luma and chroma coefficient-owner slices in the writer's exact consumption order, and updates the caller-owned reconstruction planes for subsequent predictions. Stage-by-stage scalar-oracle, physical-border, retained-syntax, superblock-to-writer synchronization, high-bit-depth precision, and steady-state zero-allocation coverage passes 8 of 8 through direct net11 VSTest in Release. This is a legal fixed baseline, not complete partition or mode analysis. - [~] A production single-tile all-intra writer now walks raster superblocks, analyzes each immediately before entropy coding, reuses one decision workspace and one block workspace, and retains decoder-identical reconstructed references across the tile. Its closed byte and high-bit-depth operators feed the existing superblock boundary without runtime sample-type checks. A byte-exact test compares this composed path with an explicit superblock-then-tile-writer oracle, so producer and writer traversal or coefficient-area drift cannot pass unnoticed. A separate clipped 2x2-superblock regression proves global raster indexing by requiring all four coefficient segments and the bottom-right reconstruction to be populated. Multi-tile ownership and the complete frame/OBU operation remain. - [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder operation now instantiates matching source and reconstruction owners with ordinary `using` lifetimes and converts packed pixels directly into the source owner before extension. diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs index 6854c7bb5e..e3b7a9a352 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs @@ -7,6 +7,7 @@ using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; using SixLabors.ImageSharp.Formats.Heif.Components; +using SixLabors.ImageSharp.Formats.Heif.Components.Alpha; using SixLabors.ImageSharp.PixelFormats; namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; @@ -33,6 +34,35 @@ internal static class Av1FrameEncoder ObuColorConfig colorConfig, int qIndex) where TPixel : unmanaged, IPixel + => Encode(configuration, image, stream, colorConfig, qIndex, false); + + /// + /// Encodes one packed alpha channel as a reduced-still-picture monochrome AV1 frame. + /// + /// The packed source pixel type. + /// The configuration providing every operation-scoped allocation. + /// The packed source frame. + /// The destination receiving the complete AV1 item payload. + /// The resolved monochrome precision configuration. + /// The frame quantizer index. + /// The sequence header describing the encoded payload. + public static ObuSequenceHeader EncodeAlpha( + Configuration configuration, + ImageFrame image, + Stream stream, + ObuColorConfig colorConfig, + int qIndex) + where TPixel : unmanaged, IPixel + => Encode(configuration, image, stream, colorConfig, qIndex, true); + + private static ObuSequenceHeader Encode( + Configuration configuration, + ImageFrame image, + Stream stream, + ObuColorConfig colorConfig, + int qIndex, + bool encodeAlpha) + where TPixel : unmanaged, IPixel { int width = image.Width; int height = image.Height; @@ -118,11 +148,11 @@ internal static class Av1FrameEncoder int initialTileSize = checked((int)Math.Max(8192L, (sampleCount * sampleSize * 5) / 2)); if (colorConfig.BitDepth == Av1BitDepth.EightBit) { - EncodeByte(configuration, image, stream, sequenceHeader, frameHeader, colorFormat, initialTileSize); + EncodeByte(configuration, image, stream, sequenceHeader, frameHeader, colorFormat, initialTileSize, encodeAlpha); } else { - EncodeHighBitDepth(configuration, image, stream, sequenceHeader, frameHeader, colorFormat, initialTileSize); + EncodeHighBitDepth(configuration, image, stream, sequenceHeader, frameHeader, colorFormat, initialTileSize, encodeAlpha); } return sequenceHeader; @@ -142,7 +172,7 @@ internal static class Av1FrameEncoder Av1EncoderFrame source, ObuColorConfig colorConfig) where TPixel : unmanaged, IPixel - => PrepareSource(configuration, image, source, colorConfig); + => PrepareSource(configuration, image, source, colorConfig, false); /// /// Converts packed pixels directly into a high-bit-depth bordered AV1 source frame. @@ -158,7 +188,7 @@ internal static class Av1FrameEncoder Av1EncoderFrame source, ObuColorConfig colorConfig) where TPixel : unmanaged, IPixel - => PrepareSource(configuration, image, source, colorConfig); + => PrepareSource(configuration, image, source, colorConfig, false); private static void EncodeByte( Configuration configuration, @@ -167,7 +197,8 @@ internal static class Av1FrameEncoder ObuSequenceHeader sequenceHeader, ObuFrameHeader frameHeader, Av1ColorFormat colorFormat, - int initialTileSize) + int initialTileSize, + bool encodeAlpha) where TPixel : unmanaged, IPixel { using Av1EncoderFrameBuffer source = new( @@ -188,7 +219,7 @@ internal static class Av1FrameEncoder chromaPositionX: 1, chromaPositionY: 1); - Encode(configuration, image, stream, sequenceHeader, frameHeader, source, reconstruction, initialTileSize); + Encode(configuration, image, stream, sequenceHeader, frameHeader, source, reconstruction, initialTileSize, encodeAlpha); } private static void EncodeHighBitDepth( @@ -198,7 +229,8 @@ internal static class Av1FrameEncoder ObuSequenceHeader sequenceHeader, ObuFrameHeader frameHeader, Av1ColorFormat colorFormat, - int initialTileSize) + int initialTileSize, + bool encodeAlpha) where TPixel : unmanaged, IPixel { int bitDepth = sequenceHeader.ColorConfig.BitDepth.GetBitCount(); @@ -220,7 +252,7 @@ internal static class Av1FrameEncoder chromaPositionX: 1, chromaPositionY: 1); - Encode(configuration, image, stream, sequenceHeader, frameHeader, source, reconstruction, initialTileSize); + Encode(configuration, image, stream, sequenceHeader, frameHeader, source, reconstruction, initialTileSize, encodeAlpha); } private static void Encode( @@ -231,10 +263,17 @@ internal static class Av1FrameEncoder ObuFrameHeader frameHeader, Av1EncoderFrameBuffer source, Av1EncoderFrameBuffer reconstruction, - int initialTileSize) + int initialTileSize, + bool encodeAlpha) where TPixel : unmanaged, IPixel { - PrepareSource(configuration, image, source.Frame, sequenceHeader.ColorConfig); + PrepareSource( + configuration, + image, + source.Frame, + sequenceHeader.ColorConfig, + encodeAlpha); + Av1ScreenContentDetector.Detect( source.Frame, out bool allowScreenContentTools, @@ -279,10 +318,17 @@ internal static class Av1FrameEncoder ObuFrameHeader frameHeader, Av1EncoderFrameBuffer source, Av1EncoderFrameBuffer reconstruction, - int initialTileSize) + int initialTileSize, + bool encodeAlpha) where TPixel : unmanaged, IPixel { - PrepareSource(configuration, image, source.Frame, sequenceHeader.ColorConfig); + PrepareSource( + configuration, + image, + source.Frame, + sequenceHeader.ColorConfig, + encodeAlpha); + Av1ScreenContentDetector.Detect( source.Frame, out bool allowScreenContentTools, @@ -326,27 +372,42 @@ internal static class Av1FrameEncoder Configuration configuration, ImageFrame image, Av1EncoderFrame source, - ObuColorConfig colorConfig) + ObuColorConfig colorConfig, + bool encodeAlpha) where TPixel : unmanaged, IPixel where TSample : unmanaged where TStorer : struct, IHeifSampleConverter { - HeifColorConversionParameters parameters = Av1YuvConverter.GetConversionParameters( - colorConfig, - out HeifColorConversionMode mode); - - // Conversion writes into the final bordered analysis planes. The later coding stages therefore consume the - // native source directly without a second full-frame copy from an intermediate component buffer. - HeifPlanarColorConverter.ConvertFromRgb< - TPixel, - Av1EncoderFrame.PlanarView, - TSample, - TStorer>( - configuration, - image, - source.View, - in parameters, - mode); + if (encodeAlpha) + { + HeifPlanarAlphaEncoder.Convert< + TPixel, + Av1EncoderFrame.PlanarView, + TSample, + TStorer>( + configuration, + image, + source.View); + } + else + { + HeifColorConversionParameters parameters = Av1YuvConverter.GetConversionParameters( + colorConfig, + out HeifColorConversionMode mode); + + // Conversion writes into the final bordered analysis planes. Later coding stages consume the native + // source without a second full-frame copy from an intermediate component buffer. + HeifPlanarColorConverter.ConvertFromRgb< + TPixel, + Av1EncoderFrame.PlanarView, + TSample, + TStorer>( + configuration, + image, + source.View, + in parameters, + mode); + } source.ExtendBorders(); } diff --git a/src/ImageSharp/Formats/Heif/Components/Alpha/HeifPlanarAlphaEncoder.cs b/src/ImageSharp/Formats/Heif/Components/Alpha/HeifPlanarAlphaEncoder.cs new file mode 100644 index 0000000000..dcb54e1ad1 --- /dev/null +++ b/src/ImageSharp/Formats/Heif/Components/Alpha/HeifPlanarAlphaEncoder.cs @@ -0,0 +1,139 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using System.Buffers; +using System.Runtime.CompilerServices; +using System.Runtime.InteropServices; +using System.Runtime.Intrinsics; +using SixLabors.ImageSharp.Memory; +using SixLabors.ImageSharp.PixelFormats; + +namespace SixLabors.ImageSharp.Formats.Heif.Components.Alpha; + +/// +/// Converts packed ImageSharp alpha values into one native HEIF monochrome plane. +/// +internal static class HeifPlanarAlphaEncoder +{ + /// + /// Converts one packed image frame into a full-range native alpha plane. + /// + /// The packed source pixel type. + /// The codec adapter exposing the destination plane. + /// The native unsigned sample storage type. + /// The SIMD narrowing and storage operations for the sample type. + /// The configuration used for row allocation and pixel conversion. + /// The packed source image frame. + /// The monochrome destination buffer. + public static void Convert( + Configuration configuration, + ImageFrame image, + TBuffer buffer) + where TPixel : unmanaged, IPixel + where TBuffer : struct, IHeifPlanarSampleBuffer + where TSample : unmanaged + where TStorer : struct, IHeifSampleConverter + { + int width = image.Width; + + // Rgba64 preserves the source pixel's normalized alpha precision before quantization to the requested AV1 + // depth. Both row views share one owner because their lifetimes never escape this conversion operation. + using IMemoryOwner rowOwner = configuration.MemoryAllocator.Allocate(width * 3); + Span rowStorage = rowOwner.GetSpan(); + Span packed = MemoryMarshal.Cast(rowStorage[..(width * 2)]); + Span alpha = rowStorage.Slice(width * 2, width); + float maximum = (1 << buffer.LumaBitDepth) - 1; + float scale = maximum / ushort.MaxValue; + for (int y = 0; y < image.Height; y++) + { + ReadOnlySpan source = image.PixelBuffer.DangerousGetRowSpan(y); + PixelOperations.Instance.ToRgba64(configuration, source, packed); + ExtractAlpha(packed, alpha); + HeifSampleConversion.WriteSamples( + alpha, + buffer.GetLumaRowSpan(y), + scale, + 0F, + maximum); + } + } + + /// + /// Deinterleaves alpha values from one packed high-precision row. + /// + private static void ExtractAlpha(ReadOnlySpan source, Span destination) + { + ref Rgba64 sourceBase = ref MemoryMarshal.GetReference(source); + ref float destinationBase = ref MemoryMarshal.GetReference(destination); + int i = 0; + + // Packed RGBA requires a gather before conversion. Constructing vectors from the alpha fields keeps the + // widening and stores SIMD-wide without copying or transposing the complete packed row. + if (Vector512.IsHardwareAccelerated) + { + nuint vectorCount = Numerics.Vector512Count(destination.Length); + for (nuint vectorIndex = 0; vectorIndex < vectorCount; vectorIndex++) + { + int offset = (int)(vectorIndex * (uint)Vector512.Count); + Vector512 values = Vector512.Create( + CreateAlphaVector256(ref Unsafe.Add(ref sourceBase, offset)), + CreateAlphaVector256(ref Unsafe.Add(ref sourceBase, offset + Vector256.Count))); + + Vector512.ConvertToSingle(values).StoreUnsafe( + ref destinationBase, + (nuint)offset); + } + + i = (int)(vectorCount * (uint)Vector512.Count); + } + + if (Vector256.IsHardwareAccelerated) + { + nuint vectorCount = Numerics.Vector256Count(destination.Length - i); + for (nuint vectorIndex = 0; vectorIndex < vectorCount; vectorIndex++) + { + int offset = i + (int)(vectorIndex * (uint)Vector256.Count); + Vector256 values = CreateAlphaVector256(ref Unsafe.Add(ref sourceBase, offset)); + Vector256.ConvertToSingle(values).StoreUnsafe( + ref destinationBase, + (nuint)offset); + } + + i += (int)(vectorCount * (uint)Vector256.Count); + } + + if (Vector128.IsHardwareAccelerated) + { + nuint vectorCount = Numerics.Vector128Count(destination.Length - i); + for (nuint vectorIndex = 0; vectorIndex < vectorCount; vectorIndex++) + { + int offset = i + (int)(vectorIndex * (uint)Vector128.Count); + Vector128 values = CreateAlphaVector128(ref Unsafe.Add(ref sourceBase, offset)); + Vector128.ConvertToSingle(values).StoreUnsafe( + ref destinationBase, + (nuint)offset); + } + + i += (int)(vectorCount * (uint)Vector128.Count); + } + + for (; i < destination.Length; i++) + { + Unsafe.Add(ref destinationBase, i) = Unsafe.Add(ref sourceBase, i).A; + } + } + + [MethodImpl(MethodImplOptions.AggressiveInlining)] + private static Vector128 CreateAlphaVector128(ref Rgba64 source) + => Vector128.Create( + (uint)source.A, + Unsafe.Add(ref source, 1).A, + Unsafe.Add(ref source, 2).A, + Unsafe.Add(ref source, 3).A); + + [MethodImpl(MethodImplOptions.AggressiveInlining)] + private static Vector256 CreateAlphaVector256(ref Rgba64 source) + => Vector256.Create( + CreateAlphaVector128(ref source), + CreateAlphaVector128(ref Unsafe.Add(ref source, Vector128.Count))); +} diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs index 7c7dd07483..15d29eb4bd 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs @@ -5,6 +5,8 @@ using SixLabors.ImageSharp.Formats.Heif.Av1; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; +using SixLabors.ImageSharp.Formats.Heif.Components; +using SixLabors.ImageSharp.Formats.Heif.Components.Alpha; using SixLabors.ImageSharp.Memory; using SixLabors.ImageSharp.PixelFormats; using SixLabors.ImageSharp.Tests.Memory; @@ -147,6 +149,108 @@ public class Av1EncoderFrameTests } } + [Theory] + [InlineData(EightBit)] + [InlineData(TenBit)] + [InlineData(TwelveBit)] + public void EncodeAlphaWritesMonochromeReducedStillPicture(int bitDepthValue) + { + const int Width = 16; + const int Height = 16; + Av1BitDepth bitDepth = (Av1BitDepth)bitDepthValue; + using Image source = new(Width, Height); + for (int y = 0; y < Height; y++) + { + Span row = source.Frames.RootFrame.PixelBuffer.DangerousGetRowSpan(y); + for (int x = 0; x < Width; x++) + { + ushort alpha = (ushort)(((x + y) * ushort.MaxValue) / (Width + Height - 2)); + row[x] = new Rgba64(ushort.MaxValue, 0, 0, alpha); + } + } + + using MemoryStream stream = new(); + ObuSequenceHeader encodedHeader = Av1FrameEncoder.EncodeAlpha( + Configuration.Default, + source.Frames.RootFrame, + stream, + CreateColorConfig(bitDepth), + qIndex: 37); + + byte[] payload = stream.ToArray(); + using Av1Decoder decoder = new(Configuration.Default); + using Image decoded = decoder.Decode(payload); + + Assert.True(encodedHeader.ColorConfig.IsMonochrome); + Assert.Equal( + bitDepth == Av1BitDepth.TwelveBit ? ObuSequenceProfile.Professional : ObuSequenceProfile.Main, + encodedHeader.SequenceProfile); + + Assert.Equal(new Size(Width, Height), decoded.Size); + Assert.True(decoded[0, 0].R < decoded[Width - 1, Height - 1].R); + Assert.Equal(decoded[0, 0].R, decoded[0, 0].G); + Assert.Equal(decoded[0, 0].R, decoded[0, 0].B); + Assert.Equal(ushort.MaxValue, decoded[0, 0].A); + + string outputDirectory = Path.Combine( + TestEnvironment.ActualOutputDirectoryFullPath, + "Formats", + "Heif", + "Av1"); + + Directory.CreateDirectory(outputDirectory); + File.WriteAllBytes( + Path.Combine(outputDirectory, $"encoder-alpha-{Width}x{Height}-{bitDepth.GetBitCount()}b.obu"), + payload); + } + + [Fact] + public void AlphaConversionUsesOnePooledRowAndPreservesTwelveBitPrecision() + { + const int Width = 19; + const int Border = Av1EncoderFrame.LumaBorder; + using Image image = new(Width, 1); + ushort[] expected = new ushort[Width]; + for (int x = 0; x < Width; x++) + { + ushort alpha = (ushort)((x * (long)ushort.MaxValue) / (Width - 1)); + image[x, 0] = new Rgba64(0, 0, 0, alpha); + expected[x] = (ushort)(((alpha * 4095L) + (ushort.MaxValue / 2)) / ushort.MaxValue); + } + + using Av1EncoderFrameBuffer frameBuffer = new( + Configuration.Default, + Width, + 1, + 12, + Av1ColorFormat.Yuv400, + 0, + 0); + + TestMemoryAllocator allocator = new(); + allocator.EnableNonThreadSafeLogging(); + Configuration configuration = Configuration.Default.Clone(); + configuration.MemoryAllocator = allocator; + + HeifPlanarAlphaEncoder.Convert< + Rgba64, + Av1EncoderFrame.PlanarView, + ushort, + HeifUShortSampleConverter>( + configuration, + image.Frames.RootFrame, + frameBuffer.Frame.View); + + frameBuffer.Frame.ExtendBorders(); + + AssertReplicatedSingleRow(frameBuffer.Luma, Border, expected); + TestMemoryAllocator.AllocationRequest allocation = Assert.Single(allocator.AllocationLog); + Assert.Equal(typeof(float), allocation.ElementType); + Assert.Equal(Width * 3, allocation.Length); + TestMemoryAllocator.ReturnRequest returned = Assert.Single(allocator.ReturnLog); + Assert.Equal(allocation.AllocationId, returned.AllocationId); + } + [Fact] public void ScreenContentDetectorMatchesLibaomFeatureThresholds() {