Browse Source

Compose AV1 intra tile encoding

pull/2633/head
James Jackson-South 1 month ago
parent
commit
7cfafc5234
  1. 6
      HEIF_IMPLEMENTATION_PLAN.md
  2. 13
      src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs
  3. 29
      src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolWriter.cs
  4. 81
      src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.Operator.cs
  5. 176
      src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.cs
  6. 200
      src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs
  7. 42
      src/ImageSharp/Formats/Heif/Av1/Tiling/Av1NeighborArrayUnit.cs
  8. 4
      src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureControlSet.cs
  9. 2
      src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs
  10. 19
      src/ImageSharp/Memory/AutoExpandingMemory.cs
  11. 6
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs
  12. 76
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderModeInfoBufferTests.cs
  13. 37
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs
  14. 183
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs

6
HEIF_IMPLEMENTATION_PLAN.md

@ -820,7 +820,7 @@ Encoder verification contract:
- [~] SIMD-first RGB-to-native-plane conversion now feeds eight-bit and high-bit-depth bordered AV1 source frames directly, preserving ImageSharp's arbitrary packed-pixel input contract without an intermediate full-frame native-plane copy. - [~] SIMD-first RGB-to-native-plane conversion now feeds eight-bit and high-bit-depth bordered AV1 source frames directly, preserving ImageSharp's arbitrary packed-pixel input contract without an intermediate full-frame native-plane copy.
- [~] Forward transform families, transform workspace, and an allocation-free DC intra block boundary exist locally. For eight-bit and high-bit-depth samples, the composed boundary now follows current libaom's encoder order: predict into the reconstruction plane, subtract prediction from source, transform, quantize into separate qcoeff and dqcoeff storage, retain EOB and transform type, and inverse-transform only when EOB is nonzero so later blocks consume decoder-identical references. Prediction and subtraction retain their SIMD-first operators, independent source and reconstruction strides are preserved, and no frame-sized or per-block buffer is introduced. The block boundary consumes the real bordered encoder-plane regions and indexes their one-segment owner directly; this preserves physical row strides without a row copy and avoids the per-call enumerator allocation exposed by the initial array-only test. One reusable 61 KiB allocator owner supplies tightly packed residual, aligned transform-coefficient, dequantized-coefficient, and transform scratch spans across transform blocks; quantized coefficients write directly to the retained frame coefficient owner instead of being duplicated. A fixed 8x8 DC-intra superblock baseline now traverses the same recursive preorder and frame-edge pruning as the tile writer, gathers left references into that reusable block workspace, writes luma and chroma coefficient-owner slices in the writer's exact consumption order, and updates the caller-owned reconstruction planes for subsequent predictions. Stage-by-stage scalar-oracle, physical-border, retained-syntax, superblock-to-writer synchronization, high-bit-depth precision, and steady-state zero-allocation coverage passes 8 of 8 through direct net11 VSTest in Release. This is a legal fixed baseline, not complete partition or mode analysis. - [~] Forward transform families, transform workspace, and an allocation-free DC intra block boundary exist locally. For eight-bit and high-bit-depth samples, the composed boundary now follows current libaom's encoder order: predict into the reconstruction plane, subtract prediction from source, transform, quantize into separate qcoeff and dqcoeff storage, retain EOB and transform type, and inverse-transform only when EOB is nonzero so later blocks consume decoder-identical references. Prediction and subtraction retain their SIMD-first operators, independent source and reconstruction strides are preserved, and no frame-sized or per-block buffer is introduced. The block boundary consumes the real bordered encoder-plane regions and indexes their one-segment owner directly; this preserves physical row strides without a row copy and avoids the per-call enumerator allocation exposed by the initial array-only test. One reusable 61 KiB allocator owner supplies tightly packed residual, aligned transform-coefficient, dequantized-coefficient, and transform scratch spans across transform blocks; quantized coefficients write directly to the retained frame coefficient owner instead of being duplicated. A fixed 8x8 DC-intra superblock baseline now traverses the same recursive preorder and frame-edge pruning as the tile writer, gathers left references into that reusable block workspace, writes luma and chroma coefficient-owner slices in the writer's exact consumption order, and updates the caller-owned reconstruction planes for subsequent predictions. Stage-by-stage scalar-oracle, physical-border, retained-syntax, superblock-to-writer synchronization, high-bit-depth precision, and steady-state zero-allocation coverage passes 8 of 8 through direct net11 VSTest in Release. This is a legal fixed baseline, not complete partition or mode analysis.
- [~] Symbol writer, coefficient writer, and tile writer fragments exist locally. - [~] A production single-tile all-intra writer now walks raster superblocks, analyzes each immediately before entropy coding, reuses one decision workspace and one block workspace, and retains decoder-identical reconstructed references across the tile. Its closed byte and high-bit-depth operators feed the existing superblock boundary without runtime sample-type checks. A byte-exact test compares this composed path with an explicit superblock-then-tile-writer oracle, so producer and writer traversal or coefficient-area drift cannot pass unnoticed. A separate clipped 2x2-superblock regression proves global raster indexing by requiring all four coefficient segments and the bottom-right reconstruction to be populated. Multi-tile ownership and the complete frame/OBU operation remain.
- [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder boundary converts packed pixels directly into the source owner before extension; the containing encode operation still needs to instantiate matching source and reconstruction owners with ordinary `using` lifetimes. - [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder boundary converts packed pixels directly into the source owner before extension; the containing encode operation still needs to instantiate matching source and reconstruction owners with ordinary `using` lifetimes.
- [~] Temporal delimiter, sequence header, frame header, and combined-frame tile-group writing exist locally. The remaining required metadata, padding, and encoder-wide syntax paths are not complete. - [~] Temporal delimiter, sequence header, frame header, and combined-frame tile-group writing exist locally. The remaining required metadata, padding, and encoder-wide syntax paths are not complete.
- [~] Implement superblock and partition analysis for every permitted block size and partition. The current baseline deliberately splits every in-frame node to 8x8 blocks and records decisions in current-libaom writer preorder; block-size selection and non-split partition analysis remain. - [~] Implement superblock and partition analysis for every permitted block size and partition. The current baseline deliberately splits every in-frame node to 8x8 blocks and records decisions in current-libaom writer preorder; block-size selection and non-split partition analysis remain.
@ -828,13 +828,13 @@ Encoder verification contract:
- [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. - [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools.
- [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain. - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain.
- [ ] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. - [ ] Implement real rate-distortion selection and make quality and effort change work, size, and output quality.
- [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state now retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One clean allocation contains the two active edges, and the exact requested length plus exactly-once return pass with the complete 77-case coefficient and entropy class in direct net11 VSTest Release. Complete tile traversal, initialized picture state, and verified CDF update behavior remain. - [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One picture owner now packs segmentation plus every tile's partition, luma, chroma, and transform edges into one clean byte allocation with typed non-owning views; together with the separately typed packed mode-information owner, the complete picture state uses two allocator rents rather than seven. Exact aligned lengths, clean initialization, and balanced exactly-once returns are covered in Release. Multi-tile payload ownership and verified CDF update behavior remain.
- [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains. - [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains.
- [~] The final-block decision workspace uses one reusable 10.3 KiB ImageSharp allocator owner. It contains 1,024 explicitly packed 10-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree, replacing separate managed arrays. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled palette, quantizer, prediction, and partition bytes cannot leak into the next decision pass. Exact allocation, size, initialization, reset, return, repeated-run, writer, entropy, and OBU coverage pass 1,957 of 1,957 direct net11 VSTest cases in Release; complete mode decision still remains. - [~] The final-block decision workspace uses one reusable 10.3 KiB ImageSharp allocator owner. It contains 1,024 explicitly packed 10-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree, replacing separate managed arrays. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled palette, quantizer, prediction, and partition bytes cannot leak into the next decision pass. Exact allocation, size, initialization, reset, return, repeated-run, writer, entropy, and OBU coverage pass 1,957 of 1,957 direct net11 VSTest cases in Release; complete mode decision still remains.
- [~] Finalized transform coefficients and packed EOB/type state now use raster-ordered, per-superblock plane segments matching current libaom's coefficient-pool geometry. One ImageSharp allocator owner replaces libaom's separate coefficient, EOB, and entropy-context allocations while preserving the full 1024 luma and 256-per-chroma 4x4 state capacity of a 128x128 4:2:0 superblock. The fixed 8x8 DC-intra traversal populates the owner's quantized coefficient and state slices while updating the caller-owned reconstruction plane directly, and a real tile-writer integration check proves that both sides consume identical luma and chroma areas. Complete mode decision still remains. - [~] Finalized transform coefficients and packed EOB/type state now use raster-ordered, per-superblock plane segments matching current libaom's coefficient-pool geometry. One ImageSharp allocator owner replaces libaom's separate coefficient, EOB, and entropy-context allocations while preserving the full 1024 luma and 256-per-chroma 4x4 state capacity of a 128x128 4:2:0 superblock. The fixed 8x8 DC-intra traversal populates the owner's quantized coefficient and state slices while updating the caller-owned reconstruction plane directly, and a real tile-writer integration check proves that both sides consume identical luma and chroma areas. Complete mode decision still remains.
- [~] Tile partition writing now follows current libaom's recursive `write_modes_sb` preorder traversal and `update_ext_partition_context` edge updates directly. Bottom-edge blocks use the horizontal-alike partition CDF and right-edge blocks use the vertical-alike CDF; byte-exact regressions cover both paths after the previous calls were found reversed. Lossless chroma-from-luma availability now uses the subsampled plane block size shared with the decoder instead of the lossy 32x32 limit, preserving the correct UV-mode alphabet for each segment. The obsolete SVT-derived global geometry catalog and its unimplemented lookup are removed; transform geometry is derived in libaom's bounded 64x64 residual order, fixed intra transform-size symbols use the reference depth and neighbor contexts, and each derived transform size is persisted to the frame-owned mode information before the entropy snapshot and coefficient traversal consume it. Frame-edge and segmentation syntax use mode-information units, and 128x128 CDEF units use libaom's 0-to-3 indexing and first-block strength ownership. The focused transform-state regression passes 3 of 3 direct net11 VSTest cases in Release. Writer, entropy, and OBU coverage passes 1,957 of 1,957 direct net11 VSTest cases in Release, with 20 of 20 focused encoder and decoder chroma-from-luma cases. Partition and mode analysis still need to populate these retained decisions; variable inter-transform syntax remains part of later inter-frame support. - [~] Tile partition writing now follows current libaom's recursive `write_modes_sb` preorder traversal and `update_ext_partition_context` edge updates directly. Bottom-edge blocks use the horizontal-alike partition CDF and right-edge blocks use the vertical-alike CDF; byte-exact regressions cover both paths after the previous calls were found reversed. Lossless chroma-from-luma availability now uses the subsampled plane block size shared with the decoder instead of the lossy 32x32 limit, preserving the correct UV-mode alphabet for each segment. The obsolete SVT-derived global geometry catalog and its unimplemented lookup are removed; transform geometry is derived in libaom's bounded 64x64 residual order, fixed intra transform-size symbols use the reference depth and neighbor contexts, and each derived transform size is persisted to the frame-owned mode information before the entropy snapshot and coefficient traversal consume it. Frame-edge and segmentation syntax use mode-information units, and 128x128 CDEF units use libaom's 0-to-3 indexing and first-block strength ownership. The focused transform-state regression passes 3 of 3 direct net11 VSTest cases in Release. Writer, entropy, and OBU coverage passes 1,957 of 1,957 direct net11 VSTest cases in Release, with 20 of 20 focused encoder and decoder chroma-from-luma cases. Partition and mode analysis still need to populate these retained decisions; variable inter-transform syntax remains part of later inter-frame support.
- [ ] Implement legal deblocking, CDEF, restoration, super-resolution, and film-grain signaling decisions. - [ ] Implement legal deblocking, CDEF, restoration, super-resolution, and film-grain signaling decisions.
- [~] The coefficient symbol encoder now reuses tile-lifetime level and context workspaces instead of allocating per transform, defers both coefficient rents until the first nonzero transform block, and disposes all tile scratch independently from the detached encoded bytes. Its range coder matches current libaom's 64-bit coding window, bulk big-endian byte flush, and backward carry propagation while using one byte of allocator scratch per estimated output byte instead of the former 16-bit pre-carry storage. The reference-type symbol encoder is passed normally through tile traversal, and the operation boundary owns the allocator-backed item payload stream for exactly one synchronous encode. Every remaining encoder fragment must be audited before it becomes active. - [~] The coefficient symbol encoder now reuses tile-lifetime level and context workspaces instead of allocating per transform, defers both coefficient rents until the first nonzero transform block, and disposes all tile scratch independently from the detached encoded bytes. Its range coder matches current libaom's 64-bit coding window, bulk big-endian byte flush, and backward carry propagation while using one byte of allocator scratch per estimated output byte instead of the former 16-bit pre-carry storage. The single-tile production path finalizes in that existing allocation and transfers its owner plus the used byte length, removing the former second rent and full-tile copy; exact-length test callers retain the original overload. The ownership regression proves that writer disposal cannot return transferred storage and that the caller returns the original allocation exactly once. The operation boundary still needs to connect this payload to the OBU and AVIF container writers before activation.
- [~] The planar conversion, DC intra prediction, residual construction, forward transform, and forward quantizer use descending SIMD dispatch: Vector512, Vector256, Vector128, then scalar. Residual construction matches current libaom's exact source-minus-prediction arithmetic for 8-bit and high-bit-depth planes, preserves independent row strides and unaligned starts, and writes directly into caller-owned signed-short storage without allocation. The composed block path delegates arithmetic to those closed operators and adds no allocation. Apply the same rule to every later hot-path family. - [~] The planar conversion, DC intra prediction, residual construction, forward transform, and forward quantizer use descending SIMD dispatch: Vector512, Vector256, Vector128, then scalar. Residual construction matches current libaom's exact source-minus-prediction arithmetic for 8-bit and high-bit-depth planes, preserves independent row strides and unaligned starts, and writes directly into caller-owned signed-short storage without allocation. The composed block path delegates arithmetic to those closed operators and adds no allocation. Apply the same rule to every later hot-path family.
- [~] Residual tests verify misaligned planes, independent source, prediction, and destination strides, SIMD remainders, untouched padding, 8-bit, 10-bit, and 12-bit precision, every operator width independently of host acceleration, the scalar fallback, and zero per-transform allocations. - [~] Residual tests verify misaligned planes, independent source, prediction, and destination strides, SIMD remainders, untouched padding, 8-bit, 10-bit, and 12-bit precision, every operator width independently of host acceleration, the scalar fallback, and zero per-transform allocations.
- [~] The unused coefficient-shape transform facade and its unimplemented N2, N4, and DC-only branches are removed. Finalized block encoding now follows the complete-transform path that current libaom uses before fast quantization; later rate-distortion search may add proven coefficient optimization without exposing inactive runtime throws. - [~] The unused coefficient-shape transform facade and its unimplemented N2, N4, and DC-only branches are removed. Finalized block encoding now follows the complete-transform path that current libaom uses before fast quantization; later rate-distortion search may add proven coefficient optimization without exposing inactive runtime throws.

13
src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs

@ -479,7 +479,18 @@ internal class Av1SymbolEncoder : IDisposable
} }
/// <summary> /// <summary>
/// Releases output memory that has not been transferred by <see cref="Exit"/>. /// Finalizes the range-coded tile payload and transfers its current allocation without copying.
/// </summary>
/// <param name="length">The number of encoded bytes at the beginning of the returned allocation.</param>
/// <returns>The complete allocation containing the encoded tile prefix.</returns>
public IMemoryOwner<byte> Exit(out int length)
{
ref Av1SymbolWriter w = ref this.writer;
return w.Exit(out length);
}
/// <summary>
/// Releases output memory that has not been transferred by <see cref="Exit()"/>.
/// </summary> /// </summary>
public void Dispose() public void Dispose()
{ {

29
src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolWriter.cs

@ -129,6 +129,30 @@ internal class Av1SymbolWriter : IDisposable
/// </summary> /// </summary>
/// <returns>An owner containing the shortest byte sequence that preserves every encoded symbol.</returns> /// <returns>An owner containing the shortest byte sequence that preserves every encoded symbol.</returns>
public IMemoryOwner<byte> Exit() public IMemoryOwner<byte> Exit()
{
int length = this.FinalizeRange();
IMemoryOwner<byte> output = this.configuration.MemoryAllocator.Allocate<byte>(length);
this.memory.GetSpan(length).CopyTo(output.GetSpan()[..length]);
return output;
}
/// <summary>
/// Finalizes the range-coded sequence and transfers its current allocation without copying.
/// </summary>
/// <param name="length">The number of encoded bytes at the beginning of the returned allocation.</param>
/// <returns>The complete allocation containing the encoded byte prefix.</returns>
public IMemoryOwner<byte> Exit(out int length)
{
length = this.FinalizeRange();
return this.memory.Detach();
}
/// <summary>
/// Terminates the range-coded sequence in the current output allocation.
/// </summary>
/// <returns>The number of encoded bytes in the allocation.</returns>
private int FinalizeRange()
{ {
// Round the low endpoint into the current interval so the emitted prefix selects every symbol encoded so far, // Round the low endpoint into the current interval so the emitted prefix selects every symbol encoded so far,
// regardless of the bits that follow it. // regardless of the bits that follow it.
@ -162,10 +186,7 @@ internal class Av1SymbolWriter : IDisposable
while (s > 0); while (s > 0);
} }
IMemoryOwner<byte> output = this.configuration.MemoryAllocator.Allocate<byte>(pos); return pos;
buffer[..pos].CopyTo(output.GetSpan()[..pos]);
return output;
} }
/// <summary> /// <summary>

81
src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.Operator.cs

@ -0,0 +1,81 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling;
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
/// <content>
/// Defines the sample-storage operations used by single-tile intra encoding.
/// </content>
internal sealed partial class Av1IntraTileWriter
{
/// <summary>
/// Defines type-specific superblock encoding without coupling tile traversal to sample storage width.
/// </summary>
/// <typeparam name="TSample">The native unsigned sample storage type.</typeparam>
private interface ITileEncodingOperator<TSample>
where TSample : unmanaged
{
/// <summary>
/// Encodes and reconstructs one superblock.
/// </summary>
/// <param name="source">The coded source frame.</param>
/// <param name="reconstruction">The reconstructed frame updated by the block transforms.</param>
/// <param name="picture">The frame coding and mode-information state.</param>
/// <param name="superblock">The reusable partition and final-block decisions.</param>
/// <param name="coefficientBuffer">The frame-owned quantized coefficient and transform state.</param>
/// <param name="blockWorkspace">The reusable block arithmetic workspace.</param>
public static abstract void EncodeSuperblock(
Av1EncoderFrame<TSample> source,
Av1EncoderFrame<TSample> reconstruction,
Av1PictureControlSet picture,
Av1Superblock superblock,
Av1EncoderCoefficientBuffer coefficientBuffer,
Av1EncoderBlockWorkspace blockWorkspace);
}
/// <summary>
/// Encodes tiles stored as eight-bit samples.
/// </summary>
private readonly struct ByteOperator : ITileEncodingOperator<byte>
{
/// <inheritdoc/>
public static void EncodeSuperblock(
Av1EncoderFrame<byte> source,
Av1EncoderFrame<byte> reconstruction,
Av1PictureControlSet picture,
Av1Superblock superblock,
Av1EncoderCoefficientBuffer coefficientBuffer,
Av1EncoderBlockWorkspace blockWorkspace)
=> Av1IntraSuperblockEncoder.Encode(
source,
reconstruction,
picture,
superblock,
coefficientBuffer,
blockWorkspace);
}
/// <summary>
/// Encodes tiles stored as high-bit-depth samples.
/// </summary>
private readonly struct UInt16Operator : ITileEncodingOperator<ushort>
{
/// <inheritdoc/>
public static void EncodeSuperblock(
Av1EncoderFrame<ushort> source,
Av1EncoderFrame<ushort> reconstruction,
Av1PictureControlSet picture,
Av1Superblock superblock,
Av1EncoderCoefficientBuffer coefficientBuffer,
Av1EncoderBlockWorkspace blockWorkspace)
=> Av1IntraSuperblockEncoder.Encode(
source,
reconstruction,
picture,
superblock,
coefficientBuffer,
blockWorkspace);
}
}

176
src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.cs

@ -0,0 +1,176 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using System.Buffers;
using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy;
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit;
using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling;
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
/// <summary>
/// Encodes and owns one range-coded all-intra tile payload.
/// </summary>
internal sealed partial class Av1IntraTileWriter : IAv1TileWriter, IDisposable
{
private IMemoryOwner<byte>? tileData;
private readonly int tileDataLength;
/// <summary>
/// Initializes a new instance of the <see cref="Av1IntraTileWriter"/> class for eight-bit samples.
/// </summary>
/// <param name="configuration">The configuration providing tile output memory.</param>
/// <param name="source">The coded source frame.</param>
/// <param name="reconstruction">The reconstructed frame updated during encoding.</param>
/// <param name="picture">The frame coding and mode-information state.</param>
/// <param name="coefficientBuffer">The frame-owned quantized coefficient and transform state.</param>
/// <param name="superblockWorkspace">The reusable partition and final-block decision workspace.</param>
/// <param name="blockWorkspace">The reusable block arithmetic workspace.</param>
/// <param name="initialSize">The estimated encoded tile size in bytes.</param>
public Av1IntraTileWriter(
Configuration configuration,
Av1EncoderFrame<byte> source,
Av1EncoderFrame<byte> reconstruction,
Av1PictureControlSet picture,
Av1EncoderCoefficientBuffer coefficientBuffer,
Av1EncoderSuperblockWorkspace superblockWorkspace,
Av1EncoderBlockWorkspace blockWorkspace,
int initialSize)
{
this.tileData = Encode<byte, ByteOperator>(
configuration,
source,
reconstruction,
picture,
coefficientBuffer,
superblockWorkspace,
blockWorkspace,
initialSize,
out this.tileDataLength);
}
/// <summary>
/// Initializes a new instance of the <see cref="Av1IntraTileWriter"/> class for high-bit-depth samples.
/// </summary>
/// <param name="configuration">The configuration providing tile output memory.</param>
/// <param name="source">The coded source frame.</param>
/// <param name="reconstruction">The reconstructed frame updated during encoding.</param>
/// <param name="picture">The frame coding and mode-information state.</param>
/// <param name="coefficientBuffer">The frame-owned quantized coefficient and transform state.</param>
/// <param name="superblockWorkspace">The reusable partition and final-block decision workspace.</param>
/// <param name="blockWorkspace">The reusable block arithmetic workspace.</param>
/// <param name="initialSize">The estimated encoded tile size in bytes.</param>
public Av1IntraTileWriter(
Configuration configuration,
Av1EncoderFrame<ushort> source,
Av1EncoderFrame<ushort> reconstruction,
Av1PictureControlSet picture,
Av1EncoderCoefficientBuffer coefficientBuffer,
Av1EncoderSuperblockWorkspace superblockWorkspace,
Av1EncoderBlockWorkspace blockWorkspace,
int initialSize)
{
this.tileData = Encode<ushort, UInt16Operator>(
configuration,
source,
reconstruction,
picture,
coefficientBuffer,
superblockWorkspace,
blockWorkspace,
initialSize,
out this.tileDataLength);
}
/// <inheritdoc/>
public ReadOnlySpan<byte> GetTileData(int tileNum)
{
ObjectDisposedException.ThrowIf(this.tileData is null, this);
return this.tileData.Memory.Span[..this.tileDataLength];
}
/// <summary>
/// Returns the detached range-coded tile allocation to the configured allocator.
/// </summary>
public void Dispose()
{
this.tileData?.Dispose();
this.tileData = null;
}
private static IMemoryOwner<byte> Encode<TSample, TOperator>(
Configuration configuration,
Av1EncoderFrame<TSample> source,
Av1EncoderFrame<TSample> reconstruction,
Av1PictureControlSet picture,
Av1EncoderCoefficientBuffer coefficientBuffer,
Av1EncoderSuperblockWorkspace superblockWorkspace,
Av1EncoderBlockWorkspace blockWorkspace,
int initialSize,
out int tileDataLength)
where TSample : unmanaged
where TOperator : struct, ITileEncodingOperator<TSample>
{
ObuFrameHeader frameHeader = picture.Parent.FrameHeader;
ObuSequenceHeader sequenceHeader = picture.Sequence.SequenceHeader;
const ushort TileIndex = 0;
Av1TileInfo tile = new(0, 0, frameHeader);
Av1Superblock superblock = new()
{
Workspace = superblockWorkspace,
TileInfo = tile
};
Point firstModeInfoPosition = new(tile.ModeInfoColumnStart, tile.ModeInfoRowStart);
Av1TileWriter.Av1EntropyCodingContext entropyContext = new()
{
MacroBlock = new Av1MacroBlockD { Tile = tile },
MacroBlockModeInfo = picture.GetMacroBlockModeInfo(firstModeInfoPosition)
};
using Av1SymbolEncoder writer = new(
configuration,
initialSize,
frameHeader.QuantizationParameters.BaseQIndex,
updateCdf: !frameHeader.DisableCdfUpdate);
int superblockModeInfoSize = sequenceHeader.SuperblockModeInfoSize;
int superblockShift = sequenceHeader.SuperblockSizeLog2 - Av1Constants.ModeInfoSizeLog2;
for (int modeInfoRow = tile.ModeInfoRowStart;
modeInfoRow < tile.ModeInfoRowEnd;
modeInfoRow += superblockModeInfoSize)
{
for (int modeInfoColumn = tile.ModeInfoColumnStart;
modeInfoColumn < tile.ModeInfoColumnEnd;
modeInfoColumn += superblockModeInfoSize)
{
int superblockRow = modeInfoRow >> superblockShift;
int superblockColumn = modeInfoColumn >> superblockShift;
superblock.Index = (superblockRow * coefficientBuffer.SuperblockColumnCount) + superblockColumn;
entropyContext.SuperblockOrigin = new Point(
modeInfoColumn << Av1Constants.ModeInfoSizeLog2,
modeInfoRow << Av1Constants.ModeInfoSizeLog2);
// Analyze immediately before entropy coding so the reusable decision workspace and reconstructed
// neighbors remain synchronized without a second superblock-sized decision allocation.
TOperator.EncodeSuperblock(
source,
reconstruction,
picture,
superblock,
coefficientBuffer,
blockWorkspace);
Av1TileWriter.WriteSuperblock(
picture,
entropyContext,
writer,
superblock,
coefficientBuffer,
TileIndex);
}
}
return writer.Exit(out tileDataLength);
}
}

200
src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs

@ -0,0 +1,200 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using System.Buffers;
using System.Runtime.CompilerServices;
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit;
using SixLabors.ImageSharp.Memory;
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Tiling;
/// <summary>
/// Owns mode information, segmentation data, and tile-neighbor contexts for one encoded AV1 picture.
/// </summary>
internal sealed class Av1EncoderPictureBuffer : IDisposable
{
private readonly Av1EncoderModeInfoBuffer modeInfo;
private readonly IMemoryOwner<byte> stateStorage;
private readonly ByteMemoryManager<Av1PartitionContext> partitionContextMemory;
private readonly Av1NeighborArrayUnit<Av1PartitionContext>[] partitionContexts;
private readonly Av1NeighborArrayUnit<byte>[] lumaCoefficientContexts;
private readonly Av1NeighborArrayUnit<byte>[] blueCoefficientContexts;
private readonly Av1NeighborArrayUnit<byte>[] redCoefficientContexts;
private readonly Av1NeighborArrayUnit<byte>[] transformContexts;
/// <summary>
/// Initializes a new instance of the <see cref="Av1EncoderPictureBuffer"/> class.
/// </summary>
/// <param name="configuration">The configuration providing picture-lifetime memory.</param>
/// <param name="sequenceHeader">The sequence header defining superblock and chroma geometry.</param>
/// <param name="frameHeader">The frame header defining dimensions and tiles.</param>
/// <param name="width">The visible luma width.</param>
/// <param name="height">The visible luma height.</param>
public Av1EncoderPictureBuffer(
Configuration configuration,
ObuSequenceHeader sequenceHeader,
ObuFrameHeader frameHeader,
int width,
int height)
{
const int ContextAlignmentLog2 = Av1Constants.MaxSuperBlockSizeLog2 - Av1Constants.ModeInfoSizeLog2;
this.modeInfo = new Av1EncoderModeInfoBuffer(
configuration,
width,
height,
disallow4x4AllFrames: true);
int alignedModeInfoRowCount = Av1Math.AlignPowerOf2(this.modeInfo.ModeInfoRowCount, ContextAlignmentLog2);
int lumaLeftLength = alignedModeInfoRowCount;
int lumaTopLength = this.modeInfo.ModeInfoStride;
ObuColorConfig colorConfig = sequenceHeader.ColorConfig;
int chromaLeftLength = colorConfig.IsMonochrome
? 0
: lumaLeftLength >> (colorConfig.SubSamplingY ? 1 : 0);
int chromaTopLength = colorConfig.IsMonochrome
? 0
: lumaTopLength >> (colorConfig.SubSamplingX ? 1 : 0);
int tileCount = frameHeader.TilesInfo.TileColumnCount * frameHeader.TilesInfo.TileRowCount;
int lumaContextLength = checked(lumaLeftLength + lumaTopLength);
int chromaContextLength = checked(chromaLeftLength + chromaTopLength);
int byteContextLengthPerTile = checked((2 * lumaContextLength) + (2 * chromaContextLength));
int partitionContextLength = checked(tileCount * lumaContextLength);
int segmentationLength = checked(this.modeInfo.ModeInfoColumnCount * this.modeInfo.ModeInfoRowCount);
int partitionContextSize = Unsafe.SizeOf<Av1PartitionContext>();
int partitionStorageOffset = checked(
((segmentationLength + partitionContextSize - 1) / partitionContextSize) * partitionContextSize);
int partitionStorageLength = checked(
partitionContextLength * Unsafe.SizeOf<Av1PartitionContext>());
int byteContextStorageOffset = checked(partitionStorageOffset + partitionStorageLength);
int stateStorageLength = checked(byteContextStorageOffset + (tileCount * byteContextLengthPerTile));
// Segmentation and every tile edge share one clean picture lifetime. The partition region begins at its
// native alignment, while typed views keep the entropy writer independent from the packed byte owner.
this.stateStorage = configuration.MemoryAllocator.Allocate<byte>(
stateStorageLength,
AllocationOptions.Clean);
Memory<byte> stateStorage = this.stateStorage.Memory[..stateStorageLength];
this.partitionContextMemory = new ByteMemoryManager<Av1PartitionContext>(
stateStorage.Slice(partitionStorageOffset, partitionStorageLength));
Memory<Av1PartitionContext> partitionStorage = this.partitionContextMemory.Memory;
Memory<byte> byteContextStorage = stateStorage[byteContextStorageOffset..];
this.partitionContexts = new Av1NeighborArrayUnit<Av1PartitionContext>[tileCount];
this.lumaCoefficientContexts = new Av1NeighborArrayUnit<byte>[tileCount];
this.blueCoefficientContexts = new Av1NeighborArrayUnit<byte>[tileCount];
this.redCoefficientContexts = new Av1NeighborArrayUnit<byte>[tileCount];
this.transformContexts = new Av1NeighborArrayUnit<byte>[tileCount];
int[][] cdefPreset = new int[tileCount][];
int[] previousQIndex = new int[tileCount];
for (int tileIndex = 0; tileIndex < tileCount; tileIndex++)
{
this.partitionContexts[tileIndex] = new Av1NeighborArrayUnit<Av1PartitionContext>(
partitionStorage.Slice(tileIndex * lumaContextLength, lumaContextLength),
lumaLeftLength,
lumaTopLength)
{
GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2
};
int byteContextOffset = tileIndex * byteContextLengthPerTile;
this.lumaCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit<byte>(
byteContextStorage.Slice(byteContextOffset, lumaContextLength),
lumaLeftLength,
lumaTopLength)
{
GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2
};
byteContextOffset += lumaContextLength;
this.blueCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit<byte>(
byteContextStorage.Slice(byteContextOffset, chromaContextLength),
chromaLeftLength,
chromaTopLength)
{
GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2
};
byteContextOffset += chromaContextLength;
this.redCoefficientContexts[tileIndex] = new Av1NeighborArrayUnit<byte>(
byteContextStorage.Slice(byteContextOffset, chromaContextLength),
chromaLeftLength,
chromaTopLength)
{
GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2
};
byteContextOffset += chromaContextLength;
this.transformContexts[tileIndex] = new Av1NeighborArrayUnit<byte>(
byteContextStorage.Slice(byteContextOffset, lumaContextLength),
lumaLeftLength,
lumaTopLength)
{
GranularityNormalLog2 = Av1Constants.ModeInfoSizeLog2
};
cdefPreset[tileIndex] = [-1, -1, -1, -1];
previousQIndex[tileIndex] = frameHeader.QuantizationParameters.BaseQIndex;
}
this.Picture = new Av1PictureControlSet
{
PartitionContexts = this.partitionContexts,
LuminanceDcSignLevelCoefficientNeighbors = this.lumaCoefficientContexts,
CbDcSignLevelCoefficientNeighbors = this.blueCoefficientContexts,
CrDcSignLevelCoefficientNeighbors = this.redCoefficientContexts,
TransformFunctionContexts = this.transformContexts,
Sequence = new Av1SequenceControlSet { SequenceHeader = sequenceHeader },
Parent = new Av1PictureParentControlSet
{
Common = new Av1EncoderCommon
{
ModeInfoColumnCount = this.modeInfo.ModeInfoColumnCount,
ModeInfoRowCount = this.modeInfo.ModeInfoRowCount,
ModeInfoStride = this.modeInfo.ModeInfoStride,
FrameSize = frameHeader.FrameSize,
TilesInfo = frameHeader.TilesInfo
},
FrameHeader = frameHeader,
PreviousQIndex = previousQIndex
},
SegmentationNeighborMap = stateStorage[..segmentationLength],
ModeInfoGrid = this.modeInfo.Grid,
ModeInfoAllocation = this.modeInfo.Allocation,
ModeInfoStride = this.modeInfo.ModeInfoStride,
Disallow4x4AllFrames = this.modeInfo.Disallow4x4AllFrames,
CdefPreset = cdefPreset
};
}
/// <summary>
/// Gets the non-owning picture state consumed by superblock analysis and tile writing.
/// </summary>
public Av1PictureControlSet Picture { get; }
/// <summary>
/// Returns every picture-lifetime allocation to the configured allocator.
/// </summary>
public void Dispose()
{
foreach (Av1NeighborArrayUnit<Av1PartitionContext> context in this.partitionContexts)
{
context.Dispose();
}
for (int tileIndex = 0; tileIndex < this.lumaCoefficientContexts.Length; tileIndex++)
{
this.lumaCoefficientContexts[tileIndex].Dispose();
this.blueCoefficientContexts[tileIndex].Dispose();
this.redCoefficientContexts[tileIndex].Dispose();
this.transformContexts[tileIndex].Dispose();
}
this.stateStorage.Dispose();
this.modeInfo.Dispose();
}
}

42
src/ImageSharp/Formats/Heif/Av1/Tiling/Av1NeighborArrayUnit.cs

@ -16,7 +16,17 @@ internal sealed class Av1NeighborArrayUnit<T> : IDisposable
/// <summary> /// <summary>
/// Owns the contiguous neighbor storage until this instance is disposed. /// Owns the contiguous neighbor storage until this instance is disposed.
/// </summary> /// </summary>
private IMemoryOwner<T>? memory; private IMemoryOwner<T>? owner;
/// <summary>
/// The contiguous neighbor storage, whether owned directly or supplied by a picture owner.
/// </summary>
private Memory<T> memory;
/// <summary>
/// Indicates whether the neighbor view has been disposed.
/// </summary>
private bool isDisposed;
/// <summary> /// <summary>
/// The number of context values exposed to blocks on the right. /// The number of context values exposed to blocks on the right.
@ -42,7 +52,21 @@ internal sealed class Av1NeighborArrayUnit<T> : IDisposable
// Both context edges share the picture lifetime, so one clean allocator-backed buffer // Both context edges share the picture lifetime, so one clean allocator-backed buffer
// preserves their zero-initialized starting state without separate owner lifetimes. // preserves their zero-initialized starting state without separate owner lifetimes.
this.memory = configuration.MemoryAllocator.Allocate<T>(totalLength, AllocationOptions.Clean); this.owner = configuration.MemoryAllocator.Allocate<T>(totalLength, AllocationOptions.Clean);
this.memory = this.owner.Memory[..totalLength];
}
/// <summary>
/// Initializes a new instance of the <see cref="Av1NeighborArrayUnit{T}"/> class over non-owning picture-lifetime storage.
/// </summary>
/// <param name="memory">The contiguous left and top context storage.</param>
/// <param name="leftSize">The number of values in the left-neighbor storage.</param>
/// <param name="topSize">The number of values in the top-neighbor storage.</param>
public Av1NeighborArrayUnit(Memory<T> memory, int leftSize, int topSize)
{
this.leftLength = leftSize;
this.topLength = topSize;
this.memory = memory[..checked(leftSize + topSize)];
} }
/// <summary> /// <summary>
@ -69,8 +93,8 @@ internal sealed class Av1NeighborArrayUnit<T> : IDisposable
{ {
get get
{ {
ObjectDisposedException.ThrowIf(this.memory is null, this); ObjectDisposedException.ThrowIf(this.isDisposed, this);
return this.memory.Memory.Span[..this.leftLength]; return this.memory.Span[..this.leftLength];
} }
} }
@ -81,8 +105,8 @@ internal sealed class Av1NeighborArrayUnit<T> : IDisposable
{ {
get get
{ {
ObjectDisposedException.ThrowIf(this.memory is null, this); ObjectDisposedException.ThrowIf(this.isDisposed, this);
return this.memory.Memory.Span.Slice(this.leftLength, this.topLength); return this.memory.Span.Slice(this.leftLength, this.topLength);
} }
} }
@ -110,8 +134,10 @@ internal sealed class Av1NeighborArrayUnit<T> : IDisposable
/// </summary> /// </summary>
public void Dispose() public void Dispose()
{ {
this.memory?.Dispose(); this.owner?.Dispose();
this.memory = null; this.owner = null;
this.memory = Memory<T>.Empty;
this.isDisposed = true;
} }
/// <summary> /// <summary>

4
src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureControlSet.cs

@ -46,7 +46,7 @@ internal class Av1PictureControlSet
/// <summary> /// <summary>
/// Gets or sets the frame segmentation identifiers used for spatial prediction. /// Gets or sets the frame segmentation identifiers used for spatial prediction.
/// </summary> /// </summary>
public required byte[] SegmentationNeighborMap { get; set; } public required Memory<byte> SegmentationNeighborMap { get; set; }
/// <summary> /// <summary>
/// Gets or sets the frame grid that maps each 4x4 position to its mode-information allocation index. /// Gets or sets the frame grid that maps each 4x4 position to its mode-information allocation index.
@ -130,7 +130,7 @@ internal class Av1PictureControlSet
public void UpdateSegmentation(Av1BlockSize blockSize, Point origin, int segmentId) public void UpdateSegmentation(Av1BlockSize blockSize, Point origin, int segmentId)
{ {
Av1EncoderCommon cm = this.Parent.Common; Av1EncoderCommon cm = this.Parent.Common;
Span<byte> segment_ids = this.SegmentationNeighborMap; Span<byte> segment_ids = this.SegmentationNeighborMap.Span;
int mi_col = origin.X >> Av1Constants.ModeInfoSizeLog2; int mi_col = origin.X >> Av1Constants.ModeInfoSizeLog2;
int mi_row = origin.Y >> Av1Constants.ModeInfoSizeLog2; int mi_row = origin.Y >> Av1Constants.ModeInfoSizeLog2;
int mi_offset = (mi_row * cm.ModeInfoColumnCount) + mi_col; int mi_offset = (mi_row * cm.ModeInfoColumnCount) + mi_col;

2
src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs

@ -1673,7 +1673,7 @@ internal partial class Av1TileWriter
bool left_available = xd.IsLeftAvailable; bool left_available = xd.IsLeftAvailable;
bool up_available = xd.IsUpAvailable; bool up_available = xd.IsUpAvailable;
Av1EncoderCommon cm = pcs.Parent.Common; Av1EncoderCommon cm = pcs.Parent.Common;
Span<byte> segmentation_map = pcs.SegmentationNeighborMap; Span<byte> segmentation_map = pcs.SegmentationNeighborMap.Span;
if (up_available && left_available) if (up_available && left_available)
{ {

19
src/ImageSharp/Memory/AutoExpandingMemory.cs

@ -14,6 +14,7 @@ internal sealed class AutoExpandingMemory<T> : IDisposable
private const int IncreaseFactor = 5; private const int IncreaseFactor = 5;
private readonly Configuration configuration; private readonly Configuration configuration;
private IMemoryOwner<T> allocation; private IMemoryOwner<T> allocation;
private bool isDetached;
public AutoExpandingMemory(Configuration configuration, int initialSize) public AutoExpandingMemory(Configuration configuration, int initialSize)
{ {
@ -45,7 +46,23 @@ internal sealed class AutoExpandingMemory<T> : IDisposable
public Span<T> GetEntireSpan() public Span<T> GetEntireSpan()
=> this.GetSpan(this.Capacity); => this.GetSpan(this.Capacity);
public void Dispose() => this.allocation.Dispose(); /// <summary>
/// Transfers the current allocation to the caller without copying its contents.
/// </summary>
/// <returns>The allocation previously owned by this instance.</returns>
public IMemoryOwner<T> Detach()
{
this.isDetached = true;
return this.allocation;
}
public void Dispose()
{
if (!this.isDetached)
{
this.allocation.Dispose();
}
}
private void EnsureCapacity(int requestedSize) private void EnsureCapacity(int requestedSize)
{ {

6
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1CoefficientsEntropyTests.cs

@ -132,7 +132,7 @@ public class Av1CoefficientsEntropyTests
FrameHeader = new ObuFrameHeader(), FrameHeader = new ObuFrameHeader(),
PreviousQIndex = [] PreviousQIndex = []
}, },
SegmentationNeighborMap = [], SegmentationNeighborMap = Memory<byte>.Empty,
ModeInfoGrid = grid, ModeInfoGrid = grid,
ModeInfoAllocation = allocation, ModeInfoAllocation = allocation,
ModeInfoStride = 4, ModeInfoStride = 4,
@ -442,7 +442,7 @@ public class Av1CoefficientsEntropyTests
for (int column = 0; column < 8; column++) for (int column = 0; column < 8; column++)
{ {
byte expected = row is 3 or 4 && column >= 2 && column < 6 ? (byte)5 : (byte)0; byte expected = row is 3 or 4 && column >= 2 && column < 6 ? (byte)5 : (byte)0;
Assert.Equal(expected, picture.SegmentationNeighborMap[(row * 8) + column]); Assert.Equal(expected, picture.SegmentationNeighborMap.Span[(row * 8) + column]);
} }
} }
} }
@ -1165,7 +1165,7 @@ public class Av1CoefficientsEntropyTests
FrameHeader = frameHeader, FrameHeader = frameHeader,
PreviousQIndex = [] PreviousQIndex = []
}, },
SegmentationNeighborMap = [], SegmentationNeighborMap = Memory<byte>.Empty,
ModeInfoGrid = modeInfoGrid, ModeInfoGrid = modeInfoGrid,
ModeInfoAllocation = modeInfoAllocation, ModeInfoAllocation = modeInfoAllocation,
ModeInfoStride = modeInfoColumnCount, ModeInfoStride = modeInfoColumnCount,

76
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderModeInfoBufferTests.cs

@ -46,6 +46,80 @@ public class Av1EncoderModeInfoBufferTests
Assert.Equal(allocation.AllocationId, returned.AllocationId); Assert.Equal(allocation.AllocationId, returned.AllocationId);
} }
[Fact]
public void PictureBufferPacksAllPictureStateIntoTwoAllocatorOwners()
{
const int Width = 16;
const int Height = 16;
TestMemoryAllocator allocator = new();
allocator.EnableNonThreadSafeLogging();
Configuration configuration = Configuration.Default.Clone();
configuration.MemoryAllocator = allocator;
ObuColorConfig colorConfig = new()
{
IsMonochrome = false,
SubSamplingX = true,
SubSamplingY = true,
BitDepth = Av1BitDepth.EightBit
};
ObuTileGroupHeader tiles = new()
{
TileColumnCount = 1,
TileRowCount = 1
};
tiles.TileColumnStartModeInfo[1] = Width >> Av1Constants.ModeInfoSizeLog2;
tiles.TileRowStartModeInfo[1] = Height >> Av1Constants.ModeInfoSizeLog2;
ObuSequenceHeader sequenceHeader = new()
{
Use128x128Superblock = true,
ColorConfig = colorConfig
};
ObuFrameHeader frameHeader = new()
{
ModeInfoColumnCount = Width >> Av1Constants.ModeInfoSizeLog2,
ModeInfoRowCount = Height >> Av1Constants.ModeInfoSizeLog2,
TilesInfo = tiles
};
TestMemoryAllocator.AllocationRequest[] allocations;
using (Av1EncoderPictureBuffer buffer = new(
configuration,
sequenceHeader,
frameHeader,
Width,
Height))
{
allocations = allocator.AllocationLog.ToArray();
Assert.Equal(2, allocations.Length);
Assert.Equal(typeof(byte), allocations[0].ElementType);
Assert.Equal(6_144, allocations[0].Length);
Assert.Equal(AllocationOptions.Clean, allocations[0].AllocationOptions);
Assert.Equal(typeof(byte), allocations[1].ElementType);
Assert.Equal(336, allocations[1].Length);
Assert.Equal(AllocationOptions.Clean, allocations[1].AllocationOptions);
Assert.Empty(allocator.ReturnLog);
Av1PictureControlSet picture = buffer.Picture;
Assert.Equal(16, picture.SegmentationNeighborMap.Length);
Assert.Equal(32, picture.PartitionContexts[0].Left.Length);
Assert.Equal(32, picture.PartitionContexts[0].Top.Length);
Assert.Equal(32, picture.LuminanceDcSignLevelCoefficientNeighbors[0].Left.Length);
Assert.Equal(32, picture.LuminanceDcSignLevelCoefficientNeighbors[0].Top.Length);
Assert.Equal(16, picture.CbDcSignLevelCoefficientNeighbors[0].Left.Length);
Assert.Equal(16, picture.CbDcSignLevelCoefficientNeighbors[0].Top.Length);
Assert.Equal(32, picture.TransformFunctionContexts[0].Left.Length);
Assert.Equal(32, picture.TransformFunctionContexts[0].Top.Length);
}
Assert.Equal(2, allocator.ReturnLog.Count);
Assert.Equal(
allocations.Select(x => x.AllocationId).Order(),
allocator.ReturnLog.Select(x => x.AllocationId).Order());
}
[Theory] [Theory]
[InlineData(false, 2, 3, 98)] [InlineData(false, 2, 3, 98)]
[InlineData(true, 2, 2, 17)] [InlineData(true, 2, 2, 17)]
@ -120,7 +194,7 @@ public class Av1EncoderModeInfoBufferTests
FrameHeader = frameHeader, FrameHeader = frameHeader,
PreviousQIndex = [] PreviousQIndex = []
}, },
SegmentationNeighborMap = [], SegmentationNeighborMap = Memory<byte>.Empty,
ModeInfoGrid = buffer.Grid, ModeInfoGrid = buffer.Grid,
ModeInfoAllocation = buffer.Allocation, ModeInfoAllocation = buffer.Allocation,
ModeInfoStride = buffer.ModeInfoStride, ModeInfoStride = buffer.ModeInfoStride,

37
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs

@ -55,6 +55,43 @@ public class Av1EntropyTests
Assert.Equal(allocation.HashCodeOfBuffer, returned.HashCodeOfBuffer); Assert.Equal(allocation.HashCodeOfBuffer, returned.HashCodeOfBuffer);
} }
[Fact]
public void SymbolWriterTransfersExistingOutputAllocationWithoutCopy()
{
const int initialSize = 257;
TestMemoryAllocator allocator = new();
allocator.EnableNonThreadSafeLogging();
Configuration configuration = Configuration.Default.Clone();
configuration.MemoryAllocator = allocator;
TestMemoryAllocator.AllocationRequest allocation;
IMemoryOwner<byte> encoded;
int length;
using (Av1SymbolWriter writer = new(configuration, initialSize, updateCdf: false))
{
writer.WriteBoolean(false, 16_384);
writer.WriteBoolean(false, 16_384);
writer.WriteBoolean(true, 512);
writer.WriteBoolean(false, 8_192);
allocation = Assert.Single(allocator.AllocationLog);
encoded = writer.Exit(out length);
Assert.Single(allocator.AllocationLog);
Assert.Empty(allocator.ReturnLog);
}
Assert.Empty(allocator.ReturnLog);
using (encoded)
{
Assert.Equal(2, length);
Assert.Equal(initialSize, encoded.Memory.Length);
Assert.Equal(63, encoded.Memory.Span[0]);
}
TestMemoryAllocator.ReturnRequest returned = Assert.Single(allocator.ReturnLog);
Assert.Equal(allocation.AllocationId, returned.AllocationId);
}
[Fact] [Fact]
public void SymbolEncoderRentsCoefficientScratchOnlyForNonzeroBlocks() public void SymbolEncoderRentsCoefficientScratchOnlyForNonzeroBlocks()
{ {

183
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs

@ -216,6 +216,46 @@ public class Av1IntraSuperblockEncoderTests
Assert.Equal(256, entropyContext.CodedAreaSuperblock); Assert.Equal(256, entropyContext.CodedAreaSuperblock);
Assert.Equal(64, entropyContext.CodedAreaSuperblockUv); Assert.Equal(64, entropyContext.CodedAreaSuperblockUv);
Assert.NotEqual(0, encoded.GetSpan().Length); Assert.NotEqual(0, encoded.GetSpan().Length);
using Av1EncoderFrameBuffer<byte> tileReconstruction = new(
Configuration.Default,
Width,
Height,
8,
Av1ColorFormat.Yuv420,
1,
1);
ClearPlane(tileReconstruction.Luma);
ClearPlane(Assert.IsType<Buffer2D<byte>>(tileReconstruction.ChromaBlue));
ClearPlane(Assert.IsType<Buffer2D<byte>>(tileReconstruction.ChromaRed));
using Av1EncoderPictureBuffer tilePicture = new(
Configuration.Default,
picture.Sequence.SequenceHeader,
picture.Parent.FrameHeader,
Width,
Height);
using Av1EncoderCoefficientBuffer tileCoefficients = new(
Configuration.Default,
picture.Sequence.SequenceHeader,
Width,
Height);
using Av1EncoderSuperblockWorkspace tileSuperblockWorkspace = new(Configuration.Default);
using Av1EncoderBlockWorkspace tileBlockWorkspace = new(Configuration.Default);
using Av1IntraTileWriter tileWriter = new(
Configuration.Default,
source.Frame,
tileReconstruction.Frame,
tilePicture.Picture,
tileCoefficients,
tileSuperblockWorkspace,
tileBlockWorkspace,
initialSize: 512);
// The production tile traversal must be byte-identical to the explicit analyze-then-write composition above.
Assert.True(encoded.GetSpan().SequenceEqual(tileWriter.GetTileData(0)));
} }
[Fact] [Fact]
@ -302,6 +342,149 @@ public class Av1IntraSuperblockEncoderTests
Assert.NotEqual((ushort)0, coefficients.GetTransformBlockSpan(0, Av1Plane.Y)[0].EndOfBlock); Assert.NotEqual((ushort)0, coefficients.GetTransformBlockSpan(0, Av1Plane.Y)[0].EndOfBlock);
Assert.Equal(0, coefficients.GetPlaneSpan(0, Av1Plane.U).Length); Assert.Equal(0, coefficients.GetPlaneSpan(0, Av1Plane.U).Length);
Assert.Equal(0, coefficients.GetPlaneSpan(0, Av1Plane.V).Length); Assert.Equal(0, coefficients.GetPlaneSpan(0, Av1Plane.V).Length);
using Av1EncoderFrameBuffer<ushort> tileReconstruction = new(
Configuration.Default,
Width,
Height,
12,
Av1ColorFormat.Yuv400,
0,
0);
ClearPlane(tileReconstruction.Luma);
using Av1EncoderPictureBuffer tilePicture = new(
Configuration.Default,
picture.Sequence.SequenceHeader,
picture.Parent.FrameHeader,
Width,
Height);
using Av1EncoderCoefficientBuffer tileCoefficients = new(
Configuration.Default,
picture.Sequence.SequenceHeader,
Width,
Height);
using Av1EncoderSuperblockWorkspace tileSuperblockWorkspace = new(Configuration.Default);
using Av1EncoderBlockWorkspace tileBlockWorkspace = new(Configuration.Default);
using Av1IntraTileWriter tileWriter = new(
Configuration.Default,
source.Frame,
tileReconstruction.Frame,
tilePicture.Picture,
tileCoefficients,
tileSuperblockWorkspace,
tileBlockWorkspace,
initialSize: 256);
Assert.NotEqual(0, tileWriter.GetTileData(0).Length);
ushort reconstructedSample = tileReconstruction.Frame.CodedView
.GetPlane(Av1Plane.Y)
.DangerousGetRowSpan(0)[0];
Assert.InRange(reconstructedSample, (ushort)(byte.MaxValue + 1), (ushort)4095);
}
[Fact]
public void TileWriterMapsClippedRasterTraversalToEverySuperblockCoefficientSegment()
{
const int Width = 72;
const int Height = 72;
const int QIndex = 53;
ObuColorConfig colorConfig = new()
{
IsMonochrome = true,
SubSamplingX = true,
SubSamplingY = true,
BitDepth = Av1BitDepth.EightBit
};
ObuTileGroupHeader tiles = new()
{
TileColumnCount = 1,
TileRowCount = 1
};
int modeInfoColumnCount = Width >> Av1Constants.ModeInfoSizeLog2;
int modeInfoRowCount = Height >> Av1Constants.ModeInfoSizeLog2;
tiles.TileColumnStartModeInfo[1] = modeInfoColumnCount;
tiles.TileRowStartModeInfo[1] = modeInfoRowCount;
ObuSequenceHeader sequenceHeader = new()
{
Use128x128Superblock = false,
ColorConfig = colorConfig
};
ObuFrameHeader frameHeader = new()
{
ModeInfoColumnCount = modeInfoColumnCount,
ModeInfoRowCount = modeInfoRowCount,
TilesInfo = tiles
};
frameHeader.QuantizationParameters.BaseQIndex = QIndex;
frameHeader.QuantizationParameters.QIndex.Fill(QIndex);
using Av1EncoderFrameBuffer<byte> source = new(
Configuration.Default,
Width,
Height,
8,
Av1ColorFormat.Yuv400,
0,
0);
using Av1EncoderFrameBuffer<byte> reconstruction = new(
Configuration.Default,
Width,
Height,
8,
Av1ColorFormat.Yuv400,
0,
0);
FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.Y), 251, 29);
ClearPlane(reconstruction.Luma);
using Av1EncoderPictureBuffer picture = new(
Configuration.Default,
sequenceHeader,
frameHeader,
Width,
Height);
using Av1EncoderCoefficientBuffer coefficients = new(
Configuration.Default,
sequenceHeader,
Width,
Height);
using Av1EncoderSuperblockWorkspace superblockWorkspace = new(Configuration.Default);
using Av1EncoderBlockWorkspace blockWorkspace = new(Configuration.Default);
using Av1IntraTileWriter tileWriter = new(
Configuration.Default,
source.Frame,
reconstruction.Frame,
picture.Picture,
coefficients,
superblockWorkspace,
blockWorkspace,
initialSize: 4096);
Assert.Equal(4, coefficients.SuperblockCount);
for (int superblockIndex = 0; superblockIndex < coefficients.SuperblockCount; superblockIndex++)
{
Assert.NotEqual(
(ushort)0,
coefficients.GetTransformBlockSpan(superblockIndex, Av1Plane.Y)[0].EndOfBlock);
}
Assert.NotEqual(
(byte)0,
reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y).DangerousGetRowSpan(Height - 1)[Width - 1]);
ref Av1MacroBlockModeInfo bottomRight = ref picture.Picture.GetMacroBlockModeInfo(new Point(16, 16));
Assert.Equal(Av1BlockSize.Block8x8, bottomRight.Block.BlockSize);
Assert.NotEqual(0, tileWriter.GetTileData(0).Length);
} }
private static Av1PictureControlSet CreatePicture( private static Av1PictureControlSet CreatePicture(

Loading…
Cancel
Save