Browse Source

Add AV1 intra-block-copy hash search

pull/2633/head
James Jackson-South 1 month ago
parent
commit
0804e375e8
  1. 2
      HEIF_IMPLEMENTATION_PLAN.md
  2. 20
      src/ImageSharp/Formats/Heif/Av1/Entropy/Av1RateDistortion.cs
  3. 9
      src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs
  4. 35
      src/ImageSharp/Formats/Heif/Av1/Motion/Av1IntraBlockCopy.cs
  5. 465
      src/ImageSharp/Formats/Heif/Av1/Motion/Av1IntraBlockCopySearchIndex.cs
  6. 176
      src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs
  7. 13
      src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.cs
  8. 24
      src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs
  9. 5
      src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureControlSet.cs
  10. 4
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderModeInfoBufferTests.cs
  11. 12
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs
  12. 134
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs

2
HEIF_IMPLEMENTATION_PLAN.md

@ -859,7 +859,7 @@ Encoder verification contract:
- [~] Paired chroma palette clustering now preserves current libaom's squared two-component distance, first-centroid tie order, independently rounded U/V means, paired deterministic empty-cluster replacement, preceding-state retention on increased distortion, and 50-iteration limit. Keeping the source planes separate avoids interleave/deinterleave copies and improves on libaom's AVX2 ceiling with Vector512, Vector256, Vector128, then scalar dispatch through ImageSharp's shared vector-count helpers. Three independent tests cover exact paired convergence, midpoint initialization, 12-bit distance and index parity, untouched destination bounds, and every intrinsic tier. The exact Release test-project build reports 1,992 baseline warnings and zero errors; the focused three-case set, complete 8,934-case AVIF set, and complete 230-case HEIF set pass direct foreground net11 Release VSTest. Roslynk reports zero compiler errors and no touched-file analyzer warnings. Candidate integration and production activation remain in the open chroma-palette checkpoint. - [~] Paired chroma palette clustering now preserves current libaom's squared two-component distance, first-centroid tie order, independently rounded U/V means, paired deterministic empty-cluster replacement, preceding-state retention on increased distortion, and 50-iteration limit. Keeping the source planes separate avoids interleave/deinterleave copies and improves on libaom's AVX2 ceiling with Vector512, Vector256, Vector128, then scalar dispatch through ImageSharp's shared vector-count helpers. Three independent tests cover exact paired convergence, midpoint initialization, 12-bit distance and index parity, untouched destination bounds, and every intrinsic tier. The exact Release test-project build reports 1,992 baseline warnings and zero errors; the focused three-case set, complete 8,934-case AVIF set, and complete 230-case HEIF set pass direct foreground net11 Release VSTest. Roslynk reports zero compiler errors and no touched-file analyzer warnings. Candidate integration and production activation remain in the open chroma-palette checkpoint.
- [~] Live paired chroma palette selection now follows current libaom's complete 2-through-8 color-size search, U-plane neighbor-cache snapping, stable U-ordered color pairs, shared U/V index map, implicit DCT-DCT transform, and strict rate-distortion winner replacement. It improves on speed-configured libaom by applying no early header-cost pruning, keeps planar U/V source data separate, and reuses the SIMD-first prediction, residual, transform, quantization, and reconstruction operators without allocator-backed candidate storage. The production tile regression proves both palette-mode probability branches, exact paired colors and indices, coefficient-free reconstruction, and nonempty syntax. The complete 58-case intra-superblock set, 8,935-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains the next checkpoint. - [~] Live paired chroma palette selection now follows current libaom's complete 2-through-8 color-size search, U-plane neighbor-cache snapping, stable U-ordered color pairs, shared U/V index map, implicit DCT-DCT transform, and strict rate-distortion winner replacement. It improves on speed-configured libaom by applying no early header-cost pruning, keeps planar U/V source data separate, and reuses the SIMD-first prediction, residual, transform, quantization, and reconstruction operators without allocator-backed candidate storage. The production tile regression proves both palette-mode probability branches, exact paired colors and indices, coefficient-free reconstruction, and nonempty syntax. The complete 58-case intra-superblock set, 8,935-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains the next checkpoint.
- [~] Production palette activation now matches current libaom's default good-quality screen detector: it scans only complete 16x16 luma blocks, normalizes high-bit-depth samples to eight bits, admits 2-through-4-color blocks, and uses the reference's strict greater-than-ten-percent frame-area threshold. A 256-bit stack bitset and a fifth-color early exit replace libaom's larger per-block histogram without changing the decision, allocation, or source precision. The adaptive sequence flag remains enabled, the frame flag is set before picture-state allocation, and intra-block copy remains disabled. Focused regressions prove strict-threshold equality, high-bit-depth normalization, five-color rejection, emitted frame-header activation, production decode, and generated payload retention. The exact Release test-project build reports 1,992 baseline warnings and zero errors; all 8,935 AVIF cases and all 230 HEIF cases pass direct foreground net11 Release VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 30 regenerated production payloads, including the 54-byte palette case. Roslynk reports zero compiler errors and no touched-file analyzer warnings. - [~] Production palette activation now matches current libaom's default good-quality screen detector: it scans only complete 16x16 luma blocks, normalizes high-bit-depth samples to eight bits, admits 2-through-4-color blocks, and uses the reference's strict greater-than-ten-percent frame-area threshold. A 256-bit stack bitset and a fifth-color early exit replace libaom's larger per-block histogram without changing the decision, allocation, or source precision. The adaptive sequence flag remains enabled, the frame flag is set before picture-state allocation, and intra-block copy remains disabled. Focused regressions prove strict-threshold equality, high-bit-depth normalization, five-color rejection, emitted frame-header activation, production decode, and generated payload retention. The exact Release test-project build reports 1,992 baseline warnings and zero errors; all 8,935 AVIF cases and all 230 HEIF cases pass direct foreground net11 Release VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 30 regenerated production payloads, including the 54-byte palette case. Roslynk reports zero compiler errors and no touched-file analyzer warnings.
- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; mode search applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. The exact net11 Release build reports 1,992 baseline warnings and zero errors; all 1,950 entropy cases, all 9,017 AV1 cases, and all 206 non-AV1 HEIF cases pass through direct foreground VSTest, and Roslynk reports zero compiler errors with no touched-file analyzer warnings. Legal source search, joint luma/chroma rate-distortion selection, production activation, and adaptive frame-flag clearing remain before intra-block copy can be enabled. - [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The exact net11 Release build reports 1,992 baseline warnings and zero errors; all 1,968 focused entropy, ownership, and intra-block-copy cases, all 9,023 AV1 cases, and all 206 non-AV1 HEIF cases pass through direct foreground VSTest, and Roslynk reports zero compiler errors. Pixel-search fallback, joint luma/chroma rate-distortion selection, production activation, and adaptive frame-flag clearing remain before intra-block copy can be enabled.
- [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips. - [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips.
- [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips. - [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips.
- [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips. - [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips.

20
src/ImageSharp/Formats/Heif/Av1/Entropy/Av1RateDistortion.cs

@ -45,4 +45,24 @@ internal static class Av1RateDistortion
long roundedRate = (weightedRate + (1 << (Av1ProbabilityCost.CostShift - 1))) >> Av1ProbabilityCost.CostShift; long roundedRate = (weightedRate + (1 << (Av1ProbabilityCost.CostShift - 1))) >> Av1ProbabilityCost.CostShift;
return roundedRate + (distortion << 7); return roundedRate + (distortion << 7);
} }
/// <summary>
/// Gets the variance-domain cost of a full-pixel motion candidate.
/// </summary>
/// <param name="rateMultiplier">The rate weight selected by the encoder quality model.</param>
/// <param name="motionVectorRate">The motion-vector syntax rate in 1/512-bit units.</param>
/// <param name="variance">The normalized sample variance.</param>
/// <returns>The variance plus the motion-vector error cost.</returns>
public static int GetMotionSearchCost(int rateMultiplier, int motionVectorRate, int variance)
{
const int RateMultiplierShift = 6;
const int MotionErrorShift = 14;
int errorPerBit = Math.Max(rateMultiplier >> RateMultiplierShift, 1);
// Motion search compares pixel variance directly, so the syntax term is reduced to the same
// error domain instead of using the final mode-decision distortion scale.
long weightedRate = (long)motionVectorRate * errorPerBit;
int motionError = (int)((weightedRate + (1 << (MotionErrorShift - 1))) >> MotionErrorShift);
return variance + motionError;
}
} }

9
src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs

@ -707,6 +707,15 @@ internal class Av1SymbolEncoder : IDisposable
return ((rate * DisplacementVectorCostWeight) + (1 << (WeightShift - 1))) >> WeightShift; return ((rate * DisplacementVectorCostWeight) + (1 << (WeightShift - 1))) >> WeightShift;
} }
/// <summary>
/// Measures an integer intra-block-copy displacement vector for variance-domain motion search.
/// </summary>
/// <param name="value">The displacement vector to measure.</param>
/// <param name="reference">The spatially derived reference vector.</param>
/// <returns>The syntax cost in 1/512-bit units.</returns>
public int GetDisplacementVectorSearchCost(Av1MotionVector value, Av1MotionVector reference)
=> this.displacementVector.GetCost(this.writer, value, reference);
/// <summary> /// <summary>
/// Writes a complete block partition type using the selected partition context. /// Writes a complete block partition type using the selected partition context.
/// </summary> /// </summary>

35
src/ImageSharp/Formats/Heif/Av1/Motion/Av1IntraBlockCopy.cs

@ -204,6 +204,31 @@ internal static class Av1IntraBlockCopy
/// <param name="sequenceHeader">The sequence-level superblock and chroma configuration.</param> /// <param name="sequenceHeader">The sequence-level superblock and chroma configuration.</param>
/// <returns><see langword="true"/> when the complete source block is a permitted reference; otherwise, <see langword="false"/>.</returns> /// <returns><see langword="true"/> when the complete source block is a permitted reference; otherwise, <see langword="false"/>.</returns>
public static bool IsValid(Av1MotionVector vector, ref Av1PartitionInfo partitionInfo, Av1TileInfo tileInfo, ObuSequenceHeader sequenceHeader) public static bool IsValid(Av1MotionVector vector, ref Av1PartitionInfo partitionInfo, Av1TileInfo tileInfo, ObuSequenceHeader sequenceHeader)
=> IsValid(
vector,
new Point(partitionInfo.ColumnIndex, partitionInfo.RowIndex),
partitionInfo.ModeInfo.BlockSize,
partitionInfo.IsChroma,
tileInfo,
sequenceHeader);
/// <summary>
/// Determines whether an encoder displacement vector references an earlier reconstructable block inside the tile.
/// </summary>
/// <param name="vector">The displacement vector in one-eighth-sample units.</param>
/// <param name="modeInfoPosition">The current block origin in 4x4 mode-information units.</param>
/// <param name="blockSize">The current block size.</param>
/// <param name="isChroma">Indicates whether chroma subsampling constraints apply.</param>
/// <param name="tileInfo">The active tile boundaries.</param>
/// <param name="sequenceHeader">The sequence-level superblock and chroma configuration.</param>
/// <returns><see langword="true"/> when the complete source block is a permitted reference; otherwise, <see langword="false"/>.</returns>
public static bool IsValid(
Av1MotionVector vector,
Point modeInfoPosition,
Av1BlockSize blockSize,
bool isChroma,
Av1TileInfo tileInfo,
ObuSequenceHeader sequenceHeader)
{ {
const int eighthSampleScale = 8; const int eighthSampleScale = 8;
const int modeInfoSampleSize = 1 << Av1Constants.ModeInfoSizeLog2; const int modeInfoSampleSize = 1 << Av1Constants.ModeInfoSizeLog2;
@ -213,10 +238,10 @@ internal static class Av1IntraBlockCopy
return false; return false;
} }
int row = partitionInfo.RowIndex; int row = modeInfoPosition.Y;
int column = partitionInfo.ColumnIndex; int column = modeInfoPosition.X;
int blockWidth = partitionInfo.ModeInfo.BlockSize.GetWidth(); int blockWidth = blockSize.GetWidth();
int blockHeight = partitionInfo.ModeInfo.BlockSize.GetHeight(); int blockHeight = blockSize.GetHeight();
int sourceTop = (row * modeInfoSampleSize * eighthSampleScale) + vector.Row; int sourceTop = (row * modeInfoSampleSize * eighthSampleScale) + vector.Row;
int sourceLeft = (column * modeInfoSampleSize * eighthSampleScale) + vector.Column; int sourceLeft = (column * modeInfoSampleSize * eighthSampleScale) + vector.Column;
int sourceBottom = (((row * modeInfoSampleSize) + blockHeight) * eighthSampleScale) + vector.Row; int sourceBottom = (((row * modeInfoSampleSize) + blockHeight) * eighthSampleScale) + vector.Row;
@ -231,7 +256,7 @@ internal static class Av1IntraBlockCopy
} }
ObuColorConfig colorConfig = sequenceHeader.ColorConfig; ObuColorConfig colorConfig = sequenceHeader.ColorConfig;
if (partitionInfo.IsChroma && colorConfig.PlaneCount > 1) if (isChroma && colorConfig.PlaneCount > 1)
{ {
// A sub-8x8 luma block can map to a chroma block whose rounded origin lies one additional luma unit // A sub-8x8 luma block can map to a chroma block whose rounded origin lies one additional luma unit
// inside the tile. These checks prevent that chroma reference from crossing the tile boundary. // inside the tile. These checks prevent that chroma reference from crossing the tile boundary.

465
src/ImageSharp/Formats/Heif/Av1/Motion/Av1IntraBlockCopySearchIndex.cs

@ -0,0 +1,465 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using System.Runtime.InteropServices;
using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy;
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit;
using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling;
using SixLabors.ImageSharp.Memory;
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Motion;
/// <summary>
/// Indexes visible 8x8 luma blocks for intra-block-copy motion search.
/// </summary>
internal readonly struct Av1IntraBlockCopySearchIndex
{
private const int BlockSize = 8;
private const int MaximumBucketCount = 1 << 16;
private const int MaximumCandidatesPerBucket = 256;
private const uint HorizontalHashMultiplier = 257;
private const uint VerticalHashMultiplier = 65599;
private static readonly uint HorizontalLeadingWeight = GetLeadingWeight(HorizontalHashMultiplier);
private static readonly uint VerticalLeadingWeight = GetLeadingWeight(VerticalHashMultiplier);
private readonly Memory<byte> storage;
private readonly int hashLinkLength;
private readonly int bucketCount;
private readonly int headOffset;
private readonly int tailOffset;
private readonly int countOffset;
/// <summary>
/// Initializes a new instance of the <see cref="Av1IntraBlockCopySearchIndex"/> struct over picture-lifetime storage.
/// </summary>
/// <param name="storage">The packed hash-link and bucket storage.</param>
/// <param name="width">The visible luma width.</param>
/// <param name="height">The visible luma height.</param>
public Av1IntraBlockCopySearchIndex(Memory<byte> storage, int width, int height)
{
this.OriginWidth = Math.Max(0, width - BlockSize + 1);
this.OriginHeight = Math.Max(0, height - BlockSize + 1);
this.hashLinkLength = this.OriginWidth == 0 || this.OriginHeight == 0
? 0
: checked(this.OriginWidth * height);
this.bucketCount = GetBucketCount(this.OriginWidth, this.OriginHeight);
this.headOffset = checked(this.hashLinkLength * sizeof(int));
this.tailOffset = checked(this.headOffset + (this.bucketCount * sizeof(int)));
this.countOffset = checked(this.tailOffset + (this.bucketCount * sizeof(int)));
this.storage = storage;
}
/// <summary>
/// Defines sample-width-specific search arithmetic for the closed generic encoder path.
/// </summary>
/// <typeparam name="TSample">The native unsigned sample storage type.</typeparam>
internal interface ISearchOperation<TSample>
where TSample : unmanaged
{
/// <summary>
/// Converts one native sample into the unsigned hash domain.
/// </summary>
/// <param name="sample">The sample to convert.</param>
/// <returns>The unsigned sample value.</returns>
public static abstract uint GetHashSample(TSample sample);
/// <summary>
/// Compares two complete 8x8 blocks.
/// </summary>
/// <param name="plane">The plane containing both blocks.</param>
/// <param name="first">The first block origin.</param>
/// <param name="second">The second block origin.</param>
/// <returns><see langword="true"/> when every sample is equal.</returns>
public static abstract bool BlocksEqual(Buffer2DRegion<TSample> plane, Point first, Point second);
/// <summary>
/// Gets the normalized 8x8 variance between a source block and reconstructed predictor.
/// </summary>
/// <param name="source">The coded source plane.</param>
/// <param name="sourceOrigin">The source block origin.</param>
/// <param name="reconstruction">The reconstructed luma plane.</param>
/// <param name="predictionOrigin">The predictor block origin.</param>
/// <param name="bitDepth">The coded sample precision.</param>
/// <returns>The variance in the eight-bit distortion domain.</returns>
public static abstract int GetVariance(
Buffer2DRegion<TSample> source,
Point sourceOrigin,
Buffer2DRegion<TSample> reconstruction,
Point predictionOrigin,
Av1BitDepth bitDepth);
}
/// <summary>
/// Gets the visible horizontal origin count represented by the index.
/// </summary>
public int OriginWidth { get; }
/// <summary>
/// Gets the visible vertical origin count represented by the index.
/// </summary>
public int OriginHeight { get; }
/// <summary>
/// Gets the packed storage length required for a visible frame.
/// </summary>
/// <param name="width">The visible luma width.</param>
/// <param name="height">The visible luma height.</param>
/// <returns>The required byte length.</returns>
public static int GetStorageLength(int width, int height)
{
int originWidth = Math.Max(0, width - BlockSize + 1);
int originHeight = Math.Max(0, height - BlockSize + 1);
if (originWidth == 0 || originHeight == 0)
{
return 0;
}
int hashLinkLength = checked(originWidth * height);
int bucketCount = GetBucketCount(originWidth, originHeight);
return checked(
(hashLinkLength * sizeof(int)) +
(bucketCount * sizeof(int) * 2) +
(bucketCount * sizeof(ushort)));
}
/// <summary>
/// Builds the complete visible-frame hash index into its picture-lifetime storage.
/// </summary>
/// <typeparam name="TSample">The native unsigned sample storage type.</typeparam>
/// <typeparam name="TOperation">The closed sample operation.</typeparam>
/// <param name="source">The coded source luma plane.</param>
public void Initialize<TSample, TOperation>(Buffer2DRegion<TSample> source)
where TSample : unmanaged
where TOperation : struct, ISearchOperation<TSample>
{
if (this.hashLinkLength == 0)
{
return;
}
Span<int> hashesAndLinks = this.GetHashesAndLinks();
Span<int> heads = this.GetHeads();
Span<int> tails = this.GetTails();
Span<ushort> counts = this.GetCounts();
heads.Clear();
tails.Clear();
counts.Clear();
for (int row = 0; row < source.Height; row++)
{
ReadOnlySpan<TSample> sourceRow = source.DangerousGetRowSpan(row);
int hashRowOffset = row * this.OriginWidth;
uint hash = 0;
for (int column = 0; column < BlockSize; column++)
{
hash = unchecked((hash * HorizontalHashMultiplier) + TOperation.GetHashSample(sourceRow[column]));
}
hashesAndLinks[hashRowOffset] = (int)hash;
for (int column = 1; column < this.OriginWidth; column++)
{
uint previous = TOperation.GetHashSample(sourceRow[column - 1]);
uint next = TOperation.GetHashSample(sourceRow[column + BlockSize - 1]);
hash = unchecked(((hash - (previous * HorizontalLeadingWeight)) * HorizontalHashMultiplier) + next);
hashesAndLinks[hashRowOffset + column] = (int)hash;
}
}
for (int column = 0; column < this.OriginWidth; column++)
{
uint hash = 0;
for (int row = 0; row < BlockSize; row++)
{
hash = unchecked((hash * VerticalHashMultiplier) + (uint)hashesAndLinks[(row * this.OriginWidth) + column]);
}
for (int row = 0; row < this.OriginHeight; row++)
{
int position = (row * this.OriginWidth) + column;
uint previous = (uint)hashesAndLinks[position];
hashesAndLinks[position] = (int)hash;
if (row + 1 < this.OriginHeight)
{
uint next = (uint)hashesAndLinks[((row + BlockSize) * this.OriginWidth) + column];
hash = unchecked(((hash - (previous * VerticalLeadingWeight)) * VerticalHashMultiplier) + next);
}
}
}
// Coarse-to-fine insertion disperses the first 256 identical blocks across the image instead of
// retaining one dense cluster. Links occupy the hash workspace after every hash has been derived.
int step = BlockSize;
int columnOffset = 0;
int rowOffset = 0;
while (step > 1)
{
for (int column = columnOffset; column < this.OriginWidth; column += step)
{
for (int row = rowOffset; row < this.OriginHeight; row += step)
{
int position = (row * this.OriginWidth) + column;
int bucket = hashesAndLinks[position] & (this.bucketCount - 1);
if (counts[bucket] < MaximumCandidatesPerBucket)
{
int encodedPosition = position + 1;
hashesAndLinks[position] = 0;
if (heads[bucket] == 0)
{
heads[bucket] = encodedPosition;
}
else
{
hashesAndLinks[tails[bucket] - 1] = encodedPosition;
}
tails[bucket] = encodedPosition;
counts[bucket]++;
}
}
}
if (columnOffset == 0 && rowOffset == 0)
{
columnOffset = step / 2;
}
else if (columnOffset == step / 2 && rowOffset == 0)
{
columnOffset = 0;
rowOffset = step / 2;
}
else if (columnOffset == 0 && rowOffset == step / 2)
{
columnOffset = step / 2;
}
else
{
step /= 2;
columnOffset = step / 2;
rowOffset = 0;
}
}
}
/// <summary>
/// Finds the best exact-source hash candidate in the reference above and left search regions.
/// </summary>
/// <typeparam name="TSample">The native unsigned sample storage type.</typeparam>
/// <typeparam name="TOperation">The closed sample operation.</typeparam>
/// <param name="source">The coded source luma plane.</param>
/// <param name="reconstruction">The coded reconstructed luma plane.</param>
/// <param name="blockOrigin">The current 8x8 block origin.</param>
/// <param name="tile">The active tile boundaries.</param>
/// <param name="sequenceHeader">The sequence geometry and sample precision.</param>
/// <param name="writer">The live tile entropy model used for displacement rate.</param>
/// <param name="reference">The spatial displacement-vector reference.</param>
/// <param name="rateMultiplier">The active rate-distortion multiplier.</param>
/// <param name="candidates">Storage receiving the above candidate followed by the left candidate.</param>
/// <returns>The number of candidates written.</returns>
public int FindCandidates<TSample, TOperation>(
Buffer2DRegion<TSample> source,
Buffer2DRegion<TSample> reconstruction,
Point blockOrigin,
Av1TileInfo tile,
ObuSequenceHeader sequenceHeader,
Av1SymbolEncoder writer,
Av1MotionVector reference,
int rateMultiplier,
Span<Av1MotionVector> candidates)
where TSample : unmanaged
where TOperation : struct, ISearchOperation<TSample>
{
if (this.hashLinkLength == 0)
{
return 0;
}
const int ModeInfoSampleSize = 1 << Av1Constants.ModeInfoSizeLog2;
int tileLeft = tile.ModeInfoColumnStart * ModeInfoSampleSize;
int tileTop = tile.ModeInfoRowStart * ModeInfoSampleSize;
int tileRight = tile.ModeInfoColumnEnd * ModeInfoSampleSize;
int tileBottom = tile.ModeInfoRowEnd * ModeInfoSampleSize;
int superblockSize = sequenceHeader.SuperblockSize.GetWidth();
int superblockLeft = (blockOrigin.X / superblockSize) * superblockSize;
int superblockTop = (blockOrigin.Y / superblockSize) * superblockSize;
int candidateCount = 0;
if (this.TryFindCandidate<TSample, TOperation>(
source,
reconstruction,
blockOrigin,
tile,
sequenceHeader,
writer,
reference,
rateMultiplier,
tileLeft,
tileTop,
tileRight - BlockSize,
superblockTop - BlockSize,
out Av1MotionVector above))
{
candidates[candidateCount++] = above;
}
if (this.TryFindCandidate<TSample, TOperation>(
source,
reconstruction,
blockOrigin,
tile,
sequenceHeader,
writer,
reference,
rateMultiplier,
tileLeft,
tileTop,
superblockLeft - BlockSize,
Math.Min(superblockTop + superblockSize, tileBottom) - BlockSize,
out Av1MotionVector left))
{
candidates[candidateCount++] = left;
}
return candidateCount;
}
private static uint GetLeadingWeight(uint multiplier)
{
uint result = 1;
for (int i = 1; i < BlockSize; i++)
{
result = unchecked(result * multiplier);
}
return result;
}
private static int GetBucketCount(int originWidth, int originHeight)
{
int originCount = checked(originWidth * originHeight);
if (originCount == 0)
{
return 0;
}
// One power-of-two bucket per possible origin avoids libaom's fixed multi-megabyte pointer table
// on small images while retaining its 16-bit upper bound and constant-time mask lookup.
return originCount >= MaximumBucketCount
? MaximumBucketCount
: 1 << (int)Av1Math.CeilLog2((uint)originCount);
}
private static uint GetBlockHash<TSample, TOperation>(Buffer2DRegion<TSample> source, Point origin)
where TSample : unmanaged
where TOperation : struct, ISearchOperation<TSample>
{
uint blockHash = 0;
for (int row = 0; row < BlockSize; row++)
{
ReadOnlySpan<TSample> sourceRow = source.DangerousGetRowSpan(origin.Y + row);
uint rowHash = 0;
for (int column = 0; column < BlockSize; column++)
{
rowHash = unchecked(
(rowHash * HorizontalHashMultiplier) +
TOperation.GetHashSample(sourceRow[origin.X + column]));
}
blockHash = unchecked((blockHash * VerticalHashMultiplier) + rowHash);
}
return blockHash;
}
private bool TryFindCandidate<TSample, TOperation>(
Buffer2DRegion<TSample> source,
Buffer2DRegion<TSample> reconstruction,
Point blockOrigin,
Av1TileInfo tile,
ObuSequenceHeader sequenceHeader,
Av1SymbolEncoder writer,
Av1MotionVector reference,
int rateMultiplier,
int minimumColumn,
int minimumRow,
int maximumColumn,
int maximumRow,
out Av1MotionVector bestVector)
where TSample : unmanaged
where TOperation : struct, ISearchOperation<TSample>
{
bestVector = default;
if (maximumColumn < minimumColumn || maximumRow < minimumRow)
{
return false;
}
uint blockHash = GetBlockHash<TSample, TOperation>(source, blockOrigin);
int bucket = (int)(blockHash & (this.bucketCount - 1));
Span<int> hashesAndLinks = this.GetHashesAndLinks();
int encodedPosition = this.GetHeads()[bucket];
int bestCost = int.MaxValue;
bool found = false;
Point modeInfoPosition = new(
blockOrigin.X >> Av1Constants.ModeInfoSizeLog2,
blockOrigin.Y >> Av1Constants.ModeInfoSizeLog2);
while (encodedPosition != 0)
{
int position = encodedPosition - 1;
int row = position / this.OriginWidth;
int column = position - (row * this.OriginWidth);
Point candidateOrigin = new(column, row);
encodedPosition = hashesAndLinks[position];
if (column < minimumColumn || column > maximumColumn || row < minimumRow || row > maximumRow ||
!TOperation.BlocksEqual(source, blockOrigin, candidateOrigin))
{
continue;
}
Av1MotionVector vector = new(
(row - blockOrigin.Y) * 8,
(column - blockOrigin.X) * 8);
if (!Av1IntraBlockCopy.IsValid(
vector,
modeInfoPosition,
Av1BlockSize.Block8x8,
isChroma: false,
tile,
sequenceHeader))
{
continue;
}
int variance = TOperation.GetVariance(
source,
blockOrigin,
reconstruction,
candidateOrigin,
sequenceHeader.ColorConfig.BitDepth);
int rate = writer.GetDisplacementVectorSearchCost(vector, reference);
int cost = Av1RateDistortion.GetMotionSearchCost(rateMultiplier, rate, variance);
if (cost < bestCost)
{
bestCost = cost;
bestVector = vector;
found = true;
}
}
return found;
}
private Span<int> GetHashesAndLinks()
=> MemoryMarshal.Cast<byte, int>(this.storage.Span[..this.headOffset]);
private Span<int> GetHeads()
=> MemoryMarshal.Cast<byte, int>(this.storage.Span.Slice(this.headOffset, this.bucketCount * sizeof(int)));
private Span<int> GetTails()
=> MemoryMarshal.Cast<byte, int>(this.storage.Span.Slice(this.tailOffset, this.bucketCount * sizeof(int)));
private Span<ushort> GetCounts()
=> MemoryMarshal.Cast<byte, ushort>(this.storage.Span.Slice(this.countOffset, this.bucketCount * sizeof(ushort)));
}

176
src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs

@ -3,6 +3,7 @@
using System.Runtime.InteropServices; using System.Runtime.InteropServices;
using System.Runtime.Intrinsics; using System.Runtime.Intrinsics;
using SixLabors.ImageSharp.Formats.Heif.Av1.Motion;
using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma; using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction.ChromaFromLuma;
using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling;
@ -287,10 +288,28 @@ internal static partial class Av1IntraSuperblockEncoder
ref Av1EncoderTransformBlockState state); ref Av1EncoderTransformBlockState state);
} }
private static int GetNormalizedVariance(int sum, int sumOfSquares, Av1BitDepth bitDepth)
{
int coefficientShift = bitDepth.GetBitCount() - 8;
if (coefficientShift > 0)
{
// Normalize both moments before subtracting them so high-bit-depth motion search uses the
// same eight-bit distortion scale as the encoder's other rate-distortion comparisons.
int squareShift = coefficientShift * 2;
sumOfSquares = (sumOfSquares + (1 << (squareShift - 1))) >> squareShift;
sum = (sum + (1 << (coefficientShift - 1))) >> coefficientShift;
}
long variance = sumOfSquares - (((long)sum * sum) / 64);
return (int)Math.Max(variance, 0);
}
/// <summary> /// <summary>
/// Encodes blocks stored as eight-bit samples. /// Encodes blocks stored as eight-bit samples.
/// </summary> /// </summary>
internal readonly struct ByteOperator : IBlockEncodingOperator<byte> internal readonly struct ByteOperator :
IBlockEncodingOperator<byte>,
Av1IntraBlockCopySearchIndex.ISearchOperation<byte>
{ {
/// <inheritdoc/> /// <inheritdoc/>
public static Span<byte> GetLeftReference(Span<short> residual, int length) public static Span<byte> GetLeftReference(Span<short> residual, int length)
@ -299,6 +318,78 @@ internal static partial class Av1IntraSuperblockEncoder
/// <inheritdoc/> /// <inheritdoc/>
public static byte CreateSample(int value) => (byte)value; public static byte CreateSample(int value) => (byte)value;
/// <inheritdoc/>
public static uint GetHashSample(byte sample) => sample;
/// <inheritdoc/>
public static bool BlocksEqual(Buffer2DRegion<byte> plane, Point first, Point second)
{
for (int row = 0; row < 8; row++)
{
ReadOnlySpan<byte> firstRow = plane.DangerousGetRowSpan(first.Y + row)[first.X..];
ReadOnlySpan<byte> secondRow = plane.DangerousGetRowSpan(second.Y + row)[second.X..];
// An 8x8 search row occupies one machine word, so one unaligned load and comparison replaces
// eight dependent scalar branches while retaining exact collision rejection.
if (MemoryMarshal.Read<ulong>(firstRow) != MemoryMarshal.Read<ulong>(secondRow))
{
return false;
}
}
return true;
}
/// <inheritdoc/>
public static int GetVariance(
Buffer2DRegion<byte> source,
Point sourceOrigin,
Buffer2DRegion<byte> reconstruction,
Point predictionOrigin,
Av1BitDepth bitDepth)
{
int sum = 0;
int sumOfSquares = 0;
if (Vector128.IsHardwareAccelerated)
{
for (int row = 0; row < 8; row++)
{
ReadOnlySpan<byte> sourceRow = source.DangerousGetRowSpan(sourceOrigin.Y + row)[sourceOrigin.X..];
ReadOnlySpan<byte> predictionRow =
reconstruction.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X..];
Vector128<short> difference =
(Vector128.WidenLower(Vector128.CreateScalarUnsafe(MemoryMarshal.Read<ulong>(sourceRow)).AsByte()) -
Vector128.WidenLower(Vector128.CreateScalarUnsafe(MemoryMarshal.Read<ulong>(predictionRow)).AsByte()))
.AsInt16();
// Widen before squaring so signed residuals cannot wrap in 16-bit lanes.
Vector128<int> lower = Vector128.WidenLower(difference);
Vector128<int> upper = Vector128.WidenUpper(difference);
sum += Vector128.Sum(difference);
sumOfSquares += Vector128.Sum(lower * lower) + Vector128.Sum(upper * upper);
}
}
else
{
for (int row = 0; row < 8; row++)
{
ReadOnlySpan<byte> sourceRow = source.DangerousGetRowSpan(sourceOrigin.Y + row)[sourceOrigin.X..];
ReadOnlySpan<byte> predictionRow =
reconstruction.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X..];
for (int column = 0; column < 8; column++)
{
int difference = sourceRow[column] - predictionRow[column];
sum += difference;
sumOfSquares += difference * difference;
}
}
}
return GetNormalizedVariance(sum, sumOfSquares, bitDepth);
}
/// <inheritdoc/> /// <inheritdoc/>
public static void CopyPaletteSamples( public static void CopyPaletteSamples(
Buffer2DRegion<byte> source, Buffer2DRegion<byte> source,
@ -584,7 +675,9 @@ internal static partial class Av1IntraSuperblockEncoder
/// <summary> /// <summary>
/// Encodes blocks stored as high-bit-depth samples. /// Encodes blocks stored as high-bit-depth samples.
/// </summary> /// </summary>
internal readonly struct UInt16Operator : IBlockEncodingOperator<ushort> internal readonly struct UInt16Operator :
IBlockEncodingOperator<ushort>,
Av1IntraBlockCopySearchIndex.ISearchOperation<ushort>
{ {
/// <inheritdoc/> /// <inheritdoc/>
public static Span<ushort> GetLeftReference(Span<short> residual, int length) public static Span<ushort> GetLeftReference(Span<short> residual, int length)
@ -593,6 +686,85 @@ internal static partial class Av1IntraSuperblockEncoder
/// <inheritdoc/> /// <inheritdoc/>
public static ushort CreateSample(int value) => (ushort)value; public static ushort CreateSample(int value) => (ushort)value;
/// <inheritdoc/>
public static uint GetHashSample(ushort sample) => sample;
/// <inheritdoc/>
public static bool BlocksEqual(Buffer2DRegion<ushort> plane, Point first, Point second)
{
for (int row = 0; row < 8; row++)
{
ReadOnlySpan<ushort> firstRow = plane.DangerousGetRowSpan(first.Y + row)[first.X..];
ReadOnlySpan<ushort> secondRow = plane.DangerousGetRowSpan(second.Y + row)[second.X..];
if (Vector128.IsHardwareAccelerated)
{
Vector128<ushort> firstSamples = Vector128.LoadUnsafe(ref MemoryMarshal.GetReference(firstRow));
Vector128<ushort> secondSamples = Vector128.LoadUnsafe(ref MemoryMarshal.GetReference(secondRow));
if (!Vector128.EqualsAll(firstSamples, secondSamples))
{
return false;
}
}
else if (!firstRow[..8].SequenceEqual(secondRow[..8]))
{
return false;
}
}
return true;
}
/// <inheritdoc/>
public static int GetVariance(
Buffer2DRegion<ushort> source,
Point sourceOrigin,
Buffer2DRegion<ushort> reconstruction,
Point predictionOrigin,
Av1BitDepth bitDepth)
{
int sum = 0;
int sumOfSquares = 0;
if (Vector128.IsHardwareAccelerated)
{
for (int row = 0; row < 8; row++)
{
ReadOnlySpan<ushort> sourceRow = source.DangerousGetRowSpan(sourceOrigin.Y + row)[sourceOrigin.X..];
ReadOnlySpan<ushort> predictionRow =
reconstruction.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X..];
// AV1's high-bit-depth domain tops out at 4095, so signed 16-bit subtraction preserves
// every possible sample difference before the square is widened to 32-bit lanes.
Vector128<short> difference =
(Vector128.LoadUnsafe(ref MemoryMarshal.GetReference(sourceRow)) -
Vector128.LoadUnsafe(ref MemoryMarshal.GetReference(predictionRow)))
.AsInt16();
Vector128<int> lower = Vector128.WidenLower(difference);
Vector128<int> upper = Vector128.WidenUpper(difference);
sum += Vector128.Sum(difference);
sumOfSquares += Vector128.Sum(lower * lower) + Vector128.Sum(upper * upper);
}
}
else
{
for (int row = 0; row < 8; row++)
{
ReadOnlySpan<ushort> sourceRow = source.DangerousGetRowSpan(sourceOrigin.Y + row)[sourceOrigin.X..];
ReadOnlySpan<ushort> predictionRow =
reconstruction.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X..];
for (int column = 0; column < 8; column++)
{
int difference = sourceRow[column] - predictionRow[column];
sum += difference;
sumOfSquares += difference * difference;
}
}
}
return GetNormalizedVariance(sum, sumOfSquares, bitDepth);
}
/// <inheritdoc/> /// <inheritdoc/>
public static void CopyPaletteSamples( public static void CopyPaletteSamples(
Buffer2DRegion<ushort> source, Buffer2DRegion<ushort> source,

13
src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.cs

@ -3,6 +3,7 @@
using System.Buffers; using System.Buffers;
using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy;
using SixLabors.ImageSharp.Formats.Heif.Av1.Motion;
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit;
using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling;
@ -109,7 +110,9 @@ internal sealed partial class Av1IntraTileWriter : IAv1TileWriter, IDisposable
int initialSize, int initialSize,
out int tileDataLength) out int tileDataLength)
where TSample : unmanaged where TSample : unmanaged
where TOperator : struct, Av1IntraSuperblockEncoder.IBlockEncodingOperator<TSample> where TOperator : struct,
Av1IntraSuperblockEncoder.IBlockEncodingOperator<TSample>,
Av1IntraBlockCopySearchIndex.ISearchOperation<TSample>
{ {
ObuFrameHeader frameHeader = picture.Parent.FrameHeader; ObuFrameHeader frameHeader = picture.Parent.FrameHeader;
ObuSequenceHeader sequenceHeader = picture.Sequence.SequenceHeader; ObuSequenceHeader sequenceHeader = picture.Sequence.SequenceHeader;
@ -136,6 +139,14 @@ internal sealed partial class Av1IntraTileWriter : IAv1TileWriter, IDisposable
int superblockModeInfoSize = sequenceHeader.SuperblockModeInfoSize; int superblockModeInfoSize = sequenceHeader.SuperblockModeInfoSize;
int superblockShift = sequenceHeader.SuperblockSizeLog2 - Av1Constants.ModeInfoSizeLog2; int superblockShift = sequenceHeader.SuperblockSizeLog2 - Av1Constants.ModeInfoSizeLog2;
if (frameHeader.AllowIntraBlockCopy)
{
// Hash the visible source once before reconstruction begins so candidate discovery never depends
// on coding order and the workspace can be reused as compact bucket links afterward.
picture.IntraBlockCopySearch.Initialize<TSample, TOperator>(
source.View.GetPlane(Av1Plane.Y));
}
for (int modeInfoRow = tile.ModeInfoRowStart; for (int modeInfoRow = tile.ModeInfoRowStart;
modeInfoRow < tile.ModeInfoRowEnd; modeInfoRow < tile.ModeInfoRowEnd;
modeInfoRow += superblockModeInfoSize) modeInfoRow += superblockModeInfoSize)

24
src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs

@ -3,6 +3,7 @@
using System.Buffers; using System.Buffers;
using System.Runtime.CompilerServices; using System.Runtime.CompilerServices;
using SixLabors.ImageSharp.Formats.Heif.Av1.Motion;
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit;
using SixLabors.ImageSharp.Memory; using SixLabors.ImageSharp.Memory;
@ -93,7 +94,16 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable
int displacementVectorStorageLength = checked( int displacementVectorStorageLength = checked(
displacementVectorLength * Unsafe.SizeOf<Av1EncoderDisplacementVector>()); displacementVectorLength * Unsafe.SizeOf<Av1EncoderDisplacementVector>());
int stateStorageLength = checked(displacementVectorStorageOffset + displacementVectorStorageLength); int displacementVectorStorageEnd = checked(displacementVectorStorageOffset + displacementVectorStorageLength);
int intraBlockCopySearchStorageOffset = frameHeader.AllowIntraBlockCopy
? Av1Math.AlignPowerOf2(displacementVectorStorageEnd, 2)
: displacementVectorStorageEnd;
int intraBlockCopySearchStorageLength = frameHeader.AllowIntraBlockCopy
? Av1IntraBlockCopySearchIndex.GetStorageLength(width, height)
: 0;
int stateStorageLength = checked(intraBlockCopySearchStorageOffset + intraBlockCopySearchStorageLength);
// Segmentation and every tile edge share one clean picture lifetime. The partition region begins at its // Segmentation and every tile edge share one clean picture lifetime. The partition region begins at its
// native alignment, while typed views keep the entropy writer independent from the packed byte owner. // native alignment, while typed views keep the entropy writer independent from the packed byte owner.
@ -138,6 +148,17 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable
displacementVectors = displacementVectorMemory.Memory; displacementVectors = displacementVectorMemory.Memory;
} }
Av1IntraBlockCopySearchIndex intraBlockCopySearch = default;
if (frameHeader.AllowIntraBlockCopy)
{
// The search index casts its packed workspace to 32-bit links, so its non-owning region begins at
// a four-byte boundary inside the existing picture-state rent.
intraBlockCopySearch = new Av1IntraBlockCopySearchIndex(
stateStorage.Slice(intraBlockCopySearchStorageOffset, intraBlockCopySearchStorageLength),
width,
height);
}
int[][] cdefPreset = new int[tileCount][]; int[][] cdefPreset = new int[tileCount][];
int[] previousQIndex = new int[tileCount]; int[] previousQIndex = new int[tileCount];
for (int tileIndex = 0; tileIndex < tileCount; tileIndex++) for (int tileIndex = 0; tileIndex < tileCount; tileIndex++)
@ -227,6 +248,7 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable
ModeInfoGrid = this.modeInfo.Grid, ModeInfoGrid = this.modeInfo.Grid,
ModeInfoAllocation = this.modeInfo.Allocation, ModeInfoAllocation = this.modeInfo.Allocation,
DisplacementVectors = displacementVectors, DisplacementVectors = displacementVectors,
IntraBlockCopySearch = intraBlockCopySearch,
ModeInfoStride = this.modeInfo.ModeInfoStride, ModeInfoStride = this.modeInfo.ModeInfoStride,
Disallow4x4AllFrames = this.modeInfo.Disallow4x4AllFrames, Disallow4x4AllFrames = this.modeInfo.Disallow4x4AllFrames,
CdefPreset = cdefPreset CdefPreset = cdefPreset

5
src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureControlSet.cs

@ -70,6 +70,11 @@ internal class Av1PictureControlSet
/// </summary> /// </summary>
public Memory<Av1EncoderDisplacementVector> DisplacementVectors { get; set; } public Memory<Av1EncoderDisplacementVector> DisplacementVectors { get; set; }
/// <summary>
/// Gets or sets the non-owning visible-frame hash index used by intra-block-copy motion search.
/// </summary>
public Av1IntraBlockCopySearchIndex IntraBlockCopySearch { get; set; }
/// <summary> /// <summary>
/// Gets or sets the row stride of <see cref="ModeInfoGrid"/> in 4x4 mode-information units. /// Gets or sets the row stride of <see cref="ModeInfoGrid"/> in 4x4 mode-information units.
/// </summary> /// </summary>

4
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderModeInfoBufferTests.cs

@ -50,7 +50,7 @@ public class Av1EncoderModeInfoBufferTests
[Theory] [Theory]
[InlineData(false, false, 336)] [InlineData(false, false, 336)]
[InlineData(true, false, 3_536)] [InlineData(true, false, 3_536)]
[InlineData(true, true, 4_560)] [InlineData(true, true, 6_416)]
public unsafe void PictureBufferPacksAllPictureStateIntoTwoAllocatorOwners( public unsafe void PictureBufferPacksAllPictureStateIntoTwoAllocatorOwners(
bool allowScreenContentTools, bool allowScreenContentTools,
bool allowIntraBlockCopy, bool allowIntraBlockCopy,
@ -148,6 +148,8 @@ public class Av1EncoderModeInfoBufferTests
{ {
Assert.Equal(256, picture.DisplacementVectors.Length); Assert.Equal(256, picture.DisplacementVectors.Length);
Assert.Equal(4, sizeof(Av1EncoderDisplacementVector)); Assert.Equal(4, sizeof(Av1EncoderDisplacementVector));
Assert.Equal(9, picture.IntraBlockCopySearch.OriginWidth);
Assert.Equal(9, picture.IntraBlockCopySearch.OriginHeight);
// The packed vector region starts at its natural 16-bit alignment inside the shared byte owner. // The packed vector region starts at its natural 16-bit alignment inside the shared byte owner.
fixed (Av1EncoderDisplacementVector* pointer = picture.DisplacementVectors.Span) fixed (Av1EncoderDisplacementVector* pointer = picture.DisplacementVectors.Span)

12
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs

@ -673,6 +673,18 @@ public class Av1EntropyTests
long expected) long expected)
=> Assert.Equal(expected, Av1RateDistortion.GetCost(rateMultiplier, rate, distortion)); => Assert.Equal(expected, Av1RateDistortion.GetCost(rateMultiplier, rate, distortion));
[Theory]
[InlineData(1, 8191, 100, 100)]
[InlineData(1, 8192, 100, 101)]
[InlineData(128, 4096, 100, 101)]
[InlineData(512, 3072, 100, 102)]
public void MotionSearchCostMatchesCurrentLibaom(
int rateMultiplier,
int motionVectorRate,
int variance,
int expected)
=> Assert.Equal(expected, Av1RateDistortion.GetMotionSearchCost(rateMultiplier, motionVectorRate, variance));
[Theory] [Theory]
[InlineData(0, 0, 52)] [InlineData(0, 0, 52)]
[InlineData(0, 1, 3)] [InlineData(0, 1, 3)]

134
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs

@ -5,6 +5,7 @@ using SixLabors.ImageSharp.Formats.Heif.Av1;
using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy;
using SixLabors.ImageSharp.Formats.Heif.Av1.Motion; using SixLabors.ImageSharp.Formats.Heif.Av1.Motion;
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit;
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling;
using SixLabors.ImageSharp.Memory; using SixLabors.ImageSharp.Memory;
@ -231,6 +232,139 @@ public class Av1IntraBlockCopyTests
Assert.False(Av1IntraBlockCopy.IsValid(new Av1MotionVector(512, -2560), ref partitionInfo, tileInfo, sequenceHeader)); Assert.False(Av1IntraBlockCopy.IsValid(new Av1MotionVector(512, -2560), ref partitionInfo, tileInfo, sequenceHeader));
} }
/// <summary>
/// Verifies exact hash matches at unaligned origins in both normative search regions.
/// </summary>
[Fact]
public void SearchIndexFindsUnalignedAboveAndLeftMatches()
{
const int Width = 640;
const int Height = 256;
const int QIndex = 23;
Point blockOrigin = new(512, 128);
Point aboveOrigin = new(515, 57);
Point leftOrigin = new(191, 131);
ObuSequenceHeader sequenceHeader = CreateSequenceHeader();
ObuFrameHeader frameHeader = CreateFrameHeader();
frameHeader.AllowScreenContentTools = true;
frameHeader.AllowIntraBlockCopy = true;
using Av1EncoderPictureBuffer pictureBuffer = new(
Configuration.Default,
sequenceHeader,
frameHeader,
Width,
Height);
using Av1EncoderFrameBuffer<byte> source = new(
Configuration.Default,
Width,
Height,
8,
Av1ColorFormat.Yuv400,
0,
0);
using Av1EncoderFrameBuffer<byte> reconstruction = new(
Configuration.Default,
Width,
Height,
8,
Av1ColorFormat.Yuv400,
0,
0);
Buffer2DRegion<byte> sourceLuma = source.Frame.View.GetPlane(Av1Plane.Y);
Buffer2DRegion<byte> reconstructionLuma = reconstruction.Frame.View.GetPlane(Av1Plane.Y);
uint randomState = 0x8F3A21C5;
for (int row = 0; row < Height; row++)
{
Span<byte> sourceRow = sourceLuma.DangerousGetRowSpan(row);
reconstructionLuma.DangerousGetRowSpan(row).Clear();
for (int column = 0; column < Width; column++)
{
randomState = unchecked((randomState * 1_664_525) + 1_013_904_223);
sourceRow[column] = (byte)(randomState >> 24);
}
}
for (int row = 0; row < 8; row++)
{
ReadOnlySpan<byte> blockRow = sourceLuma.DangerousGetRowSpan(blockOrigin.Y + row).Slice(blockOrigin.X, 8);
blockRow.CopyTo(sourceLuma.DangerousGetRowSpan(aboveOrigin.Y + row)[aboveOrigin.X..]);
blockRow.CopyTo(sourceLuma.DangerousGetRowSpan(leftOrigin.Y + row)[leftOrigin.X..]);
blockRow.CopyTo(reconstructionLuma.DangerousGetRowSpan(aboveOrigin.Y + row)[aboveOrigin.X..]);
blockRow.CopyTo(reconstructionLuma.DangerousGetRowSpan(leftOrigin.Y + row)[leftOrigin.X..]);
}
Av1PictureControlSet picture = pictureBuffer.Picture;
picture.IntraBlockCopySearch.Initialize<byte, Av1IntraSuperblockEncoder.ByteOperator>(sourceLuma);
using Av1SymbolEncoder writer = new(Configuration.Default, 64, QIndex);
Span<Av1MotionVector> candidates = stackalloc Av1MotionVector[2];
Av1MotionVector reference = new(0, -2560);
int candidateCount = picture.IntraBlockCopySearch.FindCandidates<byte, Av1IntraSuperblockEncoder.ByteOperator>(
source.Frame.CodedView.GetPlane(Av1Plane.Y),
reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y),
blockOrigin,
new Av1TileInfo(0, 0, frameHeader),
sequenceHeader,
writer,
reference,
Av1RateDistortion.GetKeyFrameRateMultiplier(QIndex, Av1BitDepth.EightBit),
candidates);
Assert.Equal(2, candidateCount);
Assert.Equal(new Av1MotionVector(-568, 24), candidates[0]);
Assert.Equal(new Av1MotionVector(24, -2568), candidates[1]);
}
/// <summary>
/// Verifies high-bit-depth SIMD variance normalization against the eight-bit search domain.
/// </summary>
[Fact]
public void SearchVarianceMatchesTwelveBitReference()
{
using Av1EncoderFrameBuffer<ushort> source = new(
Configuration.Default,
8,
8,
12,
Av1ColorFormat.Yuv400,
0,
0);
using Av1EncoderFrameBuffer<ushort> reconstruction = new(
Configuration.Default,
8,
8,
12,
Av1ColorFormat.Yuv400,
0,
0);
Buffer2DRegion<ushort> sourceLuma = source.Frame.View.GetPlane(Av1Plane.Y);
Buffer2DRegion<ushort> reconstructionLuma = reconstruction.Frame.View.GetPlane(Av1Plane.Y);
for (int row = 0; row < 8; row++)
{
Span<ushort> sourceRow = sourceLuma.DangerousGetRowSpan(row);
Span<ushort> reconstructionRow = reconstructionLuma.DangerousGetRowSpan(row);
for (int column = 0; column < 8; column++)
{
sourceRow[column] = 1000;
reconstructionRow[column] = (ushort)(1000 + (((row * 8) + column) % 2 == 0 ? 17 : 33));
}
}
int actual = Av1IntraSuperblockEncoder.UInt16Operator.GetVariance(
sourceLuma,
Point.Empty,
reconstructionLuma,
Point.Empty,
Av1BitDepth.TwelveBit);
Assert.Equal(16, actual);
}
/// <summary> /// <summary>
/// Creates the 640-by-256, 4:2:0 sequence geometry shared by the displacement tests. /// Creates the 640-by-256, 4:2:0 sequence geometry shared by the displacement tests.
/// </summary> /// </summary>

Loading…
Cancel
Save