diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md
index a9d0b63981..b5f68d1dcd 100644
--- a/HEIF_IMPLEMENTATION_PLAN.md
+++ b/HEIF_IMPLEMENTATION_PLAN.md
@@ -859,7 +859,7 @@ Encoder verification contract:
- [~] Paired chroma palette clustering now preserves current libaom's squared two-component distance, first-centroid tie order, independently rounded U/V means, paired deterministic empty-cluster replacement, preceding-state retention on increased distortion, and 50-iteration limit. Keeping the source planes separate avoids interleave/deinterleave copies and improves on libaom's AVX2 ceiling with Vector512, Vector256, Vector128, then scalar dispatch through ImageSharp's shared vector-count helpers. Three independent tests cover exact paired convergence, midpoint initialization, 12-bit distance and index parity, untouched destination bounds, and every intrinsic tier. The exact Release test-project build reports 1,992 baseline warnings and zero errors; the focused three-case set, complete 8,934-case AVIF set, and complete 230-case HEIF set pass direct foreground net11 Release VSTest. Roslynk reports zero compiler errors and no touched-file analyzer warnings. Candidate integration and production activation remain in the open chroma-palette checkpoint.
- [~] Live paired chroma palette selection now follows current libaom's complete 2-through-8 color-size search, U-plane neighbor-cache snapping, stable U-ordered color pairs, shared U/V index map, implicit DCT-DCT transform, and strict rate-distortion winner replacement. It improves on speed-configured libaom by applying no early header-cost pruning, keeps planar U/V source data separate, and reuses the SIMD-first prediction, residual, transform, quantization, and reconstruction operators without allocator-backed candidate storage. The production tile regression proves both palette-mode probability branches, exact paired colors and indices, coefficient-free reconstruction, and nonempty syntax. The complete 58-case intra-superblock set, 8,935-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains the next checkpoint.
- [~] Production palette activation now matches current libaom's default good-quality screen detector: it scans only complete 16x16 luma blocks, normalizes high-bit-depth samples to eight bits, admits 2-through-4-color blocks, and uses the reference's strict greater-than-ten-percent frame-area threshold. A 256-bit stack bitset and a fifth-color early exit replace libaom's larger per-block histogram without changing the decision, allocation, or source precision. The adaptive sequence flag remains enabled, the frame flag is set before picture-state allocation, and intra-block copy remains disabled. Focused regressions prove strict-threshold equality, high-bit-depth normalization, five-color rejection, emitted frame-header activation, production decode, and generated payload retention. The exact Release test-project build reports 1,992 baseline warnings and zero errors; all 8,935 AVIF cases and all 230 HEIF cases pass direct foreground net11 Release VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 30 regenerated production payloads, including the 54-byte palette case. Roslynk reports zero compiler errors and no touched-file analyzer warnings.
-- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The exact net11 Release build reports 1,992 baseline warnings and zero errors; all 1,968 focused entropy, ownership, and intra-block-copy cases, all 9,023 AV1 cases, and all 206 non-AV1 HEIF cases pass through direct foreground VSTest, and Roslynk reports zero compiler errors. Pixel-search fallback, joint luma/chroma rate-distortion selection, production activation, and adaptive frame-flag clearing remain before intra-block copy can be enabled.
+- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. The exact net11 Release build reports 1,992 baseline warnings and zero errors; all 1,969 focused entropy and intra-block-copy cases, all 9,068 AV1 and AVIF cases, and all 189 non-AV1 HEIF cases pass through direct foreground VSTest, and Roslynk reports zero compiler errors with no touched-file analyzer diagnostics. The exhaustive mesh fallback, joint luma/chroma rate-distortion selection, production activation, and adaptive frame-flag clearing remain before intra-block copy can be enabled.
- [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips.
- [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips.
- [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips.
diff --git a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1RateDistortion.cs b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1RateDistortion.cs
index fc0d8708d8..612a2597a6 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1RateDistortion.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1RateDistortion.cs
@@ -65,4 +65,38 @@ internal static class Av1RateDistortion
int motionError = (int)((weightedRate + (1 << (MotionErrorShift - 1))) >> MotionErrorShift);
return variance + motionError;
}
+
+ ///
+ /// Gets the sum-of-absolute-differences rate scale for a frame quantizer.
+ ///
+ /// The segment quantizer index.
+ /// The coded sample bit depth.
+ /// The multiplier that converts motion-vector rate into the absolute-difference domain.
+ public static int GetMotionSearchSadPerBit(int qIndex, Av1BitDepth bitDepth)
+ {
+ int quantizerDivisor = 1 << (bitDepth.GetBitCount() - 6);
+ double quantizer = Av1QuantizationLookup.GetAcQuant(qIndex, 0, bitDepth) / (double)quantizerDivisor;
+ return (int)((0.0418 * quantizer) + 2.4107);
+ }
+
+ ///
+ /// Gets the sum-of-absolute-differences cost of a full-pixel motion candidate.
+ ///
+ /// The quantizer-derived motion-rate scale.
+ /// The motion-vector syntax rate in 1/512-bit units.
+ /// The unnormalized sample-domain absolute difference.
+ /// The absolute difference plus the motion-vector search cost.
+ public static int GetMotionSearchSadCost(
+ int sadPerBit,
+ int motionVectorRate,
+ int sumOfAbsoluteDifferences)
+ {
+ const int MotionRateShift = 9;
+
+ // Full-pixel traversal uses absolute differences, so its quantizer-derived rate scale is deliberately
+ // distinct from the variance-domain error-per-bit scale used to compare the resulting search paths.
+ long weightedRate = (long)motionVectorRate * sadPerBit;
+ int motionError = (int)((weightedRate + (1 << (MotionRateShift - 1))) >> MotionRateShift);
+ return sumOfAbsoluteDifferences + motionError;
+ }
}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Motion/Av1IntraBlockCopySearchIndex.cs b/src/ImageSharp/Formats/Heif/Av1/Motion/Av1IntraBlockCopySearchIndex.cs
index 761549c696..98dfd5083c 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Motion/Av1IntraBlockCopySearchIndex.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Motion/Av1IntraBlockCopySearchIndex.cs
@@ -17,6 +17,9 @@ internal readonly struct Av1IntraBlockCopySearchIndex
private const int BlockSize = 8;
private const int MaximumBucketCount = 1 << 16;
private const int MaximumCandidatesPerBucket = 256;
+ private const int MaximumFullPixelSearchOffset = (1 << 10) - 1;
+ private const int MinimumFullPixelMotionVector = -(1 << 11) + 1;
+ private const int MaximumFullPixelMotionVector = (1 << 11) - 1;
private const uint HorizontalHashMultiplier = 257;
private const uint VerticalHashMultiplier = 65599;
private static readonly uint HorizontalLeadingWeight = GetLeadingWeight(HorizontalHashMultiplier);
@@ -72,6 +75,20 @@ internal readonly struct Av1IntraBlockCopySearchIndex
/// when every sample is equal.
public static abstract bool BlocksEqual(Buffer2DRegion plane, Point first, Point second);
+ ///
+ /// Gets the sum of absolute differences between the source block and reconstructed predictor.
+ ///
+ /// The coded source plane.
+ /// The source block origin.
+ /// The reconstructed luma plane.
+ /// The predictor block origin.
+ /// The unnormalized absolute difference over the complete 8x8 block.
+ public static abstract int GetSumOfAbsoluteDifferences(
+ Buffer2DRegion source,
+ Point sourceOrigin,
+ Buffer2DRegion reconstruction,
+ Point predictionOrigin);
+
///
/// Gets the normalized 8x8 variance between a source block and reconstructed predictor.
///
@@ -99,6 +116,16 @@ internal readonly struct Av1IntraBlockCopySearchIndex
///
public int OriginHeight { get; }
+ ///
+ /// Gets the expanding NSTEP radii from the final one-pixel refinement through the largest search step.
+ ///
+ private static ReadOnlySpan SearchRadii => [1, 2, 3, 5, 8, 12, 18, 27, 41, 62, 93, 140, 210, 210, 210];
+
+ ///
+ /// Gets the tangential radius paired with each NSTEP primary radius.
+ ///
+ private static ReadOnlySpan SearchTangentialRadii => [1, 2, 3, 5, 3, 4, 7, 11, 16, 25, 38, 57, 86, 86, 86];
+
///
/// Gets the packed storage length required for a visible frame.
///
@@ -322,6 +349,96 @@ internal readonly struct Av1IntraBlockCopySearchIndex
return candidateCount;
}
+ ///
+ /// Finds the best full-pixel NSTEP candidate in the reference above and left search regions.
+ ///
+ /// The native unsigned sample storage type.
+ /// The closed sample operation.
+ /// The coded source luma plane.
+ /// The coded reconstructed luma plane.
+ /// The current 8x8 block origin.
+ /// The active tile boundaries.
+ /// The sequence geometry and sample precision.
+ /// The live tile entropy model used for displacement rate.
+ /// The spatial displacement-vector reference.
+ /// The effective segment quantizer index.
+ /// The active rate-distortion multiplier.
+ /// Storage receiving the above candidate followed by the left candidate.
+ /// The number of candidates written.
+ public int FindPixelCandidates(
+ Buffer2DRegion source,
+ Buffer2DRegion reconstruction,
+ Point blockOrigin,
+ Av1TileInfo tile,
+ ObuSequenceHeader sequenceHeader,
+ Av1SymbolEncoder writer,
+ Av1MotionVector reference,
+ int qIndex,
+ int rateMultiplier,
+ Span candidates)
+ where TSample : unmanaged
+ where TOperation : struct, ISearchOperation
+ {
+ if (this.OriginWidth == 0 || this.OriginHeight == 0)
+ {
+ return 0;
+ }
+
+ const int ModeInfoSampleSize = 1 << Av1Constants.ModeInfoSizeLog2;
+ int tileLeft = tile.ModeInfoColumnStart * ModeInfoSampleSize;
+ int tileTop = tile.ModeInfoRowStart * ModeInfoSampleSize;
+ int tileRight = Math.Min((tile.ModeInfoColumnEnd * ModeInfoSampleSize) - BlockSize, this.OriginWidth - 1);
+ int tileBottom = Math.Min((tile.ModeInfoRowEnd * ModeInfoSampleSize) - BlockSize, this.OriginHeight - 1);
+ int superblockSize = sequenceHeader.SuperblockSize.GetWidth();
+ int superblockLeft = (blockOrigin.X / superblockSize) * superblockSize;
+ int superblockTop = (blockOrigin.Y / superblockSize) * superblockSize;
+ int searchStepParameter = GetSearchStepParameter(Math.Max(this.OriginWidth + BlockSize - 1, this.OriginHeight + BlockSize - 1));
+ int sadPerBit = Av1RateDistortion.GetMotionSearchSadPerBit(qIndex, sequenceHeader.ColorConfig.BitDepth);
+ int candidateCount = 0;
+
+ if (TryFindPixelCandidate(
+ source,
+ reconstruction,
+ blockOrigin,
+ tile,
+ sequenceHeader,
+ writer,
+ reference,
+ rateMultiplier,
+ sadPerBit,
+ searchStepParameter,
+ tileLeft,
+ tileTop,
+ tileRight,
+ Math.Min(superblockTop - BlockSize, tileBottom),
+ out Av1MotionVector above))
+ {
+ candidates[candidateCount++] = above;
+ }
+
+ if (TryFindPixelCandidate(
+ source,
+ reconstruction,
+ blockOrigin,
+ tile,
+ sequenceHeader,
+ writer,
+ reference,
+ rateMultiplier,
+ sadPerBit,
+ searchStepParameter,
+ tileLeft,
+ tileTop,
+ Math.Min(superblockLeft - BlockSize, tileRight),
+ Math.Min(superblockTop + superblockSize - BlockSize, tileBottom),
+ out Av1MotionVector left))
+ {
+ candidates[candidateCount++] = left;
+ }
+
+ return candidateCount;
+ }
+
private static uint GetLeadingWeight(uint multiplier)
{
uint result = 1;
@@ -451,6 +568,348 @@ internal readonly struct Av1IntraBlockCopySearchIndex
return found;
}
+ private static bool TryFindPixelCandidate(
+ Buffer2DRegion source,
+ Buffer2DRegion reconstruction,
+ Point blockOrigin,
+ Av1TileInfo tile,
+ ObuSequenceHeader sequenceHeader,
+ Av1SymbolEncoder writer,
+ Av1MotionVector reference,
+ int rateMultiplier,
+ int sadPerBit,
+ int searchStepParameter,
+ int minimumColumn,
+ int minimumRow,
+ int maximumColumn,
+ int maximumRow,
+ out Av1MotionVector bestVector)
+ where TSample : unmanaged
+ where TOperation : struct, ISearchOperation
+ {
+ bestVector = default;
+ if (maximumColumn < minimumColumn || maximumRow < minimumRow)
+ {
+ return false;
+ }
+
+ int referenceColumn = reference.Column >> 3;
+ int referenceRow = reference.Row >> 3;
+ int minimumColumnOffset = Math.Max(
+ Math.Max(minimumColumn - blockOrigin.X, referenceColumn - MaximumFullPixelSearchOffset),
+ MinimumFullPixelMotionVector);
+
+ int maximumColumnOffset = Math.Min(
+ Math.Min(maximumColumn - blockOrigin.X, referenceColumn + MaximumFullPixelSearchOffset),
+ MaximumFullPixelMotionVector);
+
+ int minimumRowOffset = Math.Max(
+ Math.Max(minimumRow - blockOrigin.Y, referenceRow - MaximumFullPixelSearchOffset),
+ MinimumFullPixelMotionVector);
+
+ int maximumRowOffset = Math.Min(
+ Math.Min(maximumRow - blockOrigin.Y, referenceRow + MaximumFullPixelSearchOffset),
+ MaximumFullPixelMotionVector);
+
+ if (maximumColumnOffset < minimumColumnOffset || maximumRowOffset < minimumRowOffset)
+ {
+ return false;
+ }
+
+ Point start = new(
+ Av1Math.Clamp(referenceColumn, minimumColumnOffset, maximumColumnOffset),
+ Av1Math.Clamp(referenceRow, minimumRowOffset, maximumRowOffset));
+
+ Point best = FindBestPixelCandidate(
+ source,
+ reconstruction,
+ blockOrigin,
+ writer,
+ reference,
+ sequenceHeader.ColorConfig.BitDepth,
+ rateMultiplier,
+ sadPerBit,
+ searchStepParameter,
+ minimumColumnOffset,
+ minimumRowOffset,
+ maximumColumnOffset,
+ maximumRowOffset,
+ start);
+
+ bestVector = new(best.Y * 8, best.X * 8);
+ Point modeInfoPosition = new(
+ blockOrigin.X >> Av1Constants.ModeInfoSizeLog2,
+ blockOrigin.Y >> Av1Constants.ModeInfoSizeLog2);
+
+ return Av1IntraBlockCopy.IsValid(
+ bestVector,
+ modeInfoPosition,
+ Av1BlockSize.Block8x8,
+ isChroma: false,
+ tile,
+ sequenceHeader);
+ }
+
+ private static Point FindBestPixelCandidate(
+ Buffer2DRegion source,
+ Buffer2DRegion reconstruction,
+ Point blockOrigin,
+ Av1SymbolEncoder writer,
+ Av1MotionVector reference,
+ Av1BitDepth bitDepth,
+ int rateMultiplier,
+ int sadPerBit,
+ int searchStepParameter,
+ int minimumColumnOffset,
+ int minimumRowOffset,
+ int maximumColumnOffset,
+ int maximumRowOffset,
+ Point start)
+ where TSample : unmanaged
+ where TOperation : struct, ISearchOperation
+ {
+ SearchDiamond(
+ source,
+ reconstruction,
+ blockOrigin,
+ writer,
+ reference,
+ sadPerBit,
+ searchStepParameter,
+ minimumColumnOffset,
+ minimumRowOffset,
+ maximumColumnOffset,
+ maximumRowOffset,
+ start,
+ out Point best,
+ out int centerSteps);
+
+ int bestCost = GetVarianceCost(
+ source,
+ reconstruction,
+ blockOrigin,
+ writer,
+ reference,
+ bitDepth,
+ rateMultiplier,
+ best);
+
+ int furtherSteps = SearchRadii.Length - 1 - searchStepParameter;
+ int shortenedBy = centerSteps;
+ while (shortenedBy < furtherSteps)
+ {
+ shortenedBy++;
+ SearchDiamond(
+ source,
+ reconstruction,
+ blockOrigin,
+ writer,
+ reference,
+ sadPerBit,
+ searchStepParameter + shortenedBy,
+ minimumColumnOffset,
+ minimumRowOffset,
+ maximumColumnOffset,
+ maximumRowOffset,
+ start,
+ out Point candidate,
+ out int additionalCenterSteps);
+
+ int candidateCost = GetVarianceCost(
+ source,
+ reconstruction,
+ blockOrigin,
+ writer,
+ reference,
+ bitDepth,
+ rateMultiplier,
+ candidate);
+
+ if (candidateCost < bestCost)
+ {
+ bestCost = candidateCost;
+ best = candidate;
+ }
+
+ shortenedBy += additionalCenterSteps;
+ }
+
+ return best;
+ }
+
+ private static void SearchDiamond(
+ Buffer2DRegion source,
+ Buffer2DRegion reconstruction,
+ Point blockOrigin,
+ Av1SymbolEncoder writer,
+ Av1MotionVector reference,
+ int sadPerBit,
+ int searchStepParameter,
+ int minimumColumnOffset,
+ int minimumRowOffset,
+ int maximumColumnOffset,
+ int maximumRowOffset,
+ Point start,
+ out Point best,
+ out int centerSteps)
+ where TSample : unmanaged
+ where TOperation : struct, ISearchOperation
+ {
+ best = start;
+ centerSteps = 0;
+ bool movedFromStart = false;
+ int bestCost = GetSadCost(
+ source,
+ reconstruction,
+ blockOrigin,
+ writer,
+ reference,
+ sadPerBit,
+ best);
+
+ for (int stage = SearchRadii.Length - 1 - searchStepParameter; stage >= 0; stage--)
+ {
+ int radius = SearchRadii[stage];
+ int tangentialRadius = SearchTangentialRadii[stage];
+ int searchSiteCount = radius <= 5 ? 8 : 12;
+ int bestSite = 0;
+ for (int site = 1; site <= searchSiteCount; site++)
+ {
+ Point delta = GetSearchOffset(site, radius, tangentialRadius);
+ Point candidate = new(best.X + delta.X, best.Y + delta.Y);
+ if (candidate.X < minimumColumnOffset || candidate.X > maximumColumnOffset ||
+ candidate.Y < minimumRowOffset || candidate.Y > maximumRowOffset)
+ {
+ continue;
+ }
+
+ Point predictionOrigin = new(blockOrigin.X + candidate.X, blockOrigin.Y + candidate.Y);
+ int sumOfAbsoluteDifferences = TOperation.GetSumOfAbsoluteDifferences(
+ source,
+ blockOrigin,
+ reconstruction,
+ predictionOrigin);
+
+ // Motion-vector cost is nonnegative, so a raw absolute difference that already reaches the
+ // best combined cost cannot win and does not need an entropy-rate lookup.
+ if (sumOfAbsoluteDifferences >= bestCost)
+ {
+ continue;
+ }
+
+ Av1MotionVector vector = new(candidate.Y * 8, candidate.X * 8);
+ int rate = writer.GetDisplacementVectorSearchCost(vector, reference);
+ int candidateCost = Av1RateDistortion.GetMotionSearchSadCost(
+ sadPerBit,
+ rate,
+ sumOfAbsoluteDifferences);
+
+ if (candidateCost < bestCost)
+ {
+ bestCost = candidateCost;
+ bestSite = site;
+ }
+ }
+
+ if (bestSite != 0)
+ {
+ Point delta = GetSearchOffset(bestSite, radius, tangentialRadius);
+ best = new(best.X + delta.X, best.Y + delta.Y);
+ movedFromStart = true;
+ }
+
+ if (!movedFromStart)
+ {
+ centerSteps++;
+ }
+
+ // The three largest NSTEP stages intentionally share one radius. When a stage remains centered,
+ // consume the equivalent duplicates exactly once instead of repeating the same candidate positions.
+ while (bestSite == 0 && stage > 2 && SearchRadii[stage - 1] == SearchRadii[stage])
+ {
+ centerSteps++;
+ stage--;
+ }
+ }
+ }
+
+ private static int GetSadCost(
+ Buffer2DRegion source,
+ Buffer2DRegion reconstruction,
+ Point blockOrigin,
+ Av1SymbolEncoder writer,
+ Av1MotionVector reference,
+ int sadPerBit,
+ Point candidate)
+ where TSample : unmanaged
+ where TOperation : struct, ISearchOperation
+ {
+ Point predictionOrigin = new(blockOrigin.X + candidate.X, blockOrigin.Y + candidate.Y);
+ int sumOfAbsoluteDifferences = TOperation.GetSumOfAbsoluteDifferences(
+ source,
+ blockOrigin,
+ reconstruction,
+ predictionOrigin);
+
+ Av1MotionVector vector = new(candidate.Y * 8, candidate.X * 8);
+ int rate = writer.GetDisplacementVectorSearchCost(vector, reference);
+ return Av1RateDistortion.GetMotionSearchSadCost(sadPerBit, rate, sumOfAbsoluteDifferences);
+ }
+
+ private static int GetVarianceCost(
+ Buffer2DRegion source,
+ Buffer2DRegion reconstruction,
+ Point blockOrigin,
+ Av1SymbolEncoder writer,
+ Av1MotionVector reference,
+ Av1BitDepth bitDepth,
+ int rateMultiplier,
+ Point candidate)
+ where TSample : unmanaged
+ where TOperation : struct, ISearchOperation
+ {
+ Point predictionOrigin = new(blockOrigin.X + candidate.X, blockOrigin.Y + candidate.Y);
+ int variance = TOperation.GetVariance(
+ source,
+ blockOrigin,
+ reconstruction,
+ predictionOrigin,
+ bitDepth);
+
+ Av1MotionVector vector = new(candidate.Y * 8, candidate.X * 8);
+ int rate = writer.GetDisplacementVectorSearchCost(vector, reference);
+ return Av1RateDistortion.GetMotionSearchCost(rateMultiplier, rate, variance);
+ }
+
+ private static int GetSearchStepParameter(int frameSize)
+ {
+ int size = Math.Max(frameSize, 16);
+ int searchStepParameter = 0;
+ while (((long)size << searchStepParameter) < MaximumFullPixelSearchOffset)
+ {
+ searchStepParameter++;
+ }
+
+ return Math.Min(searchStepParameter, 9);
+ }
+
+ private static Point GetSearchOffset(int site, int radius, int tangentialRadius)
+ => site switch
+ {
+ 1 => new Point(0, -radius),
+ 2 => new Point(0, radius),
+ 3 => new Point(-radius, 0),
+ 4 => new Point(radius, 0),
+ 5 => new Point(-tangentialRadius, -radius),
+ 6 => new Point(tangentialRadius, radius),
+ 7 => new Point(radius, -tangentialRadius),
+ 8 => new Point(-radius, tangentialRadius),
+ 9 => new Point(tangentialRadius, -radius),
+ 10 => new Point(-tangentialRadius, radius),
+ 11 => new Point(radius, tangentialRadius),
+ _ => new Point(-radius, -tangentialRadius)
+ };
+
private Span GetHashesAndLinks()
=> MemoryMarshal.Cast(this.storage.Span[..this.headOffset]);
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs
index cfdf338783..91c714e879 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs
@@ -21,7 +21,7 @@ internal static partial class Av1IntraSuperblockEncoder
/// Defines type-specific block encoding without coupling traversal to sample storage width.
///
/// The native unsigned sample storage type.
- internal interface IBlockEncodingOperator
+ internal interface IBlockEncodingOperator : Av1IntraBlockCopySearchIndex.ISearchOperation
where TSample : unmanaged
{
///
@@ -307,9 +307,7 @@ internal static partial class Av1IntraSuperblockEncoder
///
/// Encodes blocks stored as eight-bit samples.
///
- internal readonly struct ByteOperator :
- IBlockEncodingOperator,
- Av1IntraBlockCopySearchIndex.ISearchOperation
+ internal readonly struct ByteOperator : IBlockEncodingOperator
{
///
public static Span GetLeftReference(Span residual, int length)
@@ -340,6 +338,50 @@ internal static partial class Av1IntraSuperblockEncoder
return true;
}
+ ///
+ public static int GetSumOfAbsoluteDifferences(
+ Buffer2DRegion source,
+ Point sourceOrigin,
+ Buffer2DRegion reconstruction,
+ Point predictionOrigin)
+ {
+ int sum = 0;
+ if (Vector128.IsHardwareAccelerated)
+ {
+ for (int row = 0; row < 8; row++)
+ {
+ ReadOnlySpan sourceRow = source.DangerousGetRowSpan(sourceOrigin.Y + row)[sourceOrigin.X..];
+ ReadOnlySpan predictionRow =
+ reconstruction.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X..];
+
+ Vector128 difference =
+ (Vector128.WidenLower(Vector128.CreateScalarUnsafe(MemoryMarshal.Read(sourceRow)).AsByte()) -
+ Vector128.WidenLower(Vector128.CreateScalarUnsafe(MemoryMarshal.Read(predictionRow)).AsByte()))
+ .AsInt16();
+
+ // Widened signed differences retain both subtraction directions; absolute values then reduce
+ // the complete eight-sample row without scalar extraction or per-sample branches.
+ sum += Vector128.Sum(Vector128.Abs(difference));
+ }
+ }
+ else
+ {
+ for (int row = 0; row < 8; row++)
+ {
+ ReadOnlySpan sourceRow = source.DangerousGetRowSpan(sourceOrigin.Y + row)[sourceOrigin.X..];
+ ReadOnlySpan predictionRow =
+ reconstruction.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X..];
+
+ for (int column = 0; column < 8; column++)
+ {
+ sum += Math.Abs(sourceRow[column] - predictionRow[column]);
+ }
+ }
+ }
+
+ return sum;
+ }
+
///
public static int GetVariance(
Buffer2DRegion source,
@@ -675,9 +717,7 @@ internal static partial class Av1IntraSuperblockEncoder
///
/// Encodes blocks stored as high-bit-depth samples.
///
- internal readonly struct UInt16Operator :
- IBlockEncodingOperator,
- Av1IntraBlockCopySearchIndex.ISearchOperation
+ internal readonly struct UInt16Operator : IBlockEncodingOperator
{
///
public static Span GetLeftReference(Span residual, int length)
@@ -714,6 +754,50 @@ internal static partial class Av1IntraSuperblockEncoder
return true;
}
+ ///
+ public static int GetSumOfAbsoluteDifferences(
+ Buffer2DRegion source,
+ Point sourceOrigin,
+ Buffer2DRegion reconstruction,
+ Point predictionOrigin)
+ {
+ int sum = 0;
+ if (Vector128.IsHardwareAccelerated)
+ {
+ for (int row = 0; row < 8; row++)
+ {
+ ReadOnlySpan sourceRow = source.DangerousGetRowSpan(sourceOrigin.Y + row)[sourceOrigin.X..];
+ ReadOnlySpan predictionRow =
+ reconstruction.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X..];
+
+ // Twelve-bit samples remain within signed 16-bit subtraction and absolute-value ranges,
+ // allowing all eight row differences to stay packed until their horizontal reduction.
+ Vector128 difference =
+ (Vector128.LoadUnsafe(ref MemoryMarshal.GetReference(sourceRow)) -
+ Vector128.LoadUnsafe(ref MemoryMarshal.GetReference(predictionRow)))
+ .AsInt16();
+
+ sum += Vector128.Sum(Vector128.Abs(difference));
+ }
+ }
+ else
+ {
+ for (int row = 0; row < 8; row++)
+ {
+ ReadOnlySpan sourceRow = source.DangerousGetRowSpan(sourceOrigin.Y + row)[sourceOrigin.X..];
+ ReadOnlySpan predictionRow =
+ reconstruction.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X..];
+
+ for (int column = 0; column < 8; column++)
+ {
+ sum += Math.Abs(sourceRow[column] - predictionRow[column]);
+ }
+ }
+ }
+
+ return sum;
+ }
+
///
public static int GetVariance(
Buffer2DRegion source,
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.cs
index 5d1d58c17b..249eacbdfa 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraTileWriter.cs
@@ -110,9 +110,7 @@ internal sealed partial class Av1IntraTileWriter : IAv1TileWriter, IDisposable
int initialSize,
out int tileDataLength)
where TSample : unmanaged
- where TOperator : struct,
- Av1IntraSuperblockEncoder.IBlockEncodingOperator,
- Av1IntraBlockCopySearchIndex.ISearchOperation
+ where TOperator : struct, Av1IntraSuperblockEncoder.IBlockEncodingOperator
{
ObuFrameHeader frameHeader = picture.Parent.FrameHeader;
ObuSequenceHeader sequenceHeader = picture.Sequence.SequenceHeader;
diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs
index ebc8925f06..88cd147d40 100644
--- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs
+++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EntropyTests.cs
@@ -685,6 +685,27 @@ public class Av1EntropyTests
int expected)
=> Assert.Equal(expected, Av1RateDistortion.GetMotionSearchCost(rateMultiplier, motionVectorRate, variance));
+ [Theory]
+ [InlineData(0, 0, 2)]
+ [InlineData(255, 0, 21)]
+ [InlineData(255, 1, 21)]
+ [InlineData(255, 2, 21)]
+ public void MotionSearchSadPerBitMatchesCurrentLibaom(int qIndex, int bitDepth, int expected)
+ => Assert.Equal(expected, Av1RateDistortion.GetMotionSearchSadPerBit(qIndex, (Av1BitDepth)bitDepth));
+
+ [Theory]
+ [InlineData(2, 255, 100, 101)]
+ [InlineData(21, 512, 100, 121)]
+ [InlineData(21, 1000, 100, 141)]
+ public void MotionSearchSadCostMatchesCurrentLibaom(
+ int sadPerBit,
+ int motionVectorRate,
+ int sumOfAbsoluteDifferences,
+ int expected)
+ => Assert.Equal(
+ expected,
+ Av1RateDistortion.GetMotionSearchSadCost(sadPerBit, motionVectorRate, sumOfAbsoluteDifferences));
+
[Theory]
[InlineData(0, 0, 52)]
[InlineData(0, 1, 3)]
diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs
index c6b0f04295..7592855724 100644
--- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs
+++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs
@@ -318,6 +318,101 @@ public class Av1IntraBlockCopyTests
Assert.Equal(new Av1MotionVector(24, -2568), candidates[1]);
}
+ ///
+ /// Verifies that NSTEP pixel search reaches an unaligned reconstructed match which has no exact source hash match.
+ ///
+ [Fact]
+ public void PixelSearchFindsUnalignedNonHashMatch()
+ {
+ const int Width = 640;
+ const int Height = 256;
+ const int QIndex = 23;
+ Point blockOrigin = new(0, 128);
+ Point predictionOrigin = new(15, 80);
+ ObuSequenceHeader sequenceHeader = CreateSequenceHeader();
+ ObuFrameHeader frameHeader = CreateFrameHeader();
+ frameHeader.AllowScreenContentTools = true;
+ frameHeader.AllowIntraBlockCopy = true;
+
+ using Av1EncoderPictureBuffer pictureBuffer = new(
+ Configuration.Default,
+ sequenceHeader,
+ frameHeader,
+ Width,
+ Height);
+
+ using Av1EncoderFrameBuffer source = new(
+ Configuration.Default,
+ Width,
+ Height,
+ 8,
+ Av1ColorFormat.Yuv400,
+ 0,
+ 0);
+
+ using Av1EncoderFrameBuffer reconstruction = new(
+ Configuration.Default,
+ Width,
+ Height,
+ 8,
+ Av1ColorFormat.Yuv400,
+ 0,
+ 0);
+
+ Buffer2DRegion sourceLuma = source.Frame.View.GetPlane(Av1Plane.Y);
+ Buffer2DRegion reconstructionLuma = reconstruction.Frame.View.GetPlane(Av1Plane.Y);
+ for (int row = 0; row < Height; row++)
+ {
+ sourceLuma.DangerousGetRowSpan(row).Clear();
+ reconstructionLuma.DangerousGetRowSpan(row).Clear();
+ }
+
+ for (int row = 0; row < 8; row++)
+ {
+ byte value = (byte)(100 + (row * 10));
+ sourceLuma.DangerousGetRowSpan(blockOrigin.Y + row).Slice(blockOrigin.X, 8).Fill(value);
+ reconstructionLuma.DangerousGetRowSpan(predictionOrigin.Y + row).Slice(predictionOrigin.X, 8).Fill(value);
+ }
+
+ Buffer2DRegion codedSourceLuma = source.Frame.CodedView.GetPlane(Av1Plane.Y);
+ Buffer2DRegion codedReconstructionLuma = reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y);
+ Assert.False(Av1IntraSuperblockEncoder.ByteOperator.BlocksEqual(codedSourceLuma, blockOrigin, predictionOrigin));
+ Assert.Equal(
+ 0,
+ Av1IntraSuperblockEncoder.ByteOperator.GetSumOfAbsoluteDifferences(
+ codedSourceLuma,
+ blockOrigin,
+ codedReconstructionLuma,
+ predictionOrigin));
+
+ Assert.Equal(
+ 8_640,
+ Av1IntraSuperblockEncoder.ByteOperator.GetSumOfAbsoluteDifferences(
+ codedSourceLuma,
+ blockOrigin,
+ codedReconstructionLuma,
+ new Point(15, 120)));
+
+ using Av1SymbolEncoder writer = new(Configuration.Default, 64, QIndex);
+ Span candidates = stackalloc Av1MotionVector[2];
+ int candidateCount = pictureBuffer.Picture.IntraBlockCopySearch
+ .FindPixelCandidates(
+ codedSourceLuma,
+ codedReconstructionLuma,
+ blockOrigin,
+ new Av1TileInfo(0, 0, frameHeader),
+ sequenceHeader,
+ writer,
+ new Av1MotionVector(-64, 120),
+ QIndex,
+ Av1RateDistortion.GetKeyFrameRateMultiplier(QIndex, Av1BitDepth.EightBit),
+ candidates);
+
+ Assert.Equal(1, candidateCount);
+ Assert.Equal(-384, candidates[0].Row);
+ Assert.Equal(120, candidates[0].Column);
+ }
+
///
/// Verifies high-bit-depth SIMD variance normalization against the eight-bit search domain.
///
@@ -355,14 +450,21 @@ public class Av1IntraBlockCopyTests
}
}
- int actual = Av1IntraSuperblockEncoder.UInt16Operator.GetVariance(
+ int sumOfAbsoluteDifferences = Av1IntraSuperblockEncoder.UInt16Operator.GetSumOfAbsoluteDifferences(
+ sourceLuma,
+ Point.Empty,
+ reconstructionLuma,
+ Point.Empty);
+
+ int variance = Av1IntraSuperblockEncoder.UInt16Operator.GetVariance(
sourceLuma,
Point.Empty,
reconstructionLuma,
Point.Empty,
Av1BitDepth.TwelveBit);
- Assert.Equal(16, actual);
+ Assert.Equal(1600, sumOfAbsoluteDifferences);
+ Assert.Equal(16, variance);
}
///