diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md
index 35c18d22fe..553c96f52e 100644
--- a/HEIF_IMPLEMENTATION_PLAN.md
+++ b/HEIF_IMPLEMENTATION_PLAN.md
@@ -255,21 +255,46 @@ Intra-reference frame-extent correction after checkpoint `182f39ae5`, verified o
**29,668** samples, with **0** samples exceeding one. These are same-bitstream decoder/reconstruction comparisons;
they do not establish separate-encoder parity or performance. No benchmark was run.
-The intra-edge investigation also confirmed these unresolved integration requirements:
-
-- `Av1PredictionDecoder.cs:989-1837` owns separate directional preparation, edge smoothing, upsampling,
- strength selection, and neighboring-mode selection. Native `reconintra.c:989-1082,1349-1381` uses endpoint
- extension, rounded nonnegative smoothing kernels, and clipped signed four-tap half-sample interpolation.
- No new numerical discrepancy in those arithmetic kernels has been established by this comparison.
-- Encoder `Av1EncoderModeDecisionWorkspace.cs:50-53,133-135` retains four raw edge spans with one prefix sample.
- The decoder needs writable prefix positions -1 and -2 and candidate-specific filtering; mutating those raw
- encoder spans across mode trials would contaminate later candidates. `Av1EncoderBlockWorkspace.cs:143-144`
- exposes transform scratch whose lifetime must be reconciled with directional prediction before sharing it.
-- `Av1IntraSuperblockEncoder.ModeDecision.cs:2272-2466` draws tiled edges from both committed reconstruction
- and the current candidate mosaic. Enabling filtering must preserve that distinction, coded extents,
- chroma neighbor ownership, corner preparation, and smooth-neighbor-dependent thresholds across all callers.
- The sequence flag remains disabled pending that complete integration. The decoder's private kernels are
- not a substitute for the required shared closed-generic traversal and semantic-operator architecture.
+Intra-edge integration after checkpoint `2424ff9f9`, verified on 2026-09-05:
+
+- Reference `av1/av1_cx_iface.c:333,1561-1562` enables intra-edge filtering by default and propagates it to
+ sequence configuration (`av1/encoder/encoder.c:641-647`). `Av1FrameEncoder.cs:402` now enables that syntax.
+ CDEF and restoration remain disabled and unresolved. The starting-tree findings above remain historical evidence.
+- Encoder `Av1TransformBlockEncoder.cs:739-800,875-937` now prepares directional edges before prediction.
+ Luma mode trials, selected-mode transform refinement, split luma transforms, tiled planes, and chroma candidates
+ propagate both the sequence flag and the neighboring smooth-mode class. Raw references remain separate from
+ candidate copies; filtering does not mutate references used by subsequent mode or transform trials.
+- Neighbor selection at `Av1IntraSuperblockEncoder.ChromaModeDecision.cs:1035-1081` follows native
+ `av1/common/av1_common_int.h:1359-1415` for the luma units that own subsampled chroma neighbors and
+ `reconintra.c:958-986` for smooth-mode classification. Inter winners can retain a previous intra trial's UV field
+ (`Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:925-933`), so that field is only meaningful for an intra neighbor.
+- `Av1IntraEdgePreparation.cs:39-116` shares the complete corner/filter/upsampling order between encoder and decoder.
+ Native `reconintra.c:1132-1147,1204-1243,1512-1548` defines the missing-sole-edge early return and directional
+ preparation. Strength thresholds follow `reconintra.c:989-1026`; half-sample selection follows `reconintra.h:148-155`.
+ The shared code preserves a missing sole edge's constant value rather than interpolating its distinct corner.
+- `Av1IntraEdgeFilter` and `Av1IntraEdgeUpsampler` have separate closed generic traversals and semantic readonly
+ operators, with descending 512/256/128-bit widths and scalar tails. Smoothing uses rounded nonnegative kernels;
+ upsampling uses signed [-1,9,9,-1] arithmetic, rounding, clipping, and linear interleaving. Native definitions are
+ `reconintra.c:1028-1082,1349-1381`. Inline comments explain endpoint padding, lane ordering, bounds, and scaling.
+- Each candidate borrows existing transform scratch (`Av1EncoderBlockWorkspace.cs:143-144`) until prediction and
+ residual formation finish. Two 160-sample edges retain native prefix sizing; only required edges are copied.
+ Smoothing uses 132 samples including three endpoint padding positions. Upsampling needs exactly the native
+ 19 samples, including corner and endpoint extension; vector reads no longer require a larger padded window.
+ Decoder scratch is 4,548 short samples (about 8.88 KiB), replacing its previous 4,576-sample workspace.
+ No new owner or per-candidate allocation was added. This source-level sizing result is not a timing claim.
+- Existing independent scalar kernel tests now cover all SIMD tiers, lengths around lane boundaries, extrema,
+ and exact scratch capacities. Eight added preparation cases distinguish smooth-neighbor thresholds and missing
+ sole edges in both orientations. Production tests assert the emitted sequence flag; mixed-partition tests retain
+ the four unfiltered cases and add four filtered cases without weakening partition or reconstruction assertions.
+- Final Release net11.0 build: zero errors and 1,009 existing warnings. Roslynk: zero compiler errors.
+ Serialized Visual Studio VSTest passed **314/314** in `intra-edge-final.trx` (30.7293 seconds), including encoder
+ frames, intra-superblocks, transform-block contracts, predictor SIMD tiers, native decoder fixtures and fallbacks,
+ HEIF encoder contracts, and the retained empty-transform cost-helper test.
+- Fresh optimized-reference decoding of eight regenerated partition streams matches all 16,640 retained luma samples.
+ Twelve regenerated moving color streams match all 21,348 Y/U/V samples. Combined maximum error is **0** across
+ **37,988** samples, with **0** exceeding one. These remain bounded same-bitstream reconstruction comparisons;
+ separate-encoder sample parity, complete decoder coverage, and end-to-end performance are still unverified.
+ No benchmark was run. Temporary scripts, native output, and reports remain outside the commit.
### Required completion gates
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs
index e2db3b5c24..0dd214715f 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs
@@ -399,7 +399,7 @@ internal static class Av1FrameEncoder
ForceIntegerMotionVector = Av1Constants.SelectIntegerMotionVector,
EnableFilterIntra = effort >= 4,
EnableDualFilter = !isStillPicture && effort >= MinimumDualInterpolationEffort,
- EnableIntraEdgeFilter = false,
+ EnableIntraEdgeFilter = true,
EnableSuperResolution = false,
EnableCdef = false,
EnableRestoration = false,
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs
index 61ffe4e8a9..f60f263d8a 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs
@@ -266,6 +266,7 @@ internal static partial class Av1IntraSuperblockEncoder
redLeft,
hasLeft,
hasAbove,
+ this.UseSmoothIntraEdges(macroBlock, lumaOrigin, blockSize, Av1Plane.U),
blueContext,
redContext,
paletteDisabledCost,
@@ -882,6 +883,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
predictionMode,
angleDelta,
+ this.picture.Sequence.SequenceHeader.EnableIntraEdgeFilter,
+ this.UseSmoothIntraEdges(macroBlock, lumaOrigin, blockSize, plane),
residual,
transformSize,
this.bitDepth);
@@ -1029,6 +1032,57 @@ internal static partial class Av1IntraSuperblockEncoder
return distortion;
}
+ ///
+ /// Derives the directional edge-filter class from the relevant neighboring coding blocks.
+ ///
+ private bool UseSmoothIntraEdges(Av1MacroBlockD macroBlock, Point lumaOrigin, Av1BlockSize blockSize, Av1Plane plane)
+ {
+ ObuColorConfig colorConfig = this.picture.Sequence.SequenceHeader.ColorConfig;
+ int subX = plane == Av1Plane.Y ? 0 : colorConfig.SubSamplingX ? 1 : 0;
+ int subY = plane == Av1Plane.Y ? 0 : colorConfig.SubSamplingY ? 1 : 0;
+ int row = lumaOrigin.Y >> Av1Constants.ModeInfoSizeLog2;
+ int column = lumaOrigin.X >> Av1Constants.ModeInfoSizeLog2;
+ bool hasAbove = macroBlock.IsUpAvailable;
+ bool hasLeft = macroBlock.IsLeftAvailable;
+ if (subX != 0 && blockSize.Get4x4WideCount() < 2)
+ {
+ hasLeft = column - 1 > macroBlock.Tile.ModeInfoColumnStart;
+ }
+
+ if (subY != 0 && blockSize.Get4x4HighCount() < 2)
+ {
+ hasAbove = row - 1 > macroBlock.Tile.ModeInfoRowStart;
+ }
+
+ // Chroma may cover several luma units. Its neighbors are the bottom-right luma units in the
+ // adjacent chroma regions, measured from the top-left unit covered by the current chroma block.
+ int baseOffset = -((row & subY) * macroBlock.ModeInfoStride) - (column & subX);
+ if (hasAbove && IsSmoothIntraNeighbor(
+ macroBlock.GetRelativeModeInfo(baseOffset - macroBlock.ModeInfoStride + subX).Block, plane))
+ {
+ return true;
+ }
+
+ return hasLeft && IsSmoothIntraNeighbor(
+ macroBlock.GetRelativeModeInfo(baseOffset + (subY * macroBlock.ModeInfoStride) - 1).Block, plane);
+ }
+
+ ///
+ /// Determines whether a neighboring block supplies the smooth edge-filter class.
+ ///
+ private static bool IsSmoothIntraNeighbor(Av1EncoderBlockModeInfo modeInfo, Av1Plane plane)
+ {
+ if (plane == Av1Plane.Y)
+ {
+ return modeInfo.Mode is Av1PredictionMode.Smooth or Av1PredictionMode.SmoothVertical or Av1PredictionMode.SmoothHorizontal;
+ }
+
+ // An inter winner can retain the preceding intra trial's UV field. That field has no inter
+ // meaning, so only an ordinary intra neighbor can select chroma smooth-edge thresholds.
+ return !modeInfo.UseIntraBlockCopy && modeInfo.Mode < Av1PredictionMode.InterModeStart
+ && modeInfo.UvMode is Av1ChromaPredictionMode.Smooth or Av1ChromaPredictionMode.SmoothVertical or Av1ChromaPredictionMode.SmoothHorizontal;
+ }
+
private long GetChromaCandidateCost(
Av1SymbolEncoder writer,
Av1MacroBlockModeInfo modeInfo,
@@ -1046,6 +1100,7 @@ internal static partial class Av1IntraSuperblockEncoder
ReadOnlySpan redLeft,
bool hasLeft,
bool hasAbove,
+ bool smoothIntraEdges,
Av1TransformBlockContext blueContext,
Av1TransformBlockContext redContext,
int paletteDisabledCost,
@@ -1078,6 +1133,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
predictionMode,
angleDelta,
+ this.picture.Sequence.SequenceHeader.EnableIntraEdgeFilter,
+ smoothIntraEdges,
candidateBlueCoefficients,
transformSize,
transformType,
@@ -1099,6 +1156,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
predictionMode,
angleDelta,
+ this.picture.Sequence.SequenceHeader.EnableIntraEdgeFilter,
+ smoothIntraEdges,
candidateRedCoefficients,
transformSize,
transformType,
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs
index f24353c8d6..8f76f64e25 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs
@@ -1408,6 +1408,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
mode,
angleDelta,
+ this.picture.Sequence.SequenceHeader.EnableIntraEdgeFilter,
+ this.UseSmoothIntraEdges(macroBlock, blockOrigin, blockSize, Av1Plane.Y),
residual,
transformSize,
this.bitDepth);
@@ -1539,6 +1541,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
bestMode,
selectedAngleDelta,
+ this.picture.Sequence.SequenceHeader.EnableIntraEdgeFilter,
+ this.UseSmoothIntraEdges(macroBlock, blockOrigin, blockSize, Av1Plane.Y),
residual,
transformSize,
this.bitDepth);
@@ -2131,6 +2135,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
mode,
angleDelta,
+ this.picture.Sequence.SequenceHeader.EnableIntraEdgeFilter,
+ this.UseSmoothIntraEdges(macroBlock, blockOrigin, BlockSize, Av1Plane.Y),
residual,
TransformSize,
this.bitDepth);
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs
index 51bdcf4be8..7bf726c86a 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs
@@ -166,6 +166,8 @@ internal static partial class Av1IntraSuperblockEncoder
/// Whether the top reference is available.
/// The intra prediction mode.
/// The signed directional-angle adjustment.
+ /// Whether sequence syntax enables directional edge filtering.
+ /// Whether a relevant neighboring block uses smooth prediction.
/// The candidate entropy-coding coefficients.
/// The transform dimensions.
/// The compound transform applied to the residual.
@@ -187,6 +189,8 @@ internal static partial class Av1IntraSuperblockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span quantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
@@ -210,6 +214,8 @@ internal static partial class Av1IntraSuperblockEncoder
/// Whether the top reference is available.
/// The intra prediction mode.
/// The signed directional-angle adjustment.
+ /// Whether sequence syntax enables directional edge filtering.
+ /// Whether a relevant neighboring block uses smooth prediction.
/// The contiguous source-minus-prediction destination.
/// The prediction dimensions.
/// The coded sample bit depth.
@@ -224,6 +230,8 @@ internal static partial class Av1IntraSuperblockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span residual,
Av1TransformSize transformSize,
Av1BitDepth bitDepth);
@@ -685,6 +693,8 @@ internal static partial class Av1IntraSuperblockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span quantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
@@ -705,6 +715,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
mode,
angleDelta,
+ enableIntraEdgeFilter,
+ smoothIntraEdges,
quantizedCoefficients,
transformSize,
transformType,
@@ -726,6 +738,8 @@ internal static partial class Av1IntraSuperblockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span residual,
Av1TransformSize transformSize,
Av1BitDepth bitDepth)
@@ -741,6 +755,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
mode,
angleDelta,
+ enableIntraEdgeFilter,
+ smoothIntraEdges,
residual,
transformSize);
@@ -1197,6 +1213,8 @@ internal static partial class Av1IntraSuperblockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span quantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
@@ -1217,6 +1235,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
mode,
angleDelta,
+ enableIntraEdgeFilter,
+ smoothIntraEdges,
quantizedCoefficients,
transformSize,
transformType,
@@ -1239,6 +1259,8 @@ internal static partial class Av1IntraSuperblockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span residual,
Av1TransformSize transformSize,
Av1BitDepth bitDepth)
@@ -1254,6 +1276,8 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
mode,
angleDelta,
+ enableIntraEdgeFilter,
+ smoothIntraEdges,
residual,
transformSize,
bitDepth);
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs
index a61ec59951..cf295115e2 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs
@@ -1,6 +1,7 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
+using System.Runtime.CompilerServices;
using System.Runtime.InteropServices;
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers;
using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
@@ -68,6 +69,8 @@ internal static class Av1TransformBlockEncoder
hasAbove,
Av1PredictionMode.DC,
0,
+ false,
+ false,
quantizedCoefficients,
transformSize,
transformType,
@@ -91,6 +94,8 @@ internal static class Av1TransformBlockEncoder
/// Whether the top reference is available.
/// The intra prediction mode.
/// The signed directional-angle adjustment.
+ /// Whether sequence syntax enables directional edge filtering.
+ /// Whether a relevant neighboring block uses smooth prediction.
/// The candidate entropy-coding coefficients.
/// The selected transform dimensions.
/// The selected compound transform type.
@@ -111,6 +116,8 @@ internal static class Av1TransformBlockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span quantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
@@ -136,6 +143,8 @@ internal static class Av1TransformBlockEncoder
hasAbove,
mode,
angleDelta,
+ enableIntraEdgeFilter,
+ smoothIntraEdges,
quantizedCoefficients,
transformSize,
transformType,
@@ -383,6 +392,8 @@ internal static class Av1TransformBlockEncoder
hasAbove,
Av1PredictionMode.DC,
0,
+ false,
+ false,
quantizedCoefficients,
transformSize,
transformType,
@@ -407,6 +418,8 @@ internal static class Av1TransformBlockEncoder
/// Whether the top reference is available.
/// The intra prediction mode.
/// The signed directional-angle adjustment.
+ /// Whether sequence syntax enables directional edge filtering.
+ /// Whether a relevant neighboring block uses smooth prediction.
/// The candidate entropy-coding coefficients.
/// The selected transform dimensions.
/// The selected compound transform type.
@@ -428,6 +441,8 @@ internal static class Av1TransformBlockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span quantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
@@ -454,6 +469,8 @@ internal static class Av1TransformBlockEncoder
hasAbove,
mode,
angleDelta,
+ enableIntraEdgeFilter,
+ smoothIntraEdges,
quantizedCoefficients,
transformSize,
transformType,
@@ -690,6 +707,8 @@ internal static class Av1TransformBlockEncoder
/// Whether the top reference is available.
/// The intra prediction mode.
/// The signed directional-angle adjustment.
+ /// Whether sequence syntax enables directional edge filtering.
+ /// Whether a relevant neighboring block uses smooth prediction.
/// The compact source-minus-prediction destination.
/// The prediction dimensions.
public static void PrepareIntraPrediction(
@@ -704,6 +723,8 @@ internal static class Av1TransformBlockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span residual,
Av1TransformSize transformSize)
{
@@ -718,9 +739,53 @@ internal static class Av1TransformBlockEncoder
}
else if (mode.IsDirectional())
{
- // The current encoder disables intra-edge filtering in sequence syntax. Zone-three transposition
- // borrows transform scratch because prediction completes before forward transformation starts.
- Span directionalScratch = MemoryMarshal.AsBytes(workspace.TransformWorkspace)[..(width * height)];
+ int angle = mode.ToAngle() + (angleDelta * Av1Constants.AngleStep);
+ Span scratch = MemoryMarshal.AsBytes(workspace.TransformWorkspace);
+ int predictionLength = width * height;
+ Span directionalScratch = scratch[..predictionLength];
+ bool upsampleAbove = false;
+ bool upsampleLeft = false;
+ if (enableIntraEdgeFilter)
+ {
+ // Mode trials share raw references. Prepare private edge copies after the directional scratch;
+ // this entire transform workspace is reusable once prediction and residual formation finish.
+ int edgeLength = Av1IntraEdgePreparation.ReferenceBufferLength;
+ int prefixLength = Av1IntraEdgePreparation.ReferencePrefixLength;
+ Span aboveStorage = scratch.Slice(predictionLength, edgeLength);
+ Span leftStorage = scratch.Slice(predictionLength + edgeLength, edgeLength);
+ aboveStorage.Fill(127);
+ leftStorage.Fill(129);
+ if (angle < 180)
+ {
+ above.CopyTo(aboveStorage[prefixLength..]);
+ aboveStorage[prefixLength - 1] = Unsafe.Subtract(ref MemoryMarshal.GetReference(above), 1);
+ }
+
+ if (angle > 90)
+ {
+ left.CopyTo(leftStorage[prefixLength..]);
+ leftStorage[prefixLength - 1] = Unsafe.Subtract(ref MemoryMarshal.GetReference(left), 1);
+ }
+
+ Span filteredAbove = aboveStorage[prefixLength..];
+ Span filteredLeft = leftStorage[prefixLength..];
+ Av1IntraEdgePreparation.Prepare(
+ filteredAbove,
+ filteredLeft,
+ width,
+ height,
+ angle,
+ hasAbove ? width : 0,
+ hasLeft ? height : 0,
+ smoothIntraEdges,
+ 8,
+ scratch.Slice(predictionLength + (2 * edgeLength), Av1IntraEdgeFilter.ScratchLength),
+ out upsampleAbove,
+ out upsampleLeft);
+
+ above = filteredAbove;
+ left = filteredLeft;
+ }
Av1DirectionalIntraPredictor.Predict(
prediction,
@@ -728,9 +793,9 @@ internal static class Av1TransformBlockEncoder
transformSize,
above,
left,
- false,
- false,
- mode.ToAngle() + (angleDelta * Av1Constants.AngleStep),
+ upsampleAbove,
+ upsampleLeft,
+ angle,
directionalScratch);
}
else
@@ -764,6 +829,8 @@ internal static class Av1TransformBlockEncoder
/// Whether the top reference is available.
/// The intra prediction mode.
/// The signed directional-angle adjustment.
+ /// Whether sequence syntax enables directional edge filtering.
+ /// Whether a relevant neighboring block uses smooth prediction.
/// The compact source-minus-prediction destination.
/// The prediction dimensions.
/// The coded sample bit depth.
@@ -779,6 +846,8 @@ internal static class Av1TransformBlockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span residual,
Av1TransformSize transformSize,
Av1BitDepth bitDepth)
@@ -806,7 +875,54 @@ internal static class Av1TransformBlockEncoder
}
else if (mode.IsDirectional())
{
- Span directionalScratch = MemoryMarshal.Cast(workspace.TransformWorkspace)[..(width * height)];
+ int angle = mode.ToAngle() + (angleDelta * Av1Constants.AngleStep);
+ Span scratch = MemoryMarshal.Cast(workspace.TransformWorkspace);
+ int predictionLength = width * height;
+ Span directionalScratch = scratch[..predictionLength];
+ bool upsampleAbove = false;
+ bool upsampleLeft = false;
+ if (enableIntraEdgeFilter)
+ {
+ // Mode trials share raw references. Prepare private edge copies after the directional scratch;
+ // this entire transform workspace is reusable once prediction and residual formation finish.
+ int edgeLength = Av1IntraEdgePreparation.ReferenceBufferLength;
+ int prefixLength = Av1IntraEdgePreparation.ReferencePrefixLength;
+ Span aboveStorage = scratch.Slice(predictionLength, edgeLength);
+ Span leftStorage = scratch.Slice(predictionLength + edgeLength, edgeLength);
+ int midpoint = 128 << (bitDepth.GetBitCount() - 8);
+ aboveStorage.Fill((short)(midpoint - 1));
+ leftStorage.Fill((short)(midpoint + 1));
+ if (angle < 180)
+ {
+ signedAbove.CopyTo(aboveStorage[prefixLength..]);
+ aboveStorage[prefixLength - 1] = Unsafe.Subtract(ref MemoryMarshal.GetReference(signedAbove), 1);
+ }
+
+ if (angle > 90)
+ {
+ signedLeft.CopyTo(leftStorage[prefixLength..]);
+ leftStorage[prefixLength - 1] = Unsafe.Subtract(ref MemoryMarshal.GetReference(signedLeft), 1);
+ }
+
+ Span filteredAbove = aboveStorage[prefixLength..];
+ Span filteredLeft = leftStorage[prefixLength..];
+ Av1IntraEdgePreparation.Prepare(
+ filteredAbove,
+ filteredLeft,
+ width,
+ height,
+ angle,
+ hasAbove ? width : 0,
+ hasLeft ? height : 0,
+ smoothIntraEdges,
+ bitDepth.GetBitCount(),
+ scratch.Slice(predictionLength + (2 * edgeLength), Av1IntraEdgeFilter.ScratchLength),
+ out upsampleAbove,
+ out upsampleLeft);
+
+ signedAbove = filteredAbove;
+ signedLeft = filteredLeft;
+ }
Av1DirectionalIntraPredictor.Predict(
signedPrediction,
@@ -814,9 +930,9 @@ internal static class Av1TransformBlockEncoder
transformSize,
signedAbove,
signedLeft,
- false,
- false,
- mode.ToAngle() + (angleDelta * Av1Constants.AngleStep),
+ upsampleAbove,
+ upsampleLeft,
+ angle,
directionalScratch);
}
else
@@ -850,6 +966,8 @@ internal static class Av1TransformBlockEncoder
/// Whether the top reference is available.
/// The intra prediction mode.
/// The signed directional-angle adjustment.
+ /// Whether sequence syntax enables directional edge filtering.
+ /// Whether a relevant neighboring block uses smooth prediction.
/// The retained entropy-coding coefficients.
/// The selected transform dimensions.
/// The selected compound transform type.
@@ -870,6 +988,8 @@ internal static class Av1TransformBlockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span quantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
@@ -891,6 +1011,8 @@ internal static class Av1TransformBlockEncoder
hasAbove,
mode,
angleDelta,
+ enableIntraEdgeFilter,
+ smoothIntraEdges,
workspace.Residual,
transformSize);
@@ -935,6 +1057,8 @@ internal static class Av1TransformBlockEncoder
/// Whether the top reference is available.
/// The intra prediction mode.
/// The signed directional-angle adjustment.
+ /// Whether sequence syntax enables directional edge filtering.
+ /// Whether a relevant neighboring block uses smooth prediction.
/// The retained entropy-coding coefficients.
/// The selected transform dimensions.
/// The selected compound transform type.
@@ -956,6 +1080,8 @@ internal static class Av1TransformBlockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ bool enableIntraEdgeFilter,
+ bool smoothIntraEdges,
Span quantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
@@ -978,6 +1104,8 @@ internal static class Av1TransformBlockEncoder
hasAbove,
mode,
angleDelta,
+ enableIntraEdgeFilter,
+ smoothIntraEdges,
workspace.Residual,
transformSize,
bitDepth);
diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Operations.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Operations.cs
new file mode 100644
index 0000000000..f6c48c57cf
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Operations.cs
@@ -0,0 +1,210 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Runtime.CompilerServices;
+using System.Runtime.InteropServices;
+using System.Runtime.Intrinsics;
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
+
+internal static partial class Av1IntraEdgeFilter
+{
+ ///
+ /// Traverses one edge using the arithmetic of a closed smoothing operator.
+ ///
+ /// The filter-strength arithmetic.
+ private static class Filter
+ where TOperator : struct, IEdgeFilterOperator
+ {
+ ///
+ /// Filters all samples following the preserved first sample.
+ ///
+ /// The first edge sample.
+ /// The number of samples including the preserved sample.
+ /// The reusable source workspace.
+ public static void Apply(ref byte edge, int count, Span scratch)
+ {
+ // Each convolution reads the original edge. Duplicate its first sample once and its last sample
+ // twice so the five-tap windows implement endpoint clamping without per-lane boundary branches.
+ scratch[0] = edge;
+ MemoryMarshal.CreateReadOnlySpan(ref edge, count).CopyTo(scratch[1..]);
+ scratch.Slice(count + 1, 2).Fill(Unsafe.Add(ref edge, count - 1));
+
+ ref byte source = ref MemoryMarshal.GetReference(scratch);
+ int outputCount = count - 1;
+ int i = 0;
+
+ // The same offset advances through descending SIMD widths. Adjacent lanes represent adjacent
+ // output samples, and only complete windows are loaded; the final incomplete window is scalar.
+ if (Vector512.IsHardwareAccelerated)
+ {
+ int vectorEnd = outputCount - Vector512.Count;
+ for (; i <= vectorEnd; i += Vector512.Count)
+ {
+ Vector512 s0 = Vector512.WidenLower(Vector512.Create(
+ Vector256.LoadUnsafe(ref source, (nuint)(i + 0)), Vector256.Zero));
+
+ Vector512 s1 = Vector512.WidenLower(Vector512.Create(
+ Vector256.LoadUnsafe(ref source, (nuint)(i + 1)), Vector256.Zero));
+
+ Vector512 s2 = Vector512.WidenLower(Vector512.Create(
+ Vector256.LoadUnsafe(ref source, (nuint)(i + 2)), Vector256.Zero));
+
+ Vector512 s3 = Vector512.WidenLower(Vector512.Create(
+ Vector256.LoadUnsafe(ref source, (nuint)(i + 3)), Vector256.Zero));
+
+ Vector512 s4 = Vector512.WidenLower(Vector512.Create(
+ Vector256.LoadUnsafe(ref source, (nuint)(i + 4)), Vector256.Zero));
+
+ Vector512 result = TOperator.Apply(s0, s1, s2, s3, s4);
+ Vector512.Narrow(result, Vector512.Zero).GetLower().StoreUnsafe(ref edge, (nuint)(i + 1));
+ }
+ }
+
+ if (Vector256.IsHardwareAccelerated)
+ {
+ int vectorEnd = outputCount - Vector256.Count;
+ for (; i <= vectorEnd; i += Vector256.Count)
+ {
+ Vector256 s0 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 0)), Vector128.Zero));
+
+ Vector256 s1 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 1)), Vector128.Zero));
+
+ Vector256 s2 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 2)), Vector128.Zero));
+
+ Vector256 s3 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 3)), Vector128.Zero));
+
+ Vector256 s4 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 4)), Vector128.Zero));
+
+ Vector256 result = TOperator.Apply(s0, s1, s2, s3, s4);
+ Vector256.Narrow(result, Vector256.Zero).GetLower().StoreUnsafe(ref edge, (nuint)(i + 1));
+ }
+ }
+
+ if (Vector128.IsHardwareAccelerated)
+ {
+ int vectorEnd = outputCount - Vector128.Count;
+ for (; i <= vectorEnd; i += Vector128.Count)
+ {
+ Vector128 s0 = Vector128.WidenLower(Vector128.Create(
+ Vector64.LoadUnsafe(ref source, (nuint)(i + 0)), Vector64.Zero));
+
+ Vector128 s1 = Vector128.WidenLower(Vector128.Create(
+ Vector64.LoadUnsafe(ref source, (nuint)(i + 1)), Vector64.Zero));
+
+ Vector128 s2 = Vector128.WidenLower(Vector128.Create(
+ Vector64.LoadUnsafe(ref source, (nuint)(i + 2)), Vector64.Zero));
+
+ Vector128 s3 = Vector128.WidenLower(Vector128.Create(
+ Vector64.LoadUnsafe(ref source, (nuint)(i + 3)), Vector64.Zero));
+
+ Vector128 s4 = Vector128.WidenLower(Vector128.Create(
+ Vector64.LoadUnsafe(ref source, (nuint)(i + 4)), Vector64.Zero));
+
+ Vector128 result = TOperator.Apply(s0, s1, s2, s3, s4);
+ Vector128.Narrow(result, Vector128.Zero).GetLower().StoreUnsafe(ref edge, (nuint)(i + 1));
+ }
+ }
+
+ for (; i < outputCount; i++)
+ {
+ int value = TOperator.Apply(
+ Unsafe.Add(ref source, i),
+ Unsafe.Add(ref source, i + 1),
+ Unsafe.Add(ref source, i + 2),
+ Unsafe.Add(ref source, i + 3),
+ Unsafe.Add(ref source, i + 4));
+
+ Unsafe.Add(ref edge, i + 1) = (byte)value;
+ }
+ }
+
+ ///
+ /// Filters all samples following the preserved first sample.
+ ///
+ /// The first edge sample.
+ /// The number of samples including the preserved sample.
+ /// The reusable source workspace.
+ public static void Apply(ref short edge, int count, Span scratch)
+ {
+ // Each convolution reads the original edge. Duplicate its first sample once and its last sample
+ // twice so the five-tap windows implement endpoint clamping without per-lane boundary branches.
+ scratch[0] = edge;
+ MemoryMarshal.CreateReadOnlySpan(ref edge, count).CopyTo(scratch[1..]);
+ scratch.Slice(count + 1, 2).Fill(Unsafe.Add(ref edge, count - 1));
+
+ ref short source = ref MemoryMarshal.GetReference(scratch);
+ int outputCount = count - 1;
+ int i = 0;
+
+ // The same offset advances through descending SIMD widths. Adjacent lanes represent adjacent
+ // output samples, and only complete windows are loaded; the final incomplete window is scalar.
+ // Nonnegative 12-bit samples have a maximum weighted sum of 65520. The rounding bias keeps
+ // that below 65536, so unsigned 16-bit lanes preserve the normative result at all strengths.
+ if (Vector512.IsHardwareAccelerated)
+ {
+ int vectorEnd = outputCount - Vector512.Count;
+ for (; i <= vectorEnd; i += Vector512.Count)
+ {
+ Vector512 s0 = Vector512.LoadUnsafe(ref source, (nuint)(i + 0)).AsUInt16();
+ Vector512 s1 = Vector512.LoadUnsafe(ref source, (nuint)(i + 1)).AsUInt16();
+ Vector512 s2 = Vector512.LoadUnsafe(ref source, (nuint)(i + 2)).AsUInt16();
+ Vector512 s3 = Vector512.LoadUnsafe(ref source, (nuint)(i + 3)).AsUInt16();
+ Vector512 s4 = Vector512.LoadUnsafe(ref source, (nuint)(i + 4)).AsUInt16();
+
+ Vector512 result = TOperator.Apply(s0, s1, s2, s3, s4);
+ result.AsInt16().StoreUnsafe(ref edge, (nuint)(i + 1));
+ }
+ }
+
+ if (Vector256.IsHardwareAccelerated)
+ {
+ int vectorEnd = outputCount - Vector256.Count;
+ for (; i <= vectorEnd; i += Vector256.Count)
+ {
+ Vector256 s0 = Vector256.LoadUnsafe(ref source, (nuint)(i + 0)).AsUInt16();
+ Vector256 s1 = Vector256.LoadUnsafe(ref source, (nuint)(i + 1)).AsUInt16();
+ Vector256 s2 = Vector256.LoadUnsafe(ref source, (nuint)(i + 2)).AsUInt16();
+ Vector256 s3 = Vector256.LoadUnsafe(ref source, (nuint)(i + 3)).AsUInt16();
+ Vector256 s4 = Vector256.LoadUnsafe(ref source, (nuint)(i + 4)).AsUInt16();
+
+ Vector256 result = TOperator.Apply(s0, s1, s2, s3, s4);
+ result.AsInt16().StoreUnsafe(ref edge, (nuint)(i + 1));
+ }
+ }
+
+ if (Vector128.IsHardwareAccelerated)
+ {
+ int vectorEnd = outputCount - Vector128.Count;
+ for (; i <= vectorEnd; i += Vector128.Count)
+ {
+ Vector128 s0 = Vector128.LoadUnsafe(ref source, (nuint)(i + 0)).AsUInt16();
+ Vector128 s1 = Vector128.LoadUnsafe(ref source, (nuint)(i + 1)).AsUInt16();
+ Vector128 s2 = Vector128.LoadUnsafe(ref source, (nuint)(i + 2)).AsUInt16();
+ Vector128 s3 = Vector128.LoadUnsafe(ref source, (nuint)(i + 3)).AsUInt16();
+ Vector128 s4 = Vector128.LoadUnsafe(ref source, (nuint)(i + 4)).AsUInt16();
+
+ Vector128 result = TOperator.Apply(s0, s1, s2, s3, s4);
+ result.AsInt16().StoreUnsafe(ref edge, (nuint)(i + 1));
+ }
+ }
+
+ for (; i < outputCount; i++)
+ {
+ int value = TOperator.Apply(
+ Unsafe.Add(ref source, i),
+ Unsafe.Add(ref source, i + 1),
+ Unsafe.Add(ref source, i + 2),
+ Unsafe.Add(ref source, i + 3),
+ Unsafe.Add(ref source, i + 4));
+
+ Unsafe.Add(ref edge, i + 1) = (short)value;
+ }
+ }
+ }
+}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Operator.cs
new file mode 100644
index 0000000000..1c2c20dc2a
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Operator.cs
@@ -0,0 +1,113 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Runtime.Intrinsics;
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
+
+///
+/// Smooths AV1 intra-reference edges while preserving their common-corner sample.
+///
+internal static partial class Av1IntraEdgeFilter
+{
+ ///
+ /// The sample count required for a maximal edge and its repeated endpoints.
+ ///
+ public const int ScratchLength = (2 * Av1Constants.MaxTransformSize) + 4;
+
+ ///
+ /// Defines the rounded smoothing arithmetic for one AV1 filter strength.
+ ///
+ internal interface IEdgeFilterOperator
+ {
+ ///
+ /// Filters one sample using the five neighboring positions.
+ ///
+ /// The samples two positions before the output.
+ /// The preceding samples.
+ /// The centered samples.
+ /// The following samples.
+ /// The samples two positions after the output.
+ /// The rounded filtered samples.
+ public static abstract int Apply(int a, int b, int c, int d, int e);
+
+ ///
+ /// Filters eight samples using the five neighboring positions.
+ ///
+ /// The samples two positions before the output.
+ /// The preceding samples.
+ /// The centered samples.
+ /// The following samples.
+ /// The samples two positions after the output.
+ /// The rounded filtered samples.
+ public static abstract Vector128 Apply(Vector128 a, Vector128 b, Vector128 c, Vector128 d, Vector128 e);
+
+ ///
+ /// Filters sixteen samples using the five neighboring positions.
+ ///
+ /// The samples two positions before the output.
+ /// The preceding samples.
+ /// The centered samples.
+ /// The following samples.
+ /// The samples two positions after the output.
+ /// The rounded filtered samples.
+ public static abstract Vector256 Apply(Vector256 a, Vector256 b, Vector256 c, Vector256 d, Vector256 e);
+
+ ///
+ /// Filters thirty-two samples using the five neighboring positions.
+ ///
+ /// The samples two positions before the output.
+ /// The preceding samples.
+ /// The centered samples.
+ /// The following samples.
+ /// The samples two positions after the output.
+ /// The rounded filtered samples.
+ public static abstract Vector512 Apply(Vector512 a, Vector512 b, Vector512 c, Vector512 d, Vector512 e);
+ }
+
+ ///
+ /// Filters an edge in place, leaving its first sample unchanged.
+ ///
+ /// The first edge sample, including the common corner when present.
+ /// The number of edge samples.
+ /// The smoothing strength from zero through three.
+ /// The source workspace with at least samples.
+ public static void Apply(ref byte edge, int count, int strength, Span scratch)
+ {
+ switch (strength)
+ {
+ case 1:
+ Filter.Apply(ref edge, count, scratch);
+ break;
+ case 2:
+ Filter.Apply(ref edge, count, scratch);
+ break;
+ case 3:
+ Filter.Apply(ref edge, count, scratch);
+ break;
+ }
+ }
+
+ ///
+ /// Filters an edge in place, leaving its first sample unchanged.
+ ///
+ /// The first edge sample, including the common corner when present.
+ /// The number of edge samples.
+ /// The smoothing strength from zero through three.
+ /// The source workspace with at least samples.
+ public static void Apply(ref short edge, int count, int strength, Span scratch)
+ {
+ switch (strength)
+ {
+ case 1:
+ Filter.Apply(ref edge, count, scratch);
+ break;
+ case 2:
+ Filter.Apply(ref edge, count, scratch);
+ break;
+ case 3:
+ Filter.Apply(ref edge, count, scratch);
+ break;
+ }
+ }
+}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength1Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength1Operator.cs
new file mode 100644
index 0000000000..bdd76e67da
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength1Operator.cs
@@ -0,0 +1,36 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Runtime.CompilerServices;
+using System.Runtime.Intrinsics;
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
+
+internal static partial class Av1IntraEdgeFilter
+{
+ ///
+ /// Applies the strength-1 three-tap edge smoothing kernel.
+ ///
+ internal readonly struct Strength1Operator : IEdgeFilterOperator
+ {
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static int Apply(int a, int b, int c, int d, int e)
+ => (b + (c << 1) + d + 2) >> 2;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector128 Apply(Vector128 a, Vector128 b, Vector128 c, Vector128 d, Vector128 e)
+ => (b + (c << 1) + d + Vector128.Create((ushort)2)) >> 2;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector256 Apply(Vector256 a, Vector256 b, Vector256 c, Vector256 d, Vector256 e)
+ => (b + (c << 1) + d + Vector256.Create((ushort)2)) >> 2;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector512 Apply(Vector512 a, Vector512 b, Vector512 c, Vector512 d, Vector512 e)
+ => (b + (c << 1) + d + Vector512.Create((ushort)2)) >> 2;
+ }
+}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength2Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength2Operator.cs
new file mode 100644
index 0000000000..c4f2f2f40e
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength2Operator.cs
@@ -0,0 +1,36 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Runtime.CompilerServices;
+using System.Runtime.Intrinsics;
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
+
+internal static partial class Av1IntraEdgeFilter
+{
+ ///
+ /// Applies the strength-2 three-tap edge smoothing kernel.
+ ///
+ internal readonly struct Strength2Operator : IEdgeFilterOperator
+ {
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static int Apply(int a, int b, int c, int d, int e)
+ => (((b + d) * 5) + (c * 6) + 8) >> 4;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector128 Apply(Vector128 a, Vector128 b, Vector128 c, Vector128 d, Vector128 e)
+ => (((b + d) * Vector128.Create((ushort)5)) + (c * Vector128.Create((ushort)6)) + Vector128.Create((ushort)8)) >> 4;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector256 Apply(Vector256 a, Vector256 b, Vector256 c, Vector256 d, Vector256 e)
+ => (((b + d) * Vector256.Create((ushort)5)) + (c * Vector256.Create((ushort)6)) + Vector256.Create((ushort)8)) >> 4;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector512 Apply(Vector512 a, Vector512 b, Vector512 c, Vector512 d, Vector512 e)
+ => (((b + d) * Vector512.Create((ushort)5)) + (c * Vector512.Create((ushort)6)) + Vector512.Create((ushort)8)) >> 4;
+ }
+}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength3Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength3Operator.cs
new file mode 100644
index 0000000000..38aaac7658
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeFilter.Strength3Operator.cs
@@ -0,0 +1,36 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Runtime.CompilerServices;
+using System.Runtime.Intrinsics;
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
+
+internal static partial class Av1IntraEdgeFilter
+{
+ ///
+ /// Applies the strength-3 five-tap edge smoothing kernel.
+ ///
+ internal readonly struct Strength3Operator : IEdgeFilterOperator
+ {
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static int Apply(int a, int b, int c, int d, int e)
+ => (a + ((b + c + d) << 1) + e + 4) >> 3;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector128 Apply(Vector128 a, Vector128 b, Vector128 c, Vector128 d, Vector128 e)
+ => (a + ((b + c + d) << 1) + e + Vector128.Create((ushort)4)) >> 3;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector256 Apply(Vector256 a, Vector256 b, Vector256 c, Vector256 d, Vector256 e)
+ => (a + ((b + c + d) << 1) + e + Vector256.Create((ushort)4)) >> 3;
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector512 Apply(Vector512 a, Vector512 b, Vector512 c, Vector512 d, Vector512 e)
+ => (a + ((b + c + d) << 1) + e + Vector512.Create((ushort)4)) >> 3;
+ }
+}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgePreparation.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgePreparation.cs
new file mode 100644
index 0000000000..1202f39bc5
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgePreparation.cs
@@ -0,0 +1,289 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Numerics;
+using System.Runtime.CompilerServices;
+using System.Runtime.InteropServices;
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
+
+///
+/// Prepares directional intra-reference edges for AV1 smoothing and half-sample prediction.
+///
+internal static class Av1IntraEdgePreparation
+{
+ ///
+ /// The number of samples reserved before the first edge sample.
+ ///
+ public const int ReferencePrefixLength = 16;
+
+ ///
+ /// The total sample capacity of one edge including prefix and extension.
+ ///
+ public const int ReferenceBufferLength = (2 * Av1Constants.MaxTransformSize) + 32;
+
+ ///
+ /// Filters and upsamples prepared directional reference edges.
+ ///
+ /// The byte or signed high-bit-depth sample type.
+ /// The top edge with writable prefix and extension.
+ /// The left edge with writable prefix and extension.
+ /// The transform width.
+ /// The transform height.
+ /// The adjusted directional angle.
+ /// The number of available top samples before extension.
+ /// The number of available left samples before extension.
+ /// Whether a relevant neighbor uses smooth prediction.
+ /// The coded sample precision.
+ /// The original-edge workspace with at least samples.
+ /// Whether the top edge contains half-sample positions.
+ /// Whether the left edge contains half-sample positions.
+ public static void Prepare(
+ Span above,
+ Span left,
+ int width,
+ int height,
+ int angle,
+ int topCount,
+ int leftCount,
+ bool filterType,
+ int bitDepth,
+ Span scratch,
+ out bool upsampleAbove,
+ out bool upsampleLeft)
+ where T : unmanaged, IBinaryInteger
+ {
+ bool needAbove = angle < 180;
+ bool needLeft = angle > 90;
+ bool needRight = angle < 90;
+ bool needBottom = angle > 180;
+ upsampleAbove = false;
+ upsampleLeft = false;
+
+ // A missing sole edge produces a constant block from the perpendicular sample or midpoint offset.
+ // Its prepared edge already repeats that value. Upsampling its distinct corner would change it.
+ if ((!needAbove && leftCount == 0) || (!needLeft && topCount == 0))
+ {
+ return;
+ }
+
+ if (angle is not 90 and not 180)
+ {
+ if (needAbove && needLeft && width + height >= 24)
+ {
+ // The corner is one logical sample represented in both edge prefixes. Filter it first,
+ // then let both edge convolutions read the same rounded [5, 6, 5] corner value.
+ ref T corner = ref Unsafe.Subtract(ref above[0], 1);
+ int value = (5 * int.CreateChecked(left[0]))
+ + (6 * int.CreateChecked(corner))
+ + (5 * int.CreateChecked(above[0]));
+
+ corner = T.CreateChecked((value + 8) >> 4);
+ Unsafe.Subtract(ref left[0], 1) = corner;
+ }
+
+ if (needAbove && topCount > 0)
+ {
+ int strength = IntraEdgeFilterStrength(width, height, angle - 90, filterType);
+ Filter(ref Unsafe.Subtract(ref above[0], 1), topCount + 1 + (needRight ? height : 0), strength, scratch);
+ }
+
+ if (needLeft && leftCount > 0)
+ {
+ int strength = IntraEdgeFilterStrength(height, width, angle - 180, filterType);
+ Filter(ref Unsafe.Subtract(ref left[0], 1), leftCount + 1 + (needBottom ? width : 0), strength, scratch);
+ }
+ }
+
+ upsampleAbove = UseUpsampling(width, height, angle - 90, filterType);
+ if (needAbove && upsampleAbove)
+ {
+ Upsample(above, width + (needRight ? height : 0), bitDepth, scratch);
+ }
+
+ upsampleLeft = UseUpsampling(height, width, angle - 180, filterType);
+ if (needLeft && upsampleLeft)
+ {
+ Upsample(left, height + (needBottom ? width : 0), bitDepth, scratch);
+ }
+ }
+
+ ///
+ /// Selects half-sample interpolation for a transform edge.
+ ///
+ /// The transform width.
+ /// The transform height.
+ /// The angle relative to the edge's cardinal direction.
+ /// Whether a relevant neighbor uses smooth prediction.
+ /// Whether the edge uses half-sample interpolation.
+ private static bool UseUpsampling(int width, int height, int delta, bool filterType)
+ {
+ int distance = Math.Abs(delta);
+ return distance > 0 && distance < 40 && width + height <= (filterType ? 8 : 16);
+ }
+
+ ///
+ /// Dispatches edge smoothing to the concrete sample representation.
+ ///
+ /// The byte or signed high-bit-depth sample type.
+ /// The first edge sample, including the corner.
+ /// The number of edge samples.
+ /// The smoothing strength.
+ /// The reusable original-edge workspace.
+ private static void Filter(ref T edge, int count, int strength, Span scratch)
+ where T : unmanaged, IBinaryInteger
+ {
+ if (typeof(T) == typeof(byte))
+ {
+ Av1IntraEdgeFilter.Apply(ref Unsafe.As(ref edge), count, strength, MemoryMarshal.Cast(scratch));
+ }
+ else
+ {
+ Av1IntraEdgeFilter.Apply(ref Unsafe.As(ref edge), count, strength, MemoryMarshal.Cast(scratch));
+ }
+ }
+
+ ///
+ /// Dispatches half-sample interpolation to the concrete sample representation.
+ ///
+ /// The byte or signed high-bit-depth sample type.
+ /// The edge with writable prefix and extension.
+ /// The number of original edge samples.
+ /// The coded precision.
+ /// The reusable original-edge workspace.
+ private static void Upsample(Span edge, int count, int bitDepth, Span scratch)
+ where T : unmanaged, IBinaryInteger
+ {
+ if (typeof(T) == typeof(byte))
+ {
+ Av1IntraEdgeUpsampler.Apply(MemoryMarshal.Cast(edge), count, MemoryMarshal.Cast(scratch));
+ }
+ else
+ {
+ Av1IntraEdgeUpsampler.Apply(MemoryMarshal.Cast(edge), count, bitDepth, MemoryMarshal.Cast(scratch));
+ }
+ }
+
+ ///
+ /// Selects the AV1 intra-edge filter strength for the block dimensions and prediction angle.
+ ///
+ /// The edge's primary block dimension.
+ /// The edge's secondary block dimension.
+ /// The prediction angle relative to the edge's cardinal direction.
+ /// A value indicating whether a neighboring smooth mode selects the alternate thresholds.
+ /// The filter strength from zero for no filtering through three for the strongest kernel.
+ private static int IntraEdgeFilterStrength(int width, int height, int delta, bool filterType)
+ {
+ int d = Math.Abs(delta);
+ int strength = 0;
+ int widthHeight = width + height;
+ if (!filterType)
+ {
+ if (widthHeight <= 8)
+ {
+ if (d >= 56)
+ {
+ strength = 1;
+ }
+ }
+ else if (widthHeight <= 12)
+ {
+ if (d >= 40)
+ {
+ strength = 1;
+ }
+ }
+ else if (widthHeight <= 16)
+ {
+ if (d >= 40)
+ {
+ strength = 1;
+ }
+ }
+ else if (widthHeight <= 24)
+ {
+ if (d >= 8)
+ {
+ strength = 1;
+ }
+
+ if (d >= 16)
+ {
+ strength = 2;
+ }
+
+ if (d >= 32)
+ {
+ strength = 3;
+ }
+ }
+ else if (widthHeight <= 32)
+ {
+ if (d >= 1)
+ {
+ strength = 1;
+ }
+
+ if (d >= 4)
+ {
+ strength = 2;
+ }
+
+ if (d >= 32)
+ {
+ strength = 3;
+ }
+ }
+ else
+ {
+ if (d >= 1)
+ {
+ strength = 3;
+ }
+ }
+ }
+ else
+ {
+ if (widthHeight <= 8)
+ {
+ if (d >= 40)
+ {
+ strength = 1;
+ }
+
+ if (d >= 64)
+ {
+ strength = 2;
+ }
+ }
+ else if (widthHeight <= 16)
+ {
+ if (d >= 20)
+ {
+ strength = 1;
+ }
+
+ if (d >= 48)
+ {
+ strength = 2;
+ }
+ }
+ else if (widthHeight <= 24)
+ {
+ if (d >= 4)
+ {
+ strength = 3;
+ }
+ }
+ else
+ {
+ if (d >= 1)
+ {
+ strength = 3;
+ }
+ }
+ }
+
+ return strength;
+ }
+}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeUpsampler.FourTapOperator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeUpsampler.FourTapOperator.cs
new file mode 100644
index 0000000000..b99523ce02
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeUpsampler.FourTapOperator.cs
@@ -0,0 +1,48 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Runtime.CompilerServices;
+using System.Runtime.Intrinsics;
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
+
+internal static partial class Av1IntraEdgeUpsampler
+{
+ ///
+ /// Applies the AV1 [-1, 9, 9, -1] interpolation kernel with Q4 rounding and clipping.
+ ///
+ internal readonly struct FourTapOperator : IEdgeUpsamplingOperator
+ {
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static int Interpolate(int a, int b, int c, int d, int maximum)
+ => Math.Clamp((((9 * (b + c)) - a - d) + 8) >> 4, 0, maximum);
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector128 Interpolate(Vector128 a, Vector128 b, Vector128 c, Vector128 d, int maximum)
+ {
+ // Signed 32-bit lanes preserve negative overshoot and the 12-bit central sum, which can reach 73710.
+ Vector128 value = (((Vector128.Create(9) * (b + c)) - a - d) + Vector128.Create(8)) >> 4;
+ return Vector128.Clamp(value, Vector128.Zero, Vector128.Create(maximum));
+ }
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector256 Interpolate(Vector256 a, Vector256 b, Vector256 c, Vector256 d, int maximum)
+ {
+ // Signed 32-bit lanes preserve negative overshoot and the 12-bit central sum, which can reach 73710.
+ Vector256 value = (((Vector256.Create(9) * (b + c)) - a - d) + Vector256.Create(8)) >> 4;
+ return Vector256.Clamp(value, Vector256.Zero, Vector256.Create(maximum));
+ }
+
+ ///
+ [MethodImpl(MethodImplOptions.AggressiveInlining)]
+ public static Vector512 Interpolate(Vector512 a, Vector512 b, Vector512 c, Vector512 d, int maximum)
+ {
+ // Signed 32-bit lanes preserve negative overshoot and the 12-bit central sum, which can reach 73710.
+ Vector512 value = (((Vector512.Create(9) * (b + c)) - a - d) + Vector512.Create(8)) >> 4;
+ return Vector512.Clamp(value, Vector512.Zero, Vector512.Create(maximum));
+ }
+ }
+}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeUpsampler.Operations.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeUpsampler.Operations.cs
new file mode 100644
index 0000000000..106905c053
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1IntraEdgeUpsampler.Operations.cs
@@ -0,0 +1,278 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Runtime.CompilerServices;
+using System.Runtime.InteropServices;
+using System.Runtime.Intrinsics;
+using SixLabors.ImageSharp.Common.Helpers;
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Prediction;
+
+internal static partial class Av1IntraEdgeUpsampler
+{
+ ///
+ /// Traverses a bounded edge using a closed interpolation operator.
+ ///
+ /// The four-tap interpolation arithmetic.
+ private static class Upsampler
+ where TOperator : struct, IEdgeUpsamplingOperator
+ {
+ ///
+ /// Inserts half samples using the original edge values and repeated endpoints.
+ ///
+ /// The edge with prefix and doubled output capacity.
+ /// The original sample count.
+ /// The reusable original-sample workspace.
+ public static void Apply(Span edge, int count, Span scratch)
+ {
+ ref byte destination = ref MemoryMarshal.GetReference(edge);
+ ref byte source = ref MemoryMarshal.GetReference(scratch);
+
+ // Preserve the corner twice and the final sample once. Every SIMD load below covers exactly its
+ // input lanes, so the native 16+3-sample workspace also suffices for the widest interpolation.
+ source = Unsafe.Subtract(ref destination, 1);
+ Unsafe.Add(ref source, 1) = source;
+ edge[..count].CopyTo(scratch[2..]);
+ Unsafe.Add(ref source, count + 2) = edge[count - 1];
+ Unsafe.Subtract(ref destination, 2) = source;
+ ref byte firstOutput = ref Unsafe.Subtract(ref destination, 1);
+ int i = 0;
+
+ if (Vector512.IsHardwareAccelerated)
+ {
+ int vectorEnd = count - Vector512.Count;
+ for (; i <= vectorEnd; i += Vector512.Count)
+ {
+ Vector256 w0 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 0)), Vector128.Zero));
+
+ Vector512 s0 = Vector512.WidenLower(Vector512.Create(w0, Vector256.Zero)).AsInt32();
+
+ Vector256 w1 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 1)), Vector128.Zero));
+
+ Vector512 s1 = Vector512.WidenLower(Vector512.Create(w1, Vector256.Zero)).AsInt32();
+
+ Vector256 w2 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 2)), Vector128.Zero));
+
+ Vector512 s2 = Vector512.WidenLower(Vector512.Create(w2, Vector256.Zero)).AsInt32();
+
+ Vector256 w3 = Vector256.WidenLower(Vector256.Create(
+ Vector128.LoadUnsafe(ref source, (nuint)(i + 3)), Vector128.Zero));
+
+ Vector512 s3 = Vector512.WidenLower(Vector512.Create(w3, Vector256.Zero)).AsInt32();
+
+ Vector512 values = TOperator.Interpolate(s0, s1, s2, s3, 255);
+ Vector256 halfWords = Vector512.Narrow(values, Vector512.Zero).GetLower().AsUInt16();
+ Vector128 halfSamples = Vector256.Narrow(halfWords, Vector256.Zero).GetLower();
+ Vector128 originals = Vector128.LoadUnsafe(ref source, (nuint)(i + 2));
+
+ // Unpack the lower and upper eight pairs independently to retain linear sample order.
+ Vector128_.UnpackLow(halfSamples, originals).StoreUnsafe(ref firstOutput, (nuint)(2 * i));
+ Vector128_.UnpackHigh(halfSamples, originals).StoreUnsafe(ref firstOutput, (nuint)((2 * i) + 16));
+ }
+ }
+
+ if (Vector256.IsHardwareAccelerated)
+ {
+ int vectorEnd = count - Vector256.Count;
+ for (; i <= vectorEnd; i += Vector256.Count)
+ {
+ Vector128 w0 = Vector128.WidenLower(Vector128.Create(
+ Vector64.LoadUnsafe(ref source, (nuint)(i + 0)), Vector64.Zero));
+
+ Vector256 s0 = Vector256.WidenLower(Vector256.Create(w0, Vector128.Zero)).AsInt32();
+
+ Vector128 w1 = Vector128.WidenLower(Vector128.Create(
+ Vector64.LoadUnsafe(ref source, (nuint)(i + 1)), Vector64.Zero));
+
+ Vector256