diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md
index 951a33b0de..62cf022541 100644
--- a/HEIF_IMPLEMENTATION_PLAN.md
+++ b/HEIF_IMPLEMENTATION_PLAN.md
@@ -841,10 +841,11 @@ Encoder verification contract:
- [~] The unused coefficient-shape transform facade and its unimplemented N2, N4, and DC-only branches are removed. Finalized block encoding now follows the complete-transform path that current libaom uses before fast quantization; later rate-distortion search may add proven coefficient optimization without exposing inactive runtime throws.
- [~] Forward-quantizer FeatureTestRunner and zero-allocation tests compare every hardware tier with an independent scan-order scalar oracle shaped from current-main libaom. Both passed direct net11 VSTest in Release.
- [~] The combined-frame writer now completes the byte-counted uncompressed frame header before starting the optional multi-tile tile-group flag, matching current libaom's separate frame-header and tile-group writers. A non-uniform two-tile round trip verifies the explicit boundaries, both tile payloads, and complete stream consumption through direct net11 VSTest in Release.
-- [~] The first internal frame-to-OBU operation encodes 8-, 10-, and 12-bit monochrome reduced still pictures through the production tile writer and production decoder. Coefficient context initialization now stores `min(abs(level), 127)`, matching current libaom; the previous signed clamp converted every negative transform coefficient to zero and selected invalid nonzero-map distributions. Signed dense and sparse entropy round trips, direct level-buffer saturation coverage, and eight constant/gradient frame cases pass 52 of 52 direct net11 VSTest cases in Release. Current-main `aomdec` accepts all eight emitted payloads. Their decoded-frame MD5 values are `d09ea148582b9c93fa78e59426193bbc` (16x16 8-bit constant), `700698094fb922d38361e04d56e73aa6` (16x16 8-bit gradient), `f949f7422913e83dff07ee5e0a5087d3` (8x8 8-bit constant), `ae7233a94558978934469dcc4da764dd` (8x8 8-bit gradient), `09223b227f3abc3134d0a3ea15f70c0a` (8x8 10-bit constant), `539aab0e6e14bcaec271febfa8e25444` (8x8 10-bit gradient), `73117a8fc102e5d028f82444fc4d15ab` (8x8 12-bit constant), and `3cbecfd4eb5b8aeb7520e57927d6b6ba` (8x8 12-bit gradient). This is an independently decodable baseline, not completion evidence for chroma, alpha, options, containers, or the public encoder.
+- [~] The first internal frame-to-OBU operation encodes 8-, 10-, and 12-bit monochrome reduced still pictures through the production tile writer and production decoder. Coefficient context initialization now stores `min(abs(level), 127)`, matching current libaom; the previous signed clamp converted every negative transform coefficient to zero and selected invalid nonzero-map distributions. Signed dense and sparse entropy round trips, direct level-buffer saturation coverage, and eight constant/gradient frame cases pass 52 of 52 direct net11 VSTest cases in Release. Current-main `aomdec` accepts all eight emitted payloads. After winner-mode transform refinement, their decoded-frame MD5 values are `d09ea148582b9c93fa78e59426193bbc` (16x16 8-bit constant), `b83eedd5a84428f0120130253b30bdaa` (16x16 8-bit gradient), `f949f7422913e83dff07ee5e0a5087d3` (8x8 8-bit constant), `ae7233a94558978934469dcc4da764dd` (8x8 8-bit gradient), `09223b227f3abc3134d0a3ea15f70c0a` (8x8 10-bit constant), `539aab0e6e14bcaec271febfa8e25444` (8x8 10-bit gradient), `73117a8fc102e5d028f82444fc4d15ab` (8x8 12-bit constant), and `6936a2b62d7220dfb12f3763bb49965d` (8x8 12-bit gradient). This is an independently decodable baseline, not completion evidence for chroma, alpha, options, containers, or the public encoder.
- [x] The exact net11 Release rebuild completed at the established 1,005-warning repository baseline with zero errors. The complete HEIF/AV1 namespace passes 8,838 of 8,838 direct VSTest cases with zero failures or skips. Roslynk reports zero compiler errors and no diagnostics in the five changed C# files; `git diff --check` passes and `.gitattributes` is unchanged.
-- [~] The same internal frame operation now produces 4:2:0, 4:2:2, and 4:4:4 payloads at 8, 10, and 12 bits. Twenty-one color cases cover constant and spatially varying input at aligned dimensions plus odd 13x11 visible dimensions for every chroma geometry. The production decoder consumes every payload, the decoded output retains non-neutral chroma, and current-main `aomdec` accepts all 29 monochrome and color outputs. After live spatial chroma mode selection and implicit chroma-transform correction, the odd-dimension decoded-frame MD5 values are `1721dc6a018892e7fca5d4e003be1058` (4:2:0), `90dfa2d705410c86bff3ba750efec0ac` (4:2:2), and `dbd15340e2cb7d2f731c74eb1b885e0d` (4:4:4). This proves legal current-libaom payload syntax across native plane geometries; it does not yet prove target quality or native-plane equality with an independently encoded reference.
+- [~] The same internal frame operation now produces 4:2:0, 4:2:2, and 4:4:4 payloads at 8, 10, and 12 bits. Twenty-one color cases cover constant and spatially varying input at aligned dimensions plus odd 13x11 visible dimensions for every chroma geometry. The production decoder consumes every payload, the decoded output retains non-neutral chroma, and current-main `aomdec` accepts all 29 monochrome and color outputs. After live spatial chroma mode selection, implicit chroma-transform correction, and winner-mode luma-transform refinement, the odd-dimension decoded-frame MD5 values are `94ced594b3bc0fcbbfd55559a7e0088c` (4:2:0), `752777943f6e4ae7f6f879075dd311fe` (4:2:2), and `d25cc5e0f8404608530d6fec2f88915f` (4:4:4). This proves legal current-libaom payload syntax across native plane geometries; it does not yet prove target quality or native-plane equality with an independently encoded reference.
- [x] Spatial chroma candidates now use the implicit transform derived from the selected UV mode and active transform set, matching current libaom's `intra_mode_to_tx_type` and `av1_get_tx_type` behavior. The same shared derivation is consumed by the decoder, so coefficient scan order, entropy contexts, inverse reconstruction, and encoder rate estimates cannot drift between the two paths. The previous DCT-DCT candidate transform could produce syntactically accepted streams whose non-DC chroma coefficients were interpreted under a different implicit transform. Six production mode-decision cases retain nonzero U and V coefficients and assert the selected transform state across 4:2:0, 4:2:2, and 4:4:4; fifteen exact mapping cases cover every intra mode, reduced sets, and the 32x32 DCT-only fallback. The focused contract passes 21 of 21 direct net11 VSTest cases, the complete HEIF/AV1 namespace passes 8,947 of 8,947, the exact Release rebuild remains at 1,005 warnings and zero errors, and current-main `aomdec` accepts all 29 regenerated payloads.
+- [~] Luma mode selection now evaluates each of its 61 mode-and-angle candidates with the mode-derived default transform used by current libaom's fast intra path. It then refines only the winning mode across all seven transform types permitted by the 8x8 intra set in transform-enum order. This removes the fixed DCT-DCT limitation while avoiding a 61-by-7 expansion; each trial includes live transform-type and coefficient rate, reconstructed pixel-domain distortion, and the existing allocation-free stack scratch. Eighteen exact-prediction production cases prove DCT-DCT wins equal-cost ties in reference order even when the first pass used a different default, while the 72x72 textured traversal proves a non-DCT transform with nonzero coefficients reaches retained syntax. Current-main `aomdec` accepts all 29 regenerated payloads. Full partition, transform-size, and effort-dependent joint mode/transform search remain.
- [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips.
- [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips.
- [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips.
diff --git a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolContextHelper.cs b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolContextHelper.cs
index 111554d88d..ac75040cfe 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolContextHelper.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolContextHelper.cs
@@ -543,13 +543,13 @@ internal static class Av1SymbolContextHelper
}
///
- /// Gets the implicit intra transform type after applying the active transform-set restriction.
+ /// Gets the mode-derived default intra transform after applying the active transform-set restriction.
///
/// The intra prediction mode.
/// The coded transform size.
/// Indicates whether the frame restricts transform choices.
- /// The implicit transform type used to encode and decode the block.
- public static Av1TransformType GetImplicitIntraTransformType(
+ /// The permitted default transform for the mode.
+ public static Av1TransformType GetDefaultIntraTransformType(
Av1PredictionMode mode,
Av1TransformSize transformSize,
bool useReducedSet)
@@ -557,7 +557,7 @@ internal static class Av1SymbolContextHelper
Av1TransformType transformType = mode.ToTransformType();
Av1TransformSetType transformSetType = GetExtendedTransformSetType(transformSize, useReducedSet);
- // An implicit mode-derived transform falls back to DCT-DCT when its transform set omits that type.
+ // A mode-derived transform falls back to DCT-DCT when its transform set omits that type.
return transformType.IsExtendedSetUsed(transformSetType) ? transformType : Av1TransformType.DctDct;
}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolDecoder.cs b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolDecoder.cs
index 93fe1010a8..2179d6788e 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolDecoder.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolDecoder.cs
@@ -1458,7 +1458,7 @@ internal ref struct Av1SymbolDecoder
else
{
// Chroma has its own intra mode, so its implicit transform must be derived independently of luma.
- transformType = Av1SymbolContextHelper.GetImplicitIntraTransformType(
+ transformType = Av1SymbolContextHelper.GetDefaultIntraTransformType(
modeInfo.UvMode.ToLumaMode(),
transformSize,
useReducedTransformSet);
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs
index c8c9599e60..0b6c3c7af9 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs
@@ -262,7 +262,7 @@ internal static partial class Av1IntraSuperblockEncoder
{
const Av1BlockSize BlockSize = Av1BlockSize.Block8x8;
Av1PredictionMode predictionMode = chromaMode.ToLumaMode();
- Av1TransformType transformType = Av1SymbolContextHelper.GetImplicitIntraTransformType(
+ Av1TransformType transformType = Av1SymbolContextHelper.GetDefaultIntraTransformType(
predictionMode,
transformSize,
this.picture.Parent.FrameHeader.UseReducedTransformSet);
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs
index 787be3aac3..f94959545b 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs
@@ -376,6 +376,10 @@ internal static partial class Av1IntraSuperblockEncoder
int deltaCount = AngleDeltaSearchOrder.Length;
int directionalModeCount = (int)Av1PredictionMode.Directional67Degrees - (int)Av1PredictionMode.Vertical + 1;
int candidateCount = baseModeCount + (directionalModeCount * deltaCount);
+ bool useReducedTransformSet = this.picture.Parent.FrameHeader.UseReducedTransformSet;
+ Av1TransformSetType transformSetType = Av1SymbolContextHelper.GetExtendedTransformSetType(
+ TransformSize,
+ useReducedTransformSet);
// Zero-angle modes precede groups of six nonzero adjustments for each directional mode.
// A single index preserves that tie-breaking order without duplicating candidate evaluation.
@@ -395,6 +399,11 @@ internal static partial class Av1IntraSuperblockEncoder
angleDelta = AngleDeltaSearchOrder[adjustedIndex % deltaCount];
}
+ Av1TransformType defaultTransformType = Av1SymbolContextHelper.GetDefaultIntraTransformType(
+ mode,
+ TransformSize,
+ useReducedTransformSet);
+
Av1EncoderTransformBlockState candidateState = default;
long candidateCost = this.GetLumaCandidateCost(
writer,
@@ -407,6 +416,7 @@ internal static partial class Av1IntraSuperblockEncoder
hasAbove,
mode,
angleDelta,
+ defaultTransformType,
blockContext,
candidateReconstruction,
candidateCoefficients,
@@ -430,6 +440,52 @@ internal static partial class Av1IntraSuperblockEncoder
}
}
+ // Mode selection uses the mode-derived default transform, then the winning mode alone pays for
+ // an exhaustive transform refinement. This preserves reference tie order without a 61-by-7 search.
+ long bestTransformCost = long.MaxValue;
+ for (Av1TransformType transformType = Av1TransformType.DctDct;
+ transformType < Av1TransformType.AllTransformTypes;
+ transformType++)
+ {
+ if (!transformType.IsExtendedSetUsed(transformSetType))
+ {
+ continue;
+ }
+
+ Av1EncoderTransformBlockState candidateState = default;
+ long candidateCost = this.GetLumaCandidateCost(
+ writer,
+ macroBlock,
+ sourcePlane,
+ blockOrigin,
+ above,
+ left,
+ hasLeft,
+ hasAbove,
+ bestMode,
+ selectedAngleDelta,
+ transformType,
+ blockContext,
+ candidateReconstruction,
+ candidateCoefficients,
+ ref candidateState);
+
+ if (candidateCost < bestTransformCost)
+ {
+ CopyCandidate(
+ candidateReconstruction,
+ candidateCoefficients,
+ reconstructionPlane,
+ blockOrigin,
+ retainedCoefficients,
+ TransformSize,
+ candidateState,
+ ref retainedState);
+
+ bestTransformCost = candidateCost;
+ }
+ }
+
return bestMode;
}
@@ -444,6 +500,7 @@ internal static partial class Av1IntraSuperblockEncoder
bool hasAbove,
Av1PredictionMode mode,
int angleDelta,
+ Av1TransformType transformType,
Av1TransformBlockContext blockContext,
Span candidateReconstruction,
Span candidateCoefficients,
@@ -464,7 +521,7 @@ internal static partial class Av1IntraSuperblockEncoder
angleDelta,
candidateCoefficients,
TransformSize,
- Av1TransformType.DctDct,
+ transformType,
Av1Plane.Y,
this.quantization.QIndex[0],
this.quantization.DeltaQDc[(int)Av1Plane.Y],
@@ -475,7 +532,7 @@ internal static partial class Av1IntraSuperblockEncoder
int rate = Av1TileWriter.GetLumaModeCost(writer, macroBlock, BlockSize, mode, angleDelta);
rate += writer.GetCoefficientCost(
TransformSize,
- Av1TransformType.DctDct,
+ transformType,
mode,
candidateCoefficients,
Av1ComponentType.Luminance,
diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs
index 13d1afe075..98ea4a2158 100644
--- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs
+++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs
@@ -793,6 +793,13 @@ public class Av1IntraSuperblockEncoderTests
expectedAngleDelta,
superblockWorkspace.FinalBlocks[3].PredictionUnit.AngleDelta[(int)Av1PlaneType.Y]);
+ Av1EncoderTransformBlockState targetState =
+ coefficients.GetTransformBlockSpan(0, Av1Plane.Y)[12];
+
+ // Every transform has the same skip cost for this exact-prediction target, so reference enum order
+ // requires DCT-DCT to win even when the mode-derived first pass used another transform.
+ Assert.Equal((ushort)0, targetState.EndOfBlock);
+ Assert.Equal(Av1TransformType.DctDct, targetState.TransformType);
Assert.NotEqual(0, tileWriter.GetTileData(0).Length);
}
@@ -1124,13 +1131,21 @@ public class Av1IntraSuperblockEncoderTests
initialSize: 4096);
Assert.Equal(4, coefficients.SuperblockCount);
+ bool usesNonDctTransform = false;
for (int superblockIndex = 0; superblockIndex < coefficients.SuperblockCount; superblockIndex++)
{
- Assert.NotEqual(
- (ushort)0,
- coefficients.GetTransformBlockSpan(superblockIndex, Av1Plane.Y)[0].EndOfBlock);
+ Span transformBlocks =
+ coefficients.GetTransformBlockSpan(superblockIndex, Av1Plane.Y);
+
+ Assert.NotEqual((ushort)0, transformBlocks[0].EndOfBlock);
+ foreach (Av1EncoderTransformBlockState transformBlock in transformBlocks)
+ {
+ usesNonDctTransform |=
+ transformBlock.EndOfBlock > 0 && transformBlock.TransformType != Av1TransformType.DctDct;
+ }
}
+ Assert.True(usesNonDctTransform);
Assert.NotEqual(
(byte)0,
reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y).DangerousGetRowSpan(Height - 1)[Width - 1]);
diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1SymbolContextTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1SymbolContextTests.cs
index 27af710596..b21e7c5030 100644
--- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1SymbolContextTests.cs
+++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1SymbolContextTests.cs
@@ -62,13 +62,13 @@ public class Av1SymbolContextTests
[InlineData((int)Av1PredictionMode.Paeth, (int)Av1TransformSize.Size8x8, false, (int)Av1TransformType.AdstAdst)]
[InlineData((int)Av1PredictionMode.Directional135Degrees, (int)Av1TransformSize.Size8x8, true, (int)Av1TransformType.AdstAdst)]
[InlineData((int)Av1PredictionMode.Directional135Degrees, (int)Av1TransformSize.Size32x32, false, (int)Av1TransformType.DctDct)]
- public void ImplicitIntraTransformTypeMatchesCurrentLibaom(
+ public void DefaultIntraTransformTypeMatchesCurrentLibaom(
int modeValue,
int transformSizeValue,
bool useReducedSet,
int expectedValue)
{
- Av1TransformType actual = Av1SymbolContextHelper.GetImplicitIntraTransformType(
+ Av1TransformType actual = Av1SymbolContextHelper.GetDefaultIntraTransformType(
(Av1PredictionMode)modeValue,
(Av1TransformSize)transformSizeValue,
useReducedSet);