diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 366181afa5..41fa72646b 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -872,6 +872,8 @@ Encoder verification contract: - [x] Intra-block-copy transform search now prepares motion compensation and subtraction once per plane, alternates the existing candidate and selected work buffers whenever a transform improves, and performs at most one final normalization copy into the caller-owned selected span. This matches current libaom's pointer-swap ownership without adding an allocation or a third reconstruction buffer. Inline documentation now records the scratch lifetime, strict transform tie order, skip-rate replacement, unsplit transform-root syntax, joint-plane winner retention, and final publication boundary. The focused Release encoder and intra-block-copy set passes 17 of 17 cases, the complete non-HEVC HEIF/AV1 namespace passes 9,301 of 9,301 cases with zero failures or skips, and current-main `aomdec` accepts the regenerated effort-five and effort-six intra-block-copy streams. +- [x] Uniform four-by-four luma transform search now uses two compact reconstruction and coefficient views already available in the aligned mode-decision workspace. Legal transform trials write into the non-winning view and exchange span ownership only on strict rate-distortion improvement; the selected coefficients and strided reconstruction mosaic are published once after the type search so the next raster transform sees the required decoded edge. This removes reconstruction and coefficient copies on every improving transform without adding storage, changing tie order, or repeating a transform. All four representative palette, filter-intra, and effort-eight output hashes are unchanged, the focused Release set passes 13 of 13 cases, and current-main `aomdec` accepts every checked stream. + ### 7. Write complete AVIF output - [~] The encoder-side AV1 codec configuration is now derived directly from the encoded sequence header and writes the fixed four-byte `av1C` record with empty `configOBUs`. The image payload retains the required sequence header, so the property introduces no sequence-header allocation, retention, or copy. Four production-header cases cover main, high, and professional profiles; 8-, 10-, and 12-bit precision; monochrome, 4:2:0, 4:2:2, and 4:4:4 sampling; exact fixed bytes; decoder reparsing; and header/property equivalence through direct net11 Release VSTest. Property-container emission and public AVIF activation remain open. diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs index 0847a7350a..18cbe1a46d 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs @@ -1042,13 +1042,24 @@ internal static partial class Av1IntraSuperblockEncoder Av1EncoderModeDecisionWorkspace workspace = this.blockWorkspace.GetModeDecisionWorkspace(); + // One transient 64-sample plane holds the 4x4 prediction followed by two reconstruction buffers. + // The disjoint views stay live together and need no additional owner or allocator rent. Span transformSamples = workspace.GetCandidateReconstruction(1); Span prediction = transformSamples[..TransformSampleCount]; - Span transformReconstruction = transformSamples.Slice( + Span candidateTransformReconstruction = transformSamples.Slice( + TransformSampleCount, + TransformSampleCount); + + Span bestTransformReconstruction = transformSamples.Slice( + TransformSampleCount * 2, + TransformSampleCount); + + Span transformCoefficientStorage = workspace.GetCandidateCoefficients(1); + Span candidateTransformCoefficients = transformCoefficientStorage[..TransformSampleCount]; + Span bestTransformCoefficients = transformCoefficientStorage.Slice( TransformSampleCount, TransformSampleCount); - Span transformCoefficients = workspace.GetCandidateCoefficients(1)[..TransformSampleCount]; Span residual = workspace.Residual[..TransformSampleCount]; Span contexts = workspace.TransformContexts; Span topContexts = contexts[..2]; @@ -1196,6 +1207,8 @@ internal static partial class Av1IntraSuperblockEncoder transformIndex * TransformSampleCount, TransformSampleCount); + // Each transform writes into the compact buffer that does not hold the current best. + // Swapping spans on improvement keeps the winner without copying it inside the search loop. for (Av1TransformType transformType = Av1TransformType.DctDct; transformType < Av1TransformType.AllTransformTypes; transformType++) @@ -1212,9 +1225,9 @@ internal static partial class Av1IntraSuperblockEncoder transformOrigin, prediction, residual, - transformReconstruction, + candidateTransformReconstruction, TransformWidth, - transformCoefficients, + candidateTransformCoefficients, TransformSize, transformType, Av1Plane.Y, @@ -1228,7 +1241,7 @@ internal static partial class Av1IntraSuperblockEncoder TransformSize, transformType, mode, - transformCoefficients, + candidateTransformCoefficients, Av1ComponentType.Luminance, blockContext, candidateState.EndOfBlock, @@ -1243,17 +1256,13 @@ internal static partial class Av1IntraSuperblockEncoder if (candidateCost < bestTransformCost) { - // Preserve the improving 4x4 trial in the block mosaic. Later transform predictions - // consume that reconstruction, and copying the compact result avoids another transform. - transformCoefficients.CopyTo(retainedTransformCoefficients); - for (int row = 0; row < TransformWidth; row++) - { - transformReconstruction.Slice(row * TransformWidth, TransformWidth) - .CopyTo( - candidateReconstruction.Slice( - reconstructionOffset + (row * BlockWidth), - TransformWidth)); - } + Span previousBestReconstruction = bestTransformReconstruction; + bestTransformReconstruction = candidateTransformReconstruction; + candidateTransformReconstruction = previousBestReconstruction; + + Span previousBestCoefficients = bestTransformCoefficients; + bestTransformCoefficients = candidateTransformCoefficients; + candidateTransformCoefficients = previousBestCoefficients; bestTransformCost = candidateCost; bestTransformType = transformType; @@ -1263,6 +1272,18 @@ internal static partial class Av1IntraSuperblockEncoder } } + // The next 4x4 prediction consumes this reconstruction from the block mosaic. Publish the + // final winner once, after transform search, along with its entropy-context coefficients. + bestTransformCoefficients.CopyTo(retainedTransformCoefficients); + for (int row = 0; row < TransformWidth; row++) + { + bestTransformReconstruction.Slice(row * TransformWidth, TransformWidth) + .CopyTo( + candidateReconstruction.Slice( + reconstructionOffset + (row * BlockWidth), + TransformWidth)); + } + rate += bestTransformRate; distortion += bestTransformDistortion; candidateTransformBlocks[transformIndex] = bestTransformState;