diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md
index 1b686ff87b..82eac8a281 100644
--- a/HEIF_IMPLEMENTATION_PLAN.md
+++ b/HEIF_IMPLEMENTATION_PLAN.md
@@ -825,7 +825,7 @@ Encoder verification contract:
- [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder operation now instantiates matching source and reconstruction owners with ordinary `using` lifetimes and converts packed pixels directly into the source owner before extension.
- [~] Temporal delimiter, sequence header, frame header, combined-frame tile-group writing, and an internal reduced-still-picture frame operation now exist locally. The remaining required metadata, padding, multi-tile, option, and public encoder paths are not complete.
- [~] Implement superblock and partition analysis for every permitted block size and partition. The current baseline deliberately splits every in-frame node to 8x8 blocks and records decisions in current-libaom writer preorder; block-size selection and non-split partition analysis remain.
-- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes. Palette entropy, retained state, production syntax, exhaustive luma and paired chroma palette selection, adaptive palette activation, and joint intra-block-copy mode selection are complete; adaptive intra-block-copy frame activation remains.
+- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes. Palette entropy, retained state, production syntax, exhaustive luma and paired chroma palette selection, adaptive screen-content activation, and joint intra-block-copy mode selection are complete.
- [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools.
- [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain.
- [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma, filter-intra, palette, and intra-block copy, now perform live rate-distortion selection; quality mapping, effort-dependent pruning, and the remaining searches are not implemented.
@@ -858,8 +858,8 @@ Encoder verification contract:
- [~] Live luma palette selection now follows current libaom's dominant-color and one-dimensional K-means candidate families, cache-bias threshold, sorted duplicate removal, active-edge map extension, and strict winner tie order. It improves on speed-configured libaom by evaluating both candidate families at every legal 2-through-8 size without early pruning, then exhaustively evaluates every legal transform using the existing SIMD prediction, residual, transform, quantization, and reconstruction operators. Candidate storage remains bounded stack memory; the reusable 32 KiB map owner is allocated only when an eligible block enters palette search. A production tile test proves that full 8x8 and clipped 5x3 blocks at 8 and 12 bits select exact colors and indices, extend the visible edges through coded padding, reconstruct every sample without coefficients, and emit a nonempty tile. The complete 57-case intra-superblock set, 8,931-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains gated until chroma palette mode and its rate accounting are complete.
- [~] Paired chroma palette clustering now preserves current libaom's squared two-component distance, first-centroid tie order, independently rounded U/V means, paired deterministic empty-cluster replacement, preceding-state retention on increased distortion, and 50-iteration limit. Keeping the source planes separate avoids interleave/deinterleave copies and improves on libaom's AVX2 ceiling with Vector512, Vector256, Vector128, then scalar dispatch through ImageSharp's shared vector-count helpers. Three independent tests cover exact paired convergence, midpoint initialization, 12-bit distance and index parity, untouched destination bounds, and every intrinsic tier. The exact Release test-project build reports 1,992 baseline warnings and zero errors; the focused three-case set, complete 8,934-case AVIF set, and complete 230-case HEIF set pass direct foreground net11 Release VSTest. Roslynk reports zero compiler errors and no touched-file analyzer warnings. Candidate integration and production activation remain in the open chroma-palette checkpoint.
- [~] Live paired chroma palette selection now follows current libaom's complete 2-through-8 color-size search, U-plane neighbor-cache snapping, stable U-ordered color pairs, shared U/V index map, implicit DCT-DCT transform, and strict rate-distortion winner replacement. It improves on speed-configured libaom by applying no early header-cost pruning, keeps planar U/V source data separate, and reuses the SIMD-first prediction, residual, transform, quantization, and reconstruction operators without allocator-backed candidate storage. The production tile regression proves both palette-mode probability branches, exact paired colors and indices, coefficient-free reconstruction, and nonempty syntax. The complete 58-case intra-superblock set, 8,935-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains the next checkpoint.
-- [~] Production palette activation now matches current libaom's default good-quality screen detector: it scans only complete 16x16 luma blocks, normalizes high-bit-depth samples to eight bits, admits 2-through-4-color blocks, and uses the reference's strict greater-than-ten-percent frame-area threshold. A 256-bit stack bitset and a fifth-color early exit replace libaom's larger per-block histogram without changing the decision, allocation, or source precision. The adaptive sequence flag remains enabled, the frame flag is set before picture-state allocation, and intra-block copy remains disabled. Focused regressions prove strict-threshold equality, high-bit-depth normalization, five-color rejection, emitted frame-header activation, production decode, and generated payload retention. The exact Release test-project build reports 1,992 baseline warnings and zero errors; all 8,935 AVIF cases and all 230 HEIF cases pass direct foreground net11 Release VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 30 regenerated production payloads, including the 54-byte palette case. Roslynk reports zero compiler errors and no touched-file analyzer warnings.
-- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Paths above the speed-zero screen-content threshold continue through libaom's 256-pixel, one-pixel-step exhaustive mesh. Four adjacent byte or high-bit-depth candidates share each SIMD source load, strict row-major tie ordering is retained, and the final legal tail column remains searchable where libaom's current four-wide remainder loop omits it. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. Production mode decision now derives the same spatial displacement reference used by the writer, deduplicates hash and full-pixel finalists in search order, and evaluates every surviving vector through complete luma and chroma transform RD. This intentionally improves on libaom's preliminary-error pruning by permitting a hash and pixel finalist from the same search region to compete using their final syntax and reconstruction costs. Prediction is prepared once per plane and vector, including integer or half-sample chroma phase, then reused across every legal inter transform without an allocator rent or frame copy. The joint comparison includes the live intra-block-copy flag, discounted displacement rate, skip flag, coefficient syntax, and normalized Y/U/V distortion; an empty transform alternative can win only when its complete skip cost is strictly lower, while conventional intra and earlier vectors retain tie precedence. Winning reconstruction, coefficients, transform state, DC modes, cleared palette/filter/CfL state, and displacement are copied once into the existing retained stores. Production regressions force the path at 8 and 12 bits and force 4:2:0 horizontal half-sample chroma with an unaligned reference. The net11 Release solution build reports zero errors; all 2,030 focused entropy, intra-block-copy, and intra-superblock cases and all 9,240 non-HEVC HEIF, AV1, and AVIF cases pass through direct foreground VSTest, with tiered compilation disabled only for the full allocation-sensitive suite. Adaptive production activation and frame-flag clearing remain before intra-block copy can be enabled by the public encoder.
+- [x] Production screen-content activation now matches current libaom's default good-quality detector: it scans only complete 16x16 luma blocks, normalizes palette samples to eight bits, admits 2-through-4-color blocks, and uses the reference's strict greater-than-ten-percent frame-area threshold. The same pass accumulates centered sums and squared sums at native precision, applies libaom's exact 10-bit and 12-bit variance rounding, and enables intra-block copy only when positive rounded per-pixel variance exceeds its strict one-twelfth frame-area threshold. A 256-bit stack bitset and fifth-color early exit replace libaom's larger per-block histogram without a second source scan or allocation. The adaptive sequence flag remains enabled and both frame flags are fixed before picture-state allocation. Focused regressions prove strict palette-threshold equality, high-bit-depth normalization, the exact variance rounding boundary, five-color rejection, emitted frame-header activation, actual production IBC selection, and production decode. The exact Release test-project build reports 1,992 baseline warnings and zero errors; all 9,242 non-HEVC HEIF/AV1 cases pass direct foreground net11 Release VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 31 payloads regenerated by the current test tree, including an actual IBC-coded 328x16 stream with decoded MD5 `677435e5af39c930af1178f91c34af6a`. Roslynk reports zero compiler errors and no touched-file analyzer warnings.
+- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Paths above the speed-zero screen-content threshold continue through libaom's 256-pixel, one-pixel-step exhaustive mesh. Four adjacent byte or high-bit-depth candidates share each SIMD source load, strict row-major tie ordering is retained, and the final legal tail column remains searchable where libaom's current four-wide remainder loop omits it. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. Production mode decision now derives the same spatial displacement reference used by the writer, deduplicates hash and full-pixel finalists in search order, and evaluates every surviving vector through complete luma and chroma transform RD. This intentionally improves on libaom's preliminary-error pruning by permitting a hash and pixel finalist from the same search region to compete using their final syntax and reconstruction costs. Prediction is prepared once per plane and vector, including integer or half-sample chroma phase, then reused across every legal inter transform without an allocator rent or frame copy. The joint comparison includes the live intra-block-copy flag, discounted displacement rate, skip flag, coefficient syntax, and normalized Y/U/V distortion; an empty transform alternative can win only when its complete skip cost is strictly lower, while conventional intra and earlier vectors retain tie precedence. Winning reconstruction, coefficients, transform state, DC modes, cleared palette/filter/CfL state, and displacement are copied once into the existing retained stores. Production regressions force the path at 8 and 12 bits and force 4:2:0 horizontal half-sample chroma with an unaligned reference. The former bulk local workspace occupied 2.75 KiB for byte samples or 3.375 KiB for high-bit-depth samples. Prediction, candidate, winning reconstruction, residual, and coefficient scratch now occupy one naturally aligned 3.125 KiB extension of the existing frame-reused block-workspace owner, matching libaom's reusable macroblock-scratch lifetime without adding an allocation; only the 128-byte reference, weight, and finalist arrays remain on the stack. The net11 Release solution build reports zero errors; all 2,082 focused transform, entropy, intra-block-copy, intra-superblock, and frame-encoder cases and all 9,242 non-HEVC HEIF/AV1 cases pass through direct foreground VSTest, with tiered compilation disabled only for the full allocation-sensitive suite. Adaptive production activation is complete, and the emitted frame flag remains authoritative for the complete frame rather than being invalidated after tile coding.
- [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips.
- [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips.
- [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips.
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderBlockWorkspace.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderBlockWorkspace.cs
index bd502f7347..d65fe81056 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderBlockWorkspace.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderBlockWorkspace.cs
@@ -26,12 +26,40 @@ internal sealed class Av1EncoderBlockWorkspace : IDisposable
///
/// The complete workspace length in signed-integer storage elements.
///
- public const int StorageLength = ResidualStorageLength + MaximumCoefficientCount + MaximumCoefficientCount + Av1TransformWorkspace.MaximumLength;
+ public const int StorageLength =
+ ResidualStorageLength +
+ MaximumCoefficientCount +
+ MaximumCoefficientCount +
+ Av1TransformWorkspace.MaximumLength +
+ IntraBlockCopySampleStorageLength +
+ IntraBlockCopyResidualStorageLength +
+ IntraBlockCopyCoefficientStorageLength;
private const int ResidualStorageLength = MaximumResidualCount / 2;
private const int TransformCoefficientOffset = ResidualStorageLength;
private const int DequantizedCoefficientOffset = TransformCoefficientOffset + MaximumCoefficientCount;
private const int TransformWorkspaceOffset = DequantizedCoefficientOffset + MaximumCoefficientCount;
+ private const int IntraBlockCopySampleStorageOffset = TransformWorkspaceOffset + Av1TransformWorkspace.MaximumLength;
+ private const int IntraBlockCopySampleStorageLength =
+ Av1EncoderIntraBlockCopyWorkspace.SampleBufferCount *
+ Av1EncoderIntraBlockCopyWorkspace.MaximumSampleCount *
+ sizeof(ushort) /
+ sizeof(int);
+
+ private const int IntraBlockCopyResidualStorageOffset =
+ IntraBlockCopySampleStorageOffset + IntraBlockCopySampleStorageLength;
+
+ private const int IntraBlockCopyResidualStorageLength =
+ Av1EncoderIntraBlockCopyWorkspace.MaximumSampleCount *
+ sizeof(short) /
+ sizeof(int);
+
+ private const int IntraBlockCopyCoefficientStorageOffset =
+ IntraBlockCopyResidualStorageOffset + IntraBlockCopyResidualStorageLength;
+
+ private const int IntraBlockCopyCoefficientStorageLength =
+ Av1EncoderIntraBlockCopyWorkspace.CoefficientBufferCount *
+ Av1EncoderIntraBlockCopyWorkspace.MaximumSampleCount;
///
/// Owns the complete reusable block workspace in 32-bit elements so every transform region is naturally aligned.
@@ -69,6 +97,37 @@ internal sealed class Av1EncoderBlockWorkspace : IDisposable
public Span TransformWorkspace
=> this.owner.Memory.Span.Slice(TransformWorkspaceOffset, Av1TransformWorkspace.MaximumLength);
+ ///
+ /// Gets the reusable storage used while comparing intra-block-copy candidates.
+ ///
+ /// The native sample type selected by the encoder pipeline.
+ /// The typed intra-block-copy workspace.
+ public Av1EncoderIntraBlockCopyWorkspace GetIntraBlockCopyWorkspace()
+ where TSample : unmanaged
+ {
+ Span storage = this.owner.Memory.Span;
+ Span sampleStorage = MemoryMarshal
+ .Cast(storage.Slice(IntraBlockCopySampleStorageOffset, IntraBlockCopySampleStorageLength));
+
+ sampleStorage = sampleStorage[
+ ..(Av1EncoderIntraBlockCopyWorkspace.SampleBufferCount *
+ Av1EncoderIntraBlockCopyWorkspace.MaximumSampleCount)];
+
+ Span residualStorage = MemoryMarshal
+ .Cast(storage.Slice(IntraBlockCopyResidualStorageOffset, IntraBlockCopyResidualStorageLength));
+
+ residualStorage = residualStorage[..Av1EncoderIntraBlockCopyWorkspace.MaximumSampleCount];
+
+ Span coefficientStorage = storage.Slice(
+ IntraBlockCopyCoefficientStorageOffset,
+ IntraBlockCopyCoefficientStorageLength);
+
+ return new Av1EncoderIntraBlockCopyWorkspace(
+ sampleStorage,
+ residualStorage,
+ coefficientStorage);
+ }
+
///
/// Releases the reusable block workspace.
///
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderIntraBlockCopyWorkspace.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderIntraBlockCopyWorkspace.cs
new file mode 100644
index 0000000000..565a7bc008
--- /dev/null
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderIntraBlockCopyWorkspace.cs
@@ -0,0 +1,143 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
+
+///
+/// Provides disjoint reusable buffers for intra-block-copy mode decisions.
+///
+/// The native sample type selected by the encoder pipeline.
+internal readonly ref struct Av1EncoderIntraBlockCopyWorkspace
+ where TSample : unmanaged
+{
+ ///
+ /// The number of samples in the fixed 8x8 intra-block-copy transform.
+ ///
+ public const int MaximumSampleCount = 8 * 8;
+
+ ///
+ /// The number of sample buffers retained by one mode decision.
+ ///
+ public const int SampleBufferCount = 10;
+
+ ///
+ /// The number of coefficient buffers retained by one mode decision.
+ ///
+ public const int CoefficientBufferCount = 7;
+
+ private readonly Span samples;
+ private readonly Span residual;
+ private readonly Span coefficients;
+
+ ///
+ /// Initializes a new instance of the struct.
+ ///
+ /// The sample storage.
+ /// The residual storage shared by sequential plane evaluations.
+ /// The coefficient storage.
+ public Av1EncoderIntraBlockCopyWorkspace(
+ Span samples,
+ Span residual,
+ Span coefficients)
+ {
+ this.samples = samples;
+ this.residual = residual;
+ this.coefficients = coefficients;
+ }
+
+ ///
+ /// Gets the selected luma reconstruction.
+ ///
+ public Span SelectedLumaReconstruction => this.GetSamples(0);
+
+ ///
+ /// Gets the selected blue-difference chroma reconstruction.
+ ///
+ public Span SelectedBlueReconstruction => this.GetSamples(1);
+
+ ///
+ /// Gets the selected red-difference chroma reconstruction.
+ ///
+ public Span SelectedRedReconstruction => this.GetSamples(2);
+
+ ///
+ /// Gets the current luma prediction.
+ ///
+ public Span LumaPrediction => this.GetSamples(3);
+
+ ///
+ /// Gets the current blue-difference chroma prediction.
+ ///
+ public Span BluePrediction => this.GetSamples(4);
+
+ ///
+ /// Gets the current red-difference chroma prediction.
+ ///
+ public Span RedPrediction => this.GetSamples(5);
+
+ ///
+ /// Gets the current luma candidate reconstruction.
+ ///
+ public Span LumaCandidateReconstruction => this.GetSamples(6);
+
+ ///
+ /// Gets the current blue-difference chroma candidate reconstruction.
+ ///
+ public Span BlueCandidateReconstruction => this.GetSamples(7);
+
+ ///
+ /// Gets the current red-difference chroma candidate reconstruction.
+ ///
+ public Span RedCandidateReconstruction => this.GetSamples(8);
+
+ ///
+ /// Gets the reconstruction scratch overwritten by each transform trial.
+ ///
+ public Span TransformReconstruction => this.GetSamples(9);
+
+ ///
+ /// Gets the residual scratch shared by sequential plane evaluations.
+ ///
+ public Span Residual => this.residual;
+
+ ///
+ /// Gets the selected luma coefficients.
+ ///
+ public Span SelectedLumaCoefficients => this.GetCoefficients(0);
+
+ ///
+ /// Gets the selected blue-difference chroma coefficients.
+ ///
+ public Span SelectedBlueCoefficients => this.GetCoefficients(1);
+
+ ///
+ /// Gets the selected red-difference chroma coefficients.
+ ///
+ public Span SelectedRedCoefficients => this.GetCoefficients(2);
+
+ ///
+ /// Gets the current luma candidate coefficients.
+ ///
+ public Span LumaCandidateCoefficients => this.GetCoefficients(3);
+
+ ///
+ /// Gets the current blue-difference chroma candidate coefficients.
+ ///
+ public Span BlueCandidateCoefficients => this.GetCoefficients(4);
+
+ ///
+ /// Gets the current red-difference chroma candidate coefficients.
+ ///
+ public Span RedCandidateCoefficients => this.GetCoefficients(5);
+
+ ///
+ /// Gets the coefficient scratch overwritten by each transform trial.
+ ///
+ public Span TransformCoefficients => this.GetCoefficients(6);
+
+ private Span GetSamples(int index)
+ => this.samples.Slice(index * MaximumSampleCount, MaximumSampleCount);
+
+ private Span GetCoefficients(int index)
+ => this.coefficients.Slice(index * MaximumSampleCount, MaximumSampleCount);
+}
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs
index 6e3630c52a..6854c7bb5e 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs
@@ -235,8 +235,13 @@ internal static class Av1FrameEncoder
where TPixel : unmanaged, IPixel
{
PrepareSource(configuration, image, source.Frame, sequenceHeader.ColorConfig);
- frameHeader.AllowScreenContentTools = Av1ScreenContentDetector.IsPaletteLikely(source.Frame);
+ Av1ScreenContentDetector.Detect(
+ source.Frame,
+ out bool allowScreenContentTools,
+ out bool allowIntraBlockCopy);
+ frameHeader.AllowScreenContentTools = allowScreenContentTools;
+ frameHeader.AllowIntraBlockCopy = allowIntraBlockCopy;
using Av1EncoderPictureBuffer picture = new(
configuration,
sequenceHeader,
@@ -278,8 +283,13 @@ internal static class Av1FrameEncoder
where TPixel : unmanaged, IPixel
{
PrepareSource(configuration, image, source.Frame, sequenceHeader.ColorConfig);
- frameHeader.AllowScreenContentTools = Av1ScreenContentDetector.IsPaletteLikely(source.Frame);
+ Av1ScreenContentDetector.Detect(
+ source.Frame,
+ out bool allowScreenContentTools,
+ out bool allowIntraBlockCopy);
+ frameHeader.AllowScreenContentTools = allowScreenContentTools;
+ frameHeader.AllowIntraBlockCopy = allowIntraBlockCopy;
using Av1EncoderPictureBuffer picture = new(
configuration,
sequenceHeader,
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.IntraBlockCopyModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.IntraBlockCopyModeDecision.cs
index 88c5676238..1b34df4dcb 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.IntraBlockCopyModeDecision.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.IntraBlockCopyModeDecision.cs
@@ -33,7 +33,6 @@ internal static partial class Av1IntraSuperblockEncoder
{
const Av1BlockSize BlockSize = Av1BlockSize.Block8x8;
const Av1TransformSize LumaTransformSize = Av1TransformSize.Size8x8;
- const int MaximumSampleCount = 8 * 8;
Buffer2DRegion lumaSource = this.source.GetPlane(Av1Plane.Y);
Buffer2DRegion lumaReconstruction = this.reconstruction.GetPlane(Av1Plane.Y);
Point modeInfoPosition = new(
@@ -119,27 +118,8 @@ internal static partial class Av1IntraSuperblockEncoder
Av1EncoderTransformBlockState selectedLumaState = default;
Av1EncoderTransformBlockState selectedBlueState = default;
Av1EncoderTransformBlockState selectedRedState = default;
- Span selectedLumaReconstruction = stackalloc TSample[MaximumSampleCount];
- Span selectedBlueReconstruction = stackalloc TSample[MaximumSampleCount];
- Span selectedRedReconstruction = stackalloc TSample[MaximumSampleCount];
- Span selectedLumaCoefficients = stackalloc int[MaximumSampleCount];
- Span selectedBlueCoefficients = stackalloc int[MaximumSampleCount];
- Span selectedRedCoefficients = stackalloc int[MaximumSampleCount];
-
- Span lumaPrediction = stackalloc TSample[MaximumSampleCount];
- Span lumaResidual = stackalloc short[MaximumSampleCount];
- Span lumaCandidateReconstruction = stackalloc TSample[MaximumSampleCount];
- Span lumaCandidateCoefficients = stackalloc int[MaximumSampleCount];
- Span bluePrediction = stackalloc TSample[MaximumSampleCount];
- Span blueResidual = stackalloc short[MaximumSampleCount];
- Span blueCandidateReconstruction = stackalloc TSample[MaximumSampleCount];
- Span blueCandidateCoefficients = stackalloc int[MaximumSampleCount];
- Span redPrediction = stackalloc TSample[MaximumSampleCount];
- Span redResidual = stackalloc short[MaximumSampleCount];
- Span redCandidateReconstruction = stackalloc TSample[MaximumSampleCount];
- Span redCandidateCoefficients = stackalloc int[MaximumSampleCount];
- Span transformReconstruction = stackalloc TSample[MaximumSampleCount];
- Span transformCoefficients = stackalloc int[MaximumSampleCount];
+ Av1EncoderIntraBlockCopyWorkspace workspace =
+ this.blockWorkspace.GetIntraBlockCopyWorkspace();
Av1TransformBlockContext lumaContext = Av1TileWriter.GetTransformBlockContexts(
Av1ComponentType.Luminance,
@@ -192,12 +172,12 @@ internal static partial class Av1IntraSuperblockEncoder
0,
LumaTransformSize,
lumaContext,
- lumaPrediction,
- lumaResidual,
- transformReconstruction,
- transformCoefficients,
- lumaCandidateReconstruction,
- lumaCandidateCoefficients,
+ workspace.LumaPrediction,
+ workspace.Residual,
+ workspace.TransformReconstruction,
+ workspace.TransformCoefficients,
+ workspace.LumaCandidateReconstruction,
+ workspace.LumaCandidateCoefficients,
out Av1EncoderTransformBlockState lumaCandidateState,
out int lumaRate,
out long lumaDistortion,
@@ -229,12 +209,12 @@ internal static partial class Av1IntraSuperblockEncoder
subsamplingY,
chromaTransformSize,
blueContext,
- bluePrediction,
- blueResidual,
- transformReconstruction,
- transformCoefficients,
- blueCandidateReconstruction,
- blueCandidateCoefficients,
+ workspace.BluePrediction,
+ workspace.Residual,
+ workspace.TransformReconstruction,
+ workspace.TransformCoefficients,
+ workspace.BlueCandidateReconstruction,
+ workspace.BlueCandidateCoefficients,
out blueCandidateState,
out blueRate,
out blueDistortion,
@@ -252,12 +232,12 @@ internal static partial class Av1IntraSuperblockEncoder
subsamplingY,
chromaTransformSize,
redContext,
- redPrediction,
- redResidual,
- transformReconstruction,
- transformCoefficients,
- redCandidateReconstruction,
- redCandidateCoefficients,
+ workspace.RedPrediction,
+ workspace.Residual,
+ workspace.TransformReconstruction,
+ workspace.TransformCoefficients,
+ workspace.RedCandidateReconstruction,
+ workspace.RedCandidateCoefficients,
out redCandidateState,
out redRate,
out redDistortion,
@@ -304,32 +284,40 @@ internal static partial class Av1IntraSuperblockEncoder
selectedVector = candidate;
if (candidateSkip)
{
- lumaPrediction.CopyTo(selectedLumaReconstruction);
- selectedLumaCoefficients.Clear();
+ workspace.LumaPrediction.CopyTo(workspace.SelectedLumaReconstruction);
+ workspace.SelectedLumaCoefficients.Clear();
selectedLumaState = emptyLumaState;
if (!this.source.IsMonochrome)
{
int chromaSampleCount = chromaTransformSize.GetSize2d();
- bluePrediction[..chromaSampleCount].CopyTo(selectedBlueReconstruction);
- redPrediction[..chromaSampleCount].CopyTo(selectedRedReconstruction);
- selectedBlueCoefficients[..chromaSampleCount].Clear();
- selectedRedCoefficients[..chromaSampleCount].Clear();
+ workspace.BluePrediction[..chromaSampleCount].CopyTo(workspace.SelectedBlueReconstruction);
+ workspace.RedPrediction[..chromaSampleCount].CopyTo(workspace.SelectedRedReconstruction);
+ workspace.SelectedBlueCoefficients[..chromaSampleCount].Clear();
+ workspace.SelectedRedCoefficients[..chromaSampleCount].Clear();
selectedBlueState = emptyBlueState;
selectedRedState = emptyRedState;
}
}
else
{
- lumaCandidateReconstruction.CopyTo(selectedLumaReconstruction);
- lumaCandidateCoefficients.CopyTo(selectedLumaCoefficients);
+ workspace.LumaCandidateReconstruction.CopyTo(workspace.SelectedLumaReconstruction);
+ workspace.LumaCandidateCoefficients.CopyTo(workspace.SelectedLumaCoefficients);
selectedLumaState = lumaCandidateState;
if (!this.source.IsMonochrome)
{
int chromaSampleCount = chromaTransformSize.GetSize2d();
- blueCandidateReconstruction[..chromaSampleCount].CopyTo(selectedBlueReconstruction);
- redCandidateReconstruction[..chromaSampleCount].CopyTo(selectedRedReconstruction);
- blueCandidateCoefficients[..chromaSampleCount].CopyTo(selectedBlueCoefficients);
- redCandidateCoefficients[..chromaSampleCount].CopyTo(selectedRedCoefficients);
+ workspace.BlueCandidateReconstruction[..chromaSampleCount]
+ .CopyTo(workspace.SelectedBlueReconstruction);
+
+ workspace.RedCandidateReconstruction[..chromaSampleCount]
+ .CopyTo(workspace.SelectedRedReconstruction);
+
+ workspace.BlueCandidateCoefficients[..chromaSampleCount]
+ .CopyTo(workspace.SelectedBlueCoefficients);
+
+ workspace.RedCandidateCoefficients[..chromaSampleCount]
+ .CopyTo(workspace.SelectedRedCoefficients);
+
selectedBlueState = blueCandidateState;
selectedRedState = redCandidateState;
}
@@ -350,8 +338,8 @@ internal static partial class Av1IntraSuperblockEncoder
ref Av1EncoderTransformBlockState retainedLumaState = ref retainedLumaTransformBlocks[lumaTransformIndex];
CopyCandidate(
- selectedLumaReconstruction,
- selectedLumaCoefficients,
+ workspace.SelectedLumaReconstruction,
+ workspace.SelectedLumaCoefficients,
lumaReconstruction,
blockOrigin,
retainedLumaCoefficients[this.codedAreaLuma..],
@@ -375,8 +363,8 @@ internal static partial class Av1IntraSuperblockEncoder
ref Av1EncoderTransformBlockState retainedBlueState = ref retainedBlueTransformBlocks[chromaTransformIndex];
ref Av1EncoderTransformBlockState retainedRedState = ref retainedRedTransformBlocks[chromaTransformIndex];
CopyCandidate(
- selectedBlueReconstruction,
- selectedBlueCoefficients,
+ workspace.SelectedBlueReconstruction,
+ workspace.SelectedBlueCoefficients,
this.reconstruction.GetPlane(Av1Plane.U),
chromaOrigin,
retainedBlueCoefficients[this.codedAreaChroma..],
@@ -385,8 +373,8 @@ internal static partial class Av1IntraSuperblockEncoder
ref retainedBlueState);
CopyCandidate(
- selectedRedReconstruction,
- selectedRedCoefficients,
+ workspace.SelectedRedReconstruction,
+ workspace.SelectedRedCoefficients,
this.reconstruction.GetPlane(Av1Plane.V),
chromaOrigin,
retainedRedCoefficients[this.codedAreaChroma..],
diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1ScreenContentDetector.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1ScreenContentDetector.cs
index 919f5557a7..52683b8dca 100644
--- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1ScreenContentDetector.cs
+++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1ScreenContentDetector.cs
@@ -20,12 +20,11 @@ internal static class Av1ScreenContentDetector
where TSample : unmanaged
{
///
- /// Converts one native sample to eight-bit precision.
+ /// Converts one native sample to an integer without changing its precision.
///
/// The source sample.
- /// The number of low bits removed from high-bit-depth samples.
- /// The normalized sample.
- public static abstract int ToEightBit(TSample value, int bitDepthShift);
+ /// The native sample value.
+ public static abstract int ToInt32(TSample value);
}
///
@@ -34,7 +33,10 @@ internal static class Av1ScreenContentDetector
/// The converted source frame.
/// when palette tools should be enabled; otherwise, .
public static bool IsPaletteLikely(Av1EncoderFrame source)
- => IsPaletteLikely(source);
+ {
+ Detect(source, out bool allowScreenContentTools, out _);
+ return allowScreenContentTools;
+ }
///
/// Detects palette-friendly content in a high-bit-depth source frame.
@@ -42,9 +44,39 @@ internal static class Av1ScreenContentDetector
/// The converted source frame.
/// when palette tools should be enabled; otherwise, .
public static bool IsPaletteLikely(Av1EncoderFrame source)
- => IsPaletteLikely(source);
+ {
+ Detect(source, out bool allowScreenContentTools, out _);
+ return allowScreenContentTools;
+ }
- private static bool IsPaletteLikely(Av1EncoderFrame source)
+ ///
+ /// Detects palette and intra-block-copy content in an eight-bit source frame.
+ ///
+ /// The converted source frame.
+ /// Receives whether palette syntax should be enabled.
+ /// Receives whether intra-block copy should be enabled.
+ public static void Detect(
+ Av1EncoderFrame source,
+ out bool allowScreenContentTools,
+ out bool allowIntraBlockCopy)
+ => Detect(source, out allowScreenContentTools, out allowIntraBlockCopy);
+
+ ///
+ /// Detects palette and intra-block-copy content in a high-bit-depth source frame.
+ ///
+ /// The converted source frame.
+ /// Receives whether palette syntax should be enabled.
+ /// Receives whether intra-block copy should be enabled.
+ public static void Detect(
+ Av1EncoderFrame source,
+ out bool allowScreenContentTools,
+ out bool allowIntraBlockCopy)
+ => Detect(source, out allowScreenContentTools, out allowIntraBlockCopy);
+
+ private static void Detect(
+ Av1EncoderFrame source,
+ out bool allowScreenContentTools,
+ out bool allowIntraBlockCopy)
where TSample : unmanaged
where TOperator : struct, ISampleOperator
{
@@ -54,7 +86,10 @@ internal static class Av1ScreenContentDetector
long frameArea = (long)width * height;
int bitDepthShift = source.LumaBitDepth - 8;
int paletteBlockCount = 0;
+ int intraBlockCopyBlockCount = 0;
Span seenColors = stackalloc ulong[4];
+ allowScreenContentTools = false;
+ allowIntraBlockCopy = false;
// Complete 16x16 blocks and the strict frame-area threshold preserve the reference detector's decision.
for (int blockRow = 0; blockRow + DetectionBlockLength <= height; blockRow += DetectionBlockLength)
@@ -63,6 +98,8 @@ internal static class Av1ScreenContentDetector
{
seenColors.Clear();
int colorCount = 0;
+ long sum = 0;
+ long sumOfSquares = 0;
for (int row = 0; row < DetectionBlockLength && colorCount <= MaximumPaletteColorCount; row++)
{
ReadOnlySpan samples = view
@@ -72,10 +109,14 @@ internal static class Av1ScreenContentDetector
// Histogram updates depend on each sample value, so a compact scalar bitset avoids gather/scatter overhead.
for (int column = 0; column < samples.Length; column++)
{
- int value = TOperator.ToEightBit(samples[column], bitDepthShift);
+ int nativeValue = TOperator.ToInt32(samples[column]);
+ int value = nativeValue >> bitDepthShift;
int wordIndex = value >> 6;
ulong mask = 1UL << (value & 63);
ref ulong word = ref seenColors[wordIndex];
+ int centeredValue = nativeValue - (128 << bitDepthShift);
+ sum += centeredValue;
+ sumOfSquares += (long)centeredValue * centeredValue;
if ((word & mask) == 0)
{
word |= mask;
@@ -91,24 +132,43 @@ internal static class Av1ScreenContentDetector
if (colorCount > 1 && colorCount <= MaximumPaletteColorCount)
{
paletteBlockCount++;
- if ((long)paletteBlockCount * DetectionBlockArea * 10 > frameArea)
+ long normalizedSum = sum;
+ long normalizedSumOfSquares = sumOfSquares;
+ if (bitDepthShift != 0)
+ {
+ normalizedSum = RoundPowerOfTwo(sum, bitDepthShift);
+ normalizedSumOfSquares = RoundPowerOfTwo(sumOfSquares, bitDepthShift * 2);
+ }
+
+ long variance = normalizedSumOfSquares - ((normalizedSum * normalizedSum) >> 8);
+ if (variance >= DetectionBlockArea / 2)
+ {
+ intraBlockCopyBlockCount++;
+ }
+
+ allowScreenContentTools = (long)paletteBlockCount * DetectionBlockArea * 10 > frameArea;
+ allowIntraBlockCopy = allowScreenContentTools &&
+ (long)intraBlockCopyBlockCount * DetectionBlockArea * 12 > frameArea;
+
+ if (allowIntraBlockCopy)
{
- return true;
+ return;
}
}
}
}
-
- return false;
}
+ private static long RoundPowerOfTwo(long value, int shift)
+ => (value + (1L << (shift - 1))) >> shift;
+
///
/// Preserves native eight-bit samples.
///
private readonly struct ByteSampleOperator : ISampleOperator
{
///
- public static int ToEightBit(byte value, int bitDepthShift) => value;
+ public static int ToInt32(byte value) => value;
}
///
@@ -117,6 +177,6 @@ internal static class Av1ScreenContentDetector
private readonly struct UShortSampleOperator : ISampleOperator
{
///
- public static int ToEightBit(ushort value, int bitDepthShift) => value >> bitDepthShift;
+ public static int ToInt32(ushort value) => value;
}
}
diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs
index 5757dba8f5..7c7dd07483 100644
--- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs
+++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs
@@ -4,6 +4,7 @@
using SixLabors.ImageSharp.Formats.Heif.Av1;
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit;
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
+using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling;
using SixLabors.ImageSharp.Memory;
using SixLabors.ImageSharp.PixelFormats;
using SixLabors.ImageSharp.Tests.Memory;
@@ -147,7 +148,7 @@ public class Av1EncoderFrameTests
}
[Fact]
- public void ScreenContentDetectorMatchesLibaomPaletteThreshold()
+ public void ScreenContentDetectorMatchesLibaomFeatureThresholds()
{
const int width = 160;
const int height = 16;
@@ -172,6 +173,13 @@ public class Av1EncoderFrameTests
// One qualifying block is exactly ten percent of this frame, and the reference threshold is strict.
Assert.False(Av1ScreenContentDetector.IsPaletteLikely(byteFrame.Frame));
+ Av1ScreenContentDetector.Detect(
+ byteFrame.Frame,
+ out bool allowScreenContentTools,
+ out bool allowIntraBlockCopy);
+
+ Assert.False(allowScreenContentTools);
+ Assert.False(allowIntraBlockCopy);
for (int row = 0; row < height; row++)
{
Span samples = byteFrame.Frame.View.GetLumaRowSpan(row);
@@ -182,6 +190,13 @@ public class Av1EncoderFrameTests
}
Assert.True(Av1ScreenContentDetector.IsPaletteLikely(byteFrame.Frame));
+ Av1ScreenContentDetector.Detect(
+ byteFrame.Frame,
+ out allowScreenContentTools,
+ out allowIntraBlockCopy);
+
+ Assert.True(allowScreenContentTools);
+ Assert.True(allowIntraBlockCopy);
using Av1EncoderFrameBuffer highBitDepthFrame = new(
Configuration.Default,
16,
@@ -201,6 +216,13 @@ public class Av1EncoderFrameTests
}
Assert.False(Av1ScreenContentDetector.IsPaletteLikely(highBitDepthFrame.Frame));
+ Av1ScreenContentDetector.Detect(
+ highBitDepthFrame.Frame,
+ out allowScreenContentTools,
+ out allowIntraBlockCopy);
+
+ Assert.False(allowScreenContentTools);
+ Assert.False(allowIntraBlockCopy);
for (int row = 0; row < 16; row++)
{
Span samples = highBitDepthFrame.Frame.View.GetLumaRowSpan(row);
@@ -208,6 +230,13 @@ public class Av1EncoderFrameTests
}
Assert.True(Av1ScreenContentDetector.IsPaletteLikely(highBitDepthFrame.Frame));
+ Av1ScreenContentDetector.Detect(
+ highBitDepthFrame.Frame,
+ out allowScreenContentTools,
+ out allowIntraBlockCopy);
+
+ Assert.True(allowScreenContentTools);
+ Assert.True(allowIntraBlockCopy);
for (int row = 0; row < 16; row++)
{
Span samples = highBitDepthFrame.Frame.View.GetLumaRowSpan(row);
@@ -218,10 +247,58 @@ public class Av1EncoderFrameTests
}
Assert.False(Av1ScreenContentDetector.IsPaletteLikely(highBitDepthFrame.Frame));
+ Av1ScreenContentDetector.Detect(
+ highBitDepthFrame.Frame,
+ out allowScreenContentTools,
+ out allowIntraBlockCopy);
+
+ Assert.False(allowScreenContentTools);
+ Assert.False(allowIntraBlockCopy);
}
[Fact]
- public void EncodeActivatesPaletteToolsForScreenContent()
+ public void ScreenContentDetectorMatchesLibaomIntraBlockCopyVarianceThreshold()
+ {
+ const int Width = 16;
+ const int Height = 16;
+ using Av1EncoderFrameBuffer frame = new(
+ Configuration.Default,
+ Width,
+ Height,
+ 8,
+ Av1ColorFormat.Yuv400,
+ 0,
+ 0);
+
+ Buffer2DRegion luma = frame.Frame.View.GetPlane(Av1Plane.Y);
+ for (int row = 0; row < Height; row++)
+ {
+ luma.DangerousGetRowSpan(row).Fill(96);
+ }
+
+ // A single delta of eleven leaves total variance below half a sample after per-pixel rounding.
+ luma.DangerousGetRowSpan(0)[0] = 107;
+ Av1ScreenContentDetector.Detect(
+ frame.Frame,
+ out bool allowScreenContentTools,
+ out bool allowIntraBlockCopy);
+
+ Assert.True(allowScreenContentTools);
+ Assert.False(allowIntraBlockCopy);
+
+ // Raising that delta to twelve crosses the exact integer rounding boundary used by libaom.
+ luma.DangerousGetRowSpan(0)[0] = 108;
+ Av1ScreenContentDetector.Detect(
+ frame.Frame,
+ out allowScreenContentTools,
+ out allowIntraBlockCopy);
+
+ Assert.True(allowScreenContentTools);
+ Assert.True(allowIntraBlockCopy);
+ }
+
+ [Fact]
+ public void EncodeActivatesScreenContentTools()
{
const int width = 16;
const int height = 16;
@@ -253,7 +330,7 @@ public class Av1EncoderFrameTests
obuReader.ReadAll(ref reader, payload.Length, () => tileReader);
ObuFrameHeader frameHeader = Assert.IsType(obuReader.FrameHeader);
Assert.True(frameHeader.AllowScreenContentTools);
- Assert.False(frameHeader.AllowIntraBlockCopy);
+ Assert.True(frameHeader.AllowIntraBlockCopy);
using Av1Decoder decoder = new(Configuration.Default);
using Image decoded = decoder.Decode(payload);
Assert.Equal(new Size(width, height), decoded.Size);
@@ -268,6 +345,61 @@ public class Av1EncoderFrameTests
File.WriteAllBytes(Path.Combine(outputDirectory, "encoder-frame-16x16-8b-444-palette.obu"), payload);
}
+ [Fact]
+ public void EncodeSelectsIntraBlockCopyForRepeatedScreenContent()
+ {
+ const int Width = 328;
+ const int Height = 16;
+ const ulong Pattern = 0xD6A5_3C97_E18B_4F20UL;
+ using Image source = new(Width, Height);
+ for (int row = 0; row < Height; row++)
+ {
+ Span pixels = source.Frames.RootFrame.PixelBuffer.DangerousGetRowSpan(row);
+ for (int column = 0; column < Width; column++)
+ {
+ int patternIndex = ((row & 7) * 8) + (column & 7);
+ pixels[column] = ((Pattern >> patternIndex) & 1) == 0
+ ? new Rgba32(224, 32, 32)
+ : new Rgba32(32, 32, 224);
+ }
+ }
+
+ ObuColorConfig colorConfig = CreateColorConfig(Av1BitDepth.EightBit, Av1ColorFormat.Yuv444);
+ using MemoryStream stream = new();
+ _ = Av1FrameEncoder.Encode(
+ Configuration.Default,
+ source.Frames.RootFrame,
+ stream,
+ colorConfig,
+ qIndex: 37);
+
+ byte[] payload = stream.ToArray();
+ using Av1Decoder decoder = new(Configuration.Default);
+ using Image decoded = decoder.Decode(payload);
+ Assert.NotNull(decoder.FrameHeader);
+ Assert.True(decoder.FrameHeader.AllowScreenContentTools);
+ Assert.True(decoder.FrameHeader.AllowIntraBlockCopy);
+ Assert.NotNull(decoder.FrameInfo);
+ Av1SuperblockInfo targetSuperblock = decoder.FrameInfo.GetSuperblock(new Point(5, 0));
+ bool usesIntraBlockCopy = false;
+ foreach (Av1BlockModeInfo modeInfo in targetSuperblock.GetModeInfos())
+ {
+ usesIntraBlockCopy |= modeInfo.UseIntraBlockCopy;
+ }
+
+ Assert.True(usesIntraBlockCopy);
+ Assert.Equal(new Size(Width, Height), decoded.Size);
+
+ string outputDirectory = Path.Combine(
+ TestEnvironment.ActualOutputDirectoryFullPath,
+ "Formats",
+ "Heif",
+ "Av1");
+
+ Directory.CreateDirectory(outputDirectory);
+ File.WriteAllBytes(Path.Combine(outputDirectory, "encoder-frame-328x16-8b-444-intrabc.obu"), payload);
+ }
+
[Fact]
public void PrepareSourceConvertsRgba32DirectlyIntoBorderedEightBitPlane()
{