diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 5e4ef1e19d..0e50d4512c 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -824,11 +824,11 @@ Encoder verification contract: - [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder operation now instantiates matching source and reconstruction owners with ordinary `using` lifetimes and converts packed pixels directly into the source owner before extension. - [~] Temporal delimiter, sequence header, frame header, combined-frame tile-group writing, and an internal reduced-still-picture frame operation now exist locally. The remaining required metadata, padding, multi-tile, option, and public encoder paths are not complete. - [~] Implement superblock and partition analysis for every permitted block size and partition. The current baseline deliberately splits every in-frame node to 8x8 blocks and records decisions in current-libaom writer preorder; block-size selection and non-split partition analysis remain. -- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax; palette, filter intra, and intra-block copy remain. +- [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes; palette and intra-block copy remain. - [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain. -- [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma, now perform live rate-distortion selection; quality mapping, effort-dependent pruning, and the remaining searches are not implemented. -- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Every candidate includes its live mode, angle, alpha, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these stack-only selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma candidate scratch remains one 8x8 reconstruction and one 8x8 coefficient span on the stack; chroma uses one transform-sized reconstruction and coefficient span for each of U and V. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, 31 focused CfL, spatial-chroma, and entropy cases pass, all 8,959 HEIF/AV1 namespace cases pass, and current-main `aomdec` accepts all 29 emitted 8/10/12-bit 4:0:0, 4:2:0, 4:2:2, and 4:4:4 constant or gradient payloads. Remaining mode decision work includes transform-size search, broader joint mode/transform refinement, palette and filter-intra search, partition search, full block-skip RD comparison, and effort-dependent pruning. +- [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma and filter-intra, now perform live rate-distortion selection; quality mapping, effort-dependent pruning, and the remaining searches are not implemented. +- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. Each prepared reference edge retains the common-corner prefix and twice the transform dimension required by directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these stack-only selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma candidate scratch remains one 8x8 reconstruction and one 8x8 coefficient span on the stack; chroma uses one transform-sized reconstruction and coefficient span for each of U and V. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search are contractually DC and skipped, instead of relying on textured content to happen to select the baseline mode. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, 18 focused filter-intra, predictor-reference, syntax-cost, and allocation cases pass, all 8,974 HEIF/AV1 namespace cases pass, and current-main `aomdec` accepts all 29 emitted 8/10/12-bit 4:0:0, 4:2:0, 4:2:2, and 4:4:4 constant or gradient payloads. Remaining mode decision work includes transform-size search, broader joint mode/transform refinement, palette search, partition search, full block-skip RD comparison, and effort-dependent pruning. - [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One picture owner now packs segmentation plus every tile's partition, luma, chroma, and transform edges into one clean byte allocation with typed non-owning views; together with the separately typed packed mode-information owner, the complete picture state uses two allocator rents rather than seven. Exact aligned lengths, clean initialization, and balanced exactly-once returns are covered in Release. Multi-tile payload ownership and verified CDF update behavior remain. - [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains. - [~] The final-block decision workspace uses one reusable 10.3 KiB ImageSharp allocator owner. It contains 1,024 explicitly packed 10-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree, replacing separate managed arrays. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled palette, quantizer, prediction, and partition bytes cannot leak into the next decision pass. Exact allocation, size, initialization, reset, return, repeated-run, writer, entropy, and OBU coverage pass 1,957 of 1,957 direct net11 VSTest cases in Release; complete mode decision still remains. @@ -843,10 +843,11 @@ Encoder verification contract: - [~] The combined-frame writer now completes the byte-counted uncompressed frame header before starting the optional multi-tile tile-group flag, matching current libaom's separate frame-header and tile-group writers. A non-uniform two-tile round trip verifies the explicit boundaries, both tile payloads, and complete stream consumption through direct net11 VSTest in Release. - [~] The first internal frame-to-OBU operation encodes 8-, 10-, and 12-bit monochrome reduced still pictures through the production tile writer and production decoder. Coefficient context initialization now stores `min(abs(level), 127)`, matching current libaom; the previous signed clamp converted every negative transform coefficient to zero and selected invalid nonzero-map distributions. Signed dense and sparse entropy round trips, direct level-buffer saturation coverage, and eight constant/gradient frame cases pass 52 of 52 direct net11 VSTest cases in Release. Current-main `aomdec` accepts all eight emitted payloads. After winner-mode transform refinement, their decoded-frame MD5 values are `d09ea148582b9c93fa78e59426193bbc` (16x16 8-bit constant), `b83eedd5a84428f0120130253b30bdaa` (16x16 8-bit gradient), `f949f7422913e83dff07ee5e0a5087d3` (8x8 8-bit constant), `ae7233a94558978934469dcc4da764dd` (8x8 8-bit gradient), `09223b227f3abc3134d0a3ea15f70c0a` (8x8 10-bit constant), `539aab0e6e14bcaec271febfa8e25444` (8x8 10-bit gradient), `73117a8fc102e5d028f82444fc4d15ab` (8x8 12-bit constant), and `6936a2b62d7220dfb12f3763bb49965d` (8x8 12-bit gradient). This is an independently decodable baseline, not completion evidence for chroma, alpha, options, containers, or the public encoder. - [x] The exact net11 Release rebuild completed at the established 1,005-warning repository baseline with zero errors. The complete HEIF/AV1 namespace passes 8,838 of 8,838 direct VSTest cases with zero failures or skips. Roslynk reports zero compiler errors and no diagnostics in the five changed C# files; `git diff --check` passes and `.gitattributes` is unchanged. -- [~] The same internal frame operation now produces 4:2:0, 4:2:2, and 4:4:4 payloads at 8, 10, and 12 bits. Twenty-one color cases cover constant and spatially varying input at aligned dimensions plus odd 13x11 visible dimensions for every chroma geometry. The production decoder consumes every payload, the decoded output retains non-neutral chroma, and current-main `aomdec` accepts all 29 monochrome and color outputs. After live spatial chroma mode selection, implicit chroma-transform correction, winner-mode luma-transform refinement, and exhaustive chroma-from-luma alpha selection, the odd-dimension decoded-frame MD5 values are `6a9cde0e29d02bcb1e6387f62fb59c88` (4:2:0), `ed8e9e52b0c0a80d6854ded5e0439538` (4:2:2), and `fcba16f73ce3690e83cafef529b11b7c` (4:4:4). This proves legal current-libaom payload syntax across native plane geometries; it does not yet prove target quality or native-plane equality with an independently encoded reference. +- [~] The same internal frame operation now produces 4:2:0, 4:2:2, and 4:4:4 payloads at 8, 10, and 12 bits. Twenty-one color cases cover constant and spatially varying input at aligned dimensions plus odd 13x11 visible dimensions for every chroma geometry. The production decoder consumes every payload, the decoded output retains non-neutral chroma, and current-main `aomdec` accepts all 29 monochrome and color outputs. After live spatial chroma mode selection, implicit chroma-transform correction, winner-mode luma-transform refinement, exhaustive chroma-from-luma alpha selection, and filter-intra search, the odd-dimension decoded-frame MD5 values are `9985f05790d2c9f5f28723ef86d5b89b` (4:2:0), `2ba2f1d0fcfef60394a5175553c7cb8b` (4:2:2), and `6aa7a2ed0dbf76ad2ec0c222585272d0` (4:4:4). This proves legal current-libaom payload syntax across native plane geometries; it does not yet prove target quality or native-plane equality with an independently encoded reference. - [x] Spatial chroma candidates now use the implicit transform derived from the selected UV mode and active transform set, matching current libaom's `intra_mode_to_tx_type` and `av1_get_tx_type` behavior. The same shared derivation is consumed by the decoder, so coefficient scan order, entropy contexts, inverse reconstruction, and encoder rate estimates cannot drift between the two paths. The previous DCT-DCT candidate transform could produce syntactically accepted streams whose non-DC chroma coefficients were interpreted under a different implicit transform. Six production mode-decision cases retain nonzero U and V coefficients and assert the selected transform state across 4:2:0, 4:2:2, and 4:4:4; fifteen exact mapping cases cover every intra mode, reduced sets, and the 32x32 DCT-only fallback. The focused contract passes 21 of 21 direct net11 VSTest cases, the complete HEIF/AV1 namespace passes 8,947 of 8,947, the exact Release rebuild remains at 1,005 warnings and zero errors, and current-main `aomdec` accepts all 29 regenerated payloads. - [~] Luma mode selection now evaluates each of its 61 mode-and-angle candidates with the mode-derived default transform used by current libaom's fast intra path. It then refines only the winning mode across all seven transform types permitted by the 8x8 intra set in transform-enum order. This removes the fixed DCT-DCT limitation while avoiding a 61-by-7 expansion; each trial includes live transform-type and coefficient rate, reconstructed pixel-domain distortion, and the existing allocation-free stack scratch. Eighteen exact-prediction production cases prove DCT-DCT wins equal-cost ties in reference order even when the first pass used a different default, while the 72x72 textured traversal proves a non-DCT transform with nonzero coefficients reaches retained syntax. Current-main `aomdec` accepts all 29 regenerated payloads. Full partition, transform-size, and effort-dependent joint mode/transform search remain. - [x] Chroma-from-luma mode decision now reuses the decoder's SIMD-first 4:2:0, 4:2:2, and 4:4:4 reconstructed-luma preparation and prediction kernels for both byte and high-bit-depth encoder operators. The constant DC predictor for each chroma plane is computed once and its sample refills every alpha candidate, matching libaom's per-plane DC cache instead of rebuilding the same edge average 33 times. Each block uses 512 bytes of fixed stack scratch for the maximum 8-row predictor surface plus 792 bytes for complete U/V rate and distortion tables; no allocator owner, managed object, frame copy, or persistent buffer was added. Live probability costs exactly mirror current libaom's joint-sign ownership and conditional magnitude symbols. Nine production cases independently derive exact CfL targets from decoder-visible reconstructed luma at 8, 10, and 12 bits, and three entropy cases cover two nonzero signs plus each single-zero-plane form. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, all 8,959 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. +- [x] Filter-intra mode decision now runs after ordinary luma modes in current-libaom order, evaluates all five recursive predictors, and refines each predictor across every legal 8x8 transform in transform-enum order. Strictly-better replacement preserves ordinary-mode and filter-mode tie order. Each filter prediction and its source residual are prepared once and reused across transform candidates, avoiding repeated recursive prediction while retaining SIMD-first predictor and subtraction operators. The stack cost is 192 bytes for eight-bit samples or 256 bytes for high-bit-depth samples; no allocator owner or managed buffer was added. Fifteen production cases force every filter mode at 8, 10, and 12 bits and prove retained filter syntax, zero-residual reconstruction, and the DCT-DCT equal-cost transform tie. The decoded-frame MD5 values selected by this checkpoint are `d7d68803763b95827483f14515281d3a` for the 8x8 10-bit gradient, `3f7e34d44c65d7797ad26b5cd4c35bf4` for the 8x8 12-bit gradient, and `9985f05790d2c9f5f28723ef86d5b89b`, `2ba2f1d0fcfef60394a5175553c7cb8b`, and `6aa7a2ed0dbf76ad2ec0c222585272d0` for the odd 4:2:0, 4:2:2, and 4:4:4 gradients. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, 18 focused filter-intra, predictor-reference, syntax-cost, and allocation cases pass, all 8,974 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. - [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips. - [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips. - [x] Finalized fixed-block decisions now set the block-level transform-skip flag only when every retained luma and coded chroma transform has zero EOB, matching current libaom's conjunction of per-plane skip state. The previous always-false flag produced legal but redundant non-skip and zero-coefficient syntax. Monochrome and 4:2:0 regressions prove both branches from actual coefficient state; the focused decision and production-frame set passes 32 of 32 direct net11 VSTest cases. Current-main `aomdec` accepts all 29 regenerated payloads, the recorded decoded-frame MD5s are unchanged, and affected 16x16 constant 8-bit and 10-bit payloads are one byte smaller. The complete HEIF/AV1 namespace passes 8,862 of 8,862 cases with zero failures or skips. diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs index b816b05f66..c803b4526c 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs @@ -57,7 +57,7 @@ internal static class Av1FrameEncoder Use128x128Superblock = false, ForceScreenContentTools = 2, ForceIntegerMotionVector = 2, - EnableFilterIntra = false, + EnableFilterIntra = true, EnableIntraEdgeFilter = false, EnableSuperResolution = false, EnableCdef = false, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs index 1bcf0b2b14..64525eb37d 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs @@ -192,9 +192,11 @@ internal static partial class Av1IntraSuperblockEncoder tileIndex, lumaCoefficients[this.codedAreaLuma..], ref lumaState, - out int lumaAngleDelta); + out int lumaAngleDelta, + out Av1FilterIntraMode filterIntraMode); block.PredictionUnit.AngleDelta[(int)Av1PlaneType.Y] = (sbyte)lumaAngleDelta; + block.FilterIntraMode = filterIntraMode; this.codedAreaLuma += LumaTransformSize.GetSize2d(); bool skipTransform = lumaState.EndOfBlock == 0; @@ -257,7 +259,8 @@ internal static partial class Av1IntraSuperblockEncoder ushort tileIndex, Span retainedCoefficients, ref Av1EncoderTransformBlockState retainedState, - out int selectedAngleDelta) + out int selectedAngleDelta, + out Av1FilterIntraMode selectedFilterIntraMode) { const Av1BlockSize BlockSize = Av1BlockSize.Block8x8; const Av1TransformSize TransformSize = Av1TransformSize.Size8x8; @@ -376,6 +379,7 @@ internal static partial class Av1IntraSuperblockEncoder long bestCost = long.MaxValue; Av1PredictionMode bestMode = Av1PredictionMode.DC; selectedAngleDelta = 0; + selectedFilterIntraMode = Av1FilterIntraMode.AllFilterIntraModes; int baseModeCount = LumaModeSearchOrder.Length; int deltaCount = AngleDeltaSearchOrder.Length; int directionalModeCount = (int)Av1PredictionMode.Directional67Degrees - (int)Av1PredictionMode.Vertical + 1; @@ -490,6 +494,74 @@ internal static partial class Av1IntraSuperblockEncoder } } + if (this.picture.Sequence.SequenceHeader.EnableFilterIntra) + { + Span filterPrediction = stackalloc TSample[SampleCount]; + Span filterResidual = stackalloc short[SampleCount]; + + // Each recursive filter prediction and its source residual are independent of transform type. + // Prepare them once per filter mode so all legal transforms reuse the same samples. + for (Av1FilterIntraMode filterIntraMode = Av1FilterIntraMode.DC; + filterIntraMode < Av1FilterIntraMode.AllFilterIntraModes; + filterIntraMode++) + { + TOperator.PrepareFilterIntra( + this.blockWorkspace, + sourcePlane, + blockOrigin, + filterPrediction, + above, + left, + filterResidual, + filterIntraMode, + TransformSize, + this.bitDepth); + + for (Av1TransformType transformType = Av1TransformType.DctDct; + transformType < Av1TransformType.AllTransformTypes; + transformType++) + { + if (!transformType.IsExtendedSetUsed(transformSetType)) + { + continue; + } + + Av1EncoderTransformBlockState candidateState = default; + long candidateCost = this.GetFilterIntraCandidateCost( + writer, + macroBlock, + sourcePlane, + blockOrigin, + filterPrediction, + filterResidual, + filterIntraMode, + transformType, + blockContext, + candidateReconstruction, + candidateCoefficients, + ref candidateState); + + if (candidateCost < bestTransformCost) + { + CopyCandidate( + candidateReconstruction, + candidateCoefficients, + reconstructionPlane, + blockOrigin, + retainedCoefficients, + TransformSize, + candidateState, + ref retainedState); + + bestTransformCost = candidateCost; + bestMode = Av1PredictionMode.DC; + selectedAngleDelta = 0; + selectedFilterIntraMode = filterIntraMode; + } + } + } + } + return bestMode; } @@ -534,6 +606,11 @@ internal static partial class Av1IntraSuperblockEncoder ref candidateState); int rate = Av1TileWriter.GetLumaModeCost(writer, macroBlock, BlockSize, mode, angleDelta); + if (mode == Av1PredictionMode.DC && this.picture.Sequence.SequenceHeader.EnableFilterIntra) + { + rate += writer.GetFilterIntraModeCost(Av1FilterIntraMode.AllFilterIntraModes, BlockSize); + } + rate += writer.GetCoefficientCost( TransformSize, transformType, @@ -548,6 +625,61 @@ internal static partial class Av1IntraSuperblockEncoder return Av1RateDistortion.GetCost(this.rateMultiplier, rate, distortion); } + private long GetFilterIntraCandidateCost( + Av1SymbolEncoder writer, + Av1MacroBlockD macroBlock, + Buffer2DRegion sourcePlane, + Point blockOrigin, + ReadOnlySpan prediction, + ReadOnlySpan residual, + Av1FilterIntraMode filterIntraMode, + Av1TransformType transformType, + Av1TransformBlockContext blockContext, + Span candidateReconstruction, + Span candidateCoefficients, + ref Av1EncoderTransformBlockState candidateState) + { + const Av1BlockSize BlockSize = Av1BlockSize.Block8x8; + const Av1TransformSize TransformSize = Av1TransformSize.Size8x8; + long distortion = TOperator.EncodePredictionCandidate( + this.blockWorkspace, + sourcePlane, + blockOrigin, + prediction, + residual, + candidateReconstruction, + candidateCoefficients, + TransformSize, + transformType, + Av1Plane.Y, + this.quantization.QIndex[0], + this.quantization.DeltaQDc[(int)Av1Plane.Y], + this.quantization.DeltaQAc[(int)Av1Plane.Y], + this.bitDepth, + ref candidateState); + + int rate = Av1TileWriter.GetLumaModeCost( + writer, + macroBlock, + BlockSize, + Av1PredictionMode.DC, + 0); + + rate += writer.GetFilterIntraModeCost(filterIntraMode, BlockSize); + rate += writer.GetCoefficientCost( + TransformSize, + transformType, + Av1PredictionMode.DC, + candidateCoefficients, + Av1ComponentType.Luminance, + blockContext, + candidateState.EndOfBlock, + this.picture.Parent.FrameHeader.UseReducedTransformSet, + filterIntraMode); + + return Av1RateDistortion.GetCost(this.rateMultiplier, rate, distortion); + } + private static void CopyCandidate( ReadOnlySpan candidateReconstruction, ReadOnlySpan candidateCoefficients, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs index 9d59e76599..ebb78069c8 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs @@ -154,6 +154,67 @@ internal static partial class Av1IntraSuperblockEncoder Av1BitDepth bitDepth, ref Av1EncoderTransformBlockState state); + /// + /// Builds one filter-intra prediction for reuse across transform candidates. + /// + /// The reusable block workspace. + /// The coded source plane. + /// The transform-block origin in plane samples. + /// The contiguous prediction destination. + /// The top reference samples, with prefix storage for the shared corner. + /// The left reference samples. + /// The contiguous source-minus-prediction destination. + /// The selected filter-intra mode. + /// The prediction dimensions. + /// The coded sample bit depth. + public static abstract void PrepareFilterIntra( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span prediction, + ReadOnlySpan above, + ReadOnlySpan left, + Span residual, + Av1FilterIntraMode filterIntraMode, + Av1TransformSize transformSize, + Av1BitDepth bitDepth); + + /// + /// Encodes one prepared prediction with the selected transform into decision scratch. + /// + /// The reusable block workspace. + /// The coded source plane. + /// The transform-block origin in plane samples. + /// The contiguous prediction samples. + /// The contiguous source-minus-prediction samples. + /// The contiguous candidate reconstruction. + /// The candidate entropy-coding coefficients. + /// The transform dimensions. + /// The compound transform applied to the residual. + /// The component plane containing the block. + /// The effective segment quantizer index. + /// The plane DC quantizer adjustment. + /// The plane AC quantizer adjustment. + /// The coded sample bit depth. + /// The candidate transform state. + /// The normalized pixel-domain distortion in AV1 transform units. + public static abstract long EncodePredictionCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + ReadOnlySpan prediction, + ReadOnlySpan residual, + Span reconstruction, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + Av1Plane plane, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth, + ref Av1EncoderTransformBlockState state); + /// /// Encodes one chroma-from-luma candidate into contiguous decision scratch. /// @@ -318,6 +379,74 @@ internal static partial class Av1IntraSuperblockEncoder plane, ref state); + /// + public static void PrepareFilterIntra( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span prediction, + ReadOnlySpan above, + ReadOnlySpan left, + Span residual, + Av1FilterIntraMode filterIntraMode, + Av1TransformSize transformSize, + Av1BitDepth bitDepth) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + + // Prediction finishes before transform search, so its temporary rows can borrow the transform workspace. + Span filterScratch = MemoryMarshal.AsBytes(workspace.TransformWorkspace).Slice( + 0, + Av1FilterIntraPredictorBase.ScratchLength); + + Av1FilterIntraPredictorBase.GetPredictor(filterIntraMode) + .Predict(prediction, width, above, left, width, height, filterScratch); + + Av1ResidualBuilder.Subtract( + Av1TransformBlockEncoder.GetPlaneSpan(source, blockOrigin), + source.Stride, + prediction, + width, + residual, + width, + width, + height); + } + + /// + public static long EncodePredictionCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + ReadOnlySpan prediction, + ReadOnlySpan residual, + Span reconstruction, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + Av1Plane plane, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth, + ref Av1EncoderTransformBlockState state) + => Av1TransformBlockEncoder.EncodePredictionLossyCandidate( + workspace, + source, + blockOrigin, + prediction, + residual, + reconstruction, + quantizedCoefficients, + transformSize, + transformType, + qIndex, + dcDeltaQ, + acDeltaQ, + plane, + ref state); + /// public static long EncodeChromaFromLumaCandidate( Av1EncoderBlockWorkspace workspace, @@ -482,6 +611,83 @@ internal static partial class Av1IntraSuperblockEncoder bitDepth, ref state); + /// + public static void PrepareFilterIntra( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + Span prediction, + ReadOnlySpan above, + ReadOnlySpan left, + Span residual, + Av1FilterIntraMode filterIntraMode, + Av1TransformSize transformSize, + Av1BitDepth bitDepth) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + + // Prediction finishes before transform search, so its temporary rows can borrow the transform workspace. + Span filterScratch = MemoryMarshal.Cast(workspace.TransformWorkspace).Slice( + 0, + Av1FilterIntraPredictorBase.ScratchLength); + + Av1FilterIntraPredictorBase.GetPredictor(filterIntraMode) + .Predict( + MemoryMarshal.Cast(prediction), + width, + MemoryMarshal.Cast(above), + MemoryMarshal.Cast(left), + width, + height, + bitDepth.GetBitCount(), + filterScratch); + + Av1ResidualBuilder.Subtract( + Av1TransformBlockEncoder.GetPlaneSpan(source, blockOrigin), + source.Stride, + prediction, + width, + residual, + width, + width, + height); + } + + /// + public static long EncodePredictionCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + ReadOnlySpan prediction, + ReadOnlySpan residual, + Span reconstruction, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + Av1Plane plane, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth, + ref Av1EncoderTransformBlockState state) + => Av1TransformBlockEncoder.EncodePredictionLossyCandidate( + workspace, + source, + blockOrigin, + prediction, + residual, + reconstruction, + quantizedCoefficients, + transformSize, + transformType, + qIndex, + dcDeltaQ, + acDeltaQ, + plane, + bitDepth, + ref state); + /// public static long EncodeChromaFromLumaCandidate( Av1EncoderBlockWorkspace workspace, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs index 466accc03c..b70ade1a11 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TransformBlockEncoder.cs @@ -158,6 +158,86 @@ internal static class Av1TransformBlockEncoder return Av1ResidualBuilder.SumSquares(workspace.Residual[..transformSize.GetSize2d()]) << 4; } + /// + /// Encodes one eight-bit candidate from a cached prediction and source residual. + /// + /// The reusable residual, coefficient, and transform storage. + /// The coded source plane. + /// The block origin in plane samples. + /// The contiguous prediction samples. + /// The contiguous source-minus-prediction samples. + /// The contiguous candidate reconstruction. + /// The candidate entropy-coding coefficients. + /// The selected transform dimensions. + /// The selected compound transform type. + /// The segment quantizer index. + /// The plane DC quantizer adjustment. + /// The plane AC quantizer adjustment. + /// The component plane containing the block. + /// The candidate transform type and end-of-block syntax. + /// The normalized pixel-domain distortion in AV1 transform units. + public static long EncodePredictionLossyCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + ReadOnlySpan prediction, + ReadOnlySpan residual, + Span reconstruction, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1Plane plane, + ref Av1EncoderTransformBlockState state) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + int sampleCount = transformSize.GetSize2d(); + ReadOnlySpan sourceSamples = GetPlaneSpan(source, blockOrigin); + + // Each transform trial mutates reconstruction and residual scratch, so restore both prepared inputs. + prediction[..sampleCount].CopyTo(reconstruction); + residual[..sampleCount].CopyTo(workspace.Residual); + EncodeLossy( + workspace, + quantizedCoefficients, + transformSize, + transformType, + qIndex, + dcDeltaQ, + acDeltaQ, + Av1BitDepth.EightBit, + ref state); + + if (state.EndOfBlock > 0) + { + Av1InverseTransformer.Reconstruct8Bit( + workspace.DequantizedCoefficients, + reconstruction, + width, + transformSize, + transformType, + (int)plane, + state.EndOfBlock, + false, + workspace.TransformWorkspace); + } + + Av1ResidualBuilder.Subtract( + sourceSamples, + source.Stride, + reconstruction, + width, + workspace.Residual, + width, + width, + height); + + return Av1ResidualBuilder.SumSquares(workspace.Residual[..sampleCount]) << 4; + } + /// /// Encodes one eight-bit chroma-from-luma candidate into contiguous decision scratch. /// @@ -402,6 +482,95 @@ internal static class Av1TransformBlockEncoder return normalizedDistortion << 4; } + /// + /// Encodes one high-bit-depth candidate from a cached prediction and source residual. + /// + /// The reusable residual, coefficient, and transform storage. + /// The coded source plane. + /// The block origin in plane samples. + /// The contiguous prediction samples. + /// The contiguous source-minus-prediction samples. + /// The contiguous candidate reconstruction. + /// The candidate entropy-coding coefficients. + /// The selected transform dimensions. + /// The selected compound transform type. + /// The segment quantizer index. + /// The plane DC quantizer adjustment. + /// The plane AC quantizer adjustment. + /// The component plane containing the block. + /// The coded sample bit depth. + /// The candidate transform type and end-of-block syntax. + /// The normalized pixel-domain distortion in AV1 transform units. + public static long EncodePredictionLossyCandidate( + Av1EncoderBlockWorkspace workspace, + Buffer2DRegion source, + Point blockOrigin, + ReadOnlySpan prediction, + ReadOnlySpan residual, + Span reconstruction, + Span quantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1Plane plane, + Av1BitDepth bitDepth, + ref Av1EncoderTransformBlockState state) + { + int width = transformSize.GetWidth(); + int height = transformSize.GetHeight(); + int sampleCount = transformSize.GetSize2d(); + ReadOnlySpan sourceSamples = GetPlaneSpan(source, blockOrigin); + + // Each transform trial mutates reconstruction and residual scratch, so restore both prepared inputs. + prediction[..sampleCount].CopyTo(reconstruction); + residual[..sampleCount].CopyTo(workspace.Residual); + EncodeLossy( + workspace, + quantizedCoefficients, + transformSize, + transformType, + qIndex, + dcDeltaQ, + acDeltaQ, + bitDepth, + ref state); + + if (state.EndOfBlock > 0) + { + Av1InverseTransformer.ReconstructHighBitDepth( + workspace.DequantizedCoefficients, + MemoryMarshal.Cast(reconstruction), + width, + transformSize, + transformType, + (int)plane, + state.EndOfBlock, + false, + bitDepth, + workspace.TransformWorkspace); + } + + Av1ResidualBuilder.Subtract( + sourceSamples, + source.Stride, + reconstruction, + width, + workspace.Residual, + width, + width, + height); + + long distortion = Av1ResidualBuilder.SumSquares(workspace.Residual[..sampleCount]); + int shift = (bitDepth.GetBitCount() - 8) * 2; + long normalizedDistortion = shift == 0 + ? distortion + : (distortion + (1L << (shift - 1))) >> shift; + + return normalizedDistortion << 4; + } + /// /// Encodes one high-bit-depth chroma-from-luma candidate into contiguous decision scratch. /// diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs index bf1c415563..d047ba8f98 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs @@ -3,6 +3,7 @@ using System.Buffers; using System.Numerics; +using System.Runtime.InteropServices; using SixLabors.ImageSharp.Formats.Heif.Av1; using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; @@ -66,6 +67,7 @@ public class Av1IntraSuperblockEncoderTests using Av1EncoderModeInfoBuffer modeInfo = new(Configuration.Default, Width, Height, disallow4x4AllFrames: true); Av1PictureControlSet picture = CreatePicture(modeInfo, colorConfig, use128x128Superblock: true, qIndex: 73); + picture.Sequence.SequenceHeader.EnableFilterIntra = true; using Av1EncoderCoefficientBuffer coefficients = new( Configuration.Default, picture.Sequence.SequenceHeader, @@ -1194,6 +1196,246 @@ public class Av1IntraSuperblockEncoderTests Assert.NotEqual(0, tileWriter.GetTileData(0).Length); } + [Theory] + [InlineData((int)Av1FilterIntraMode.DC)] + [InlineData((int)Av1FilterIntraMode.Vertical)] + [InlineData((int)Av1FilterIntraMode.Horizontal)] + [InlineData((int)Av1FilterIntraMode.Directional157)] + [InlineData((int)Av1FilterIntraMode.Paeth)] + public void ProductionTileSelectsFilterIntraMode(int filterIntraModeValue) + => VerifyProductionTileSelectsFilterIntraMode( + filterIntraModeValue, + 8, + static (source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new( + Configuration.Default, + source, + reconstruction, + picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialSize: 512), + static (mode, destination, above, left, _, scratch) => + Av1FilterIntraPredictorBase.GetPredictor(mode) + .Predict(destination, 8, above, left, 8, 8, scratch)); + + [Theory] + [InlineData((int)Av1FilterIntraMode.DC, 10)] + [InlineData((int)Av1FilterIntraMode.DC, 12)] + [InlineData((int)Av1FilterIntraMode.Vertical, 10)] + [InlineData((int)Av1FilterIntraMode.Vertical, 12)] + [InlineData((int)Av1FilterIntraMode.Horizontal, 10)] + [InlineData((int)Av1FilterIntraMode.Horizontal, 12)] + [InlineData((int)Av1FilterIntraMode.Directional157, 10)] + [InlineData((int)Av1FilterIntraMode.Directional157, 12)] + [InlineData((int)Av1FilterIntraMode.Paeth, 10)] + [InlineData((int)Av1FilterIntraMode.Paeth, 12)] + public void ProductionTileSelectsFilterIntraModeHighBitDepth( + int filterIntraModeValue, + int bitDepth) + => VerifyProductionTileSelectsFilterIntraMode( + filterIntraModeValue, + bitDepth, + static (source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new( + Configuration.Default, + source, + reconstruction, + picture, + coefficients, + superblockWorkspace, + blockWorkspace, + initialSize: 512), + static (mode, destination, above, left, sampleBitDepth, scratch) => + Av1FilterIntraPredictorBase.GetPredictor(mode) + .Predict( + MemoryMarshal.Cast(destination), + 8, + MemoryMarshal.Cast(above), + MemoryMarshal.Cast(left), + 8, + 8, + sampleBitDepth, + MemoryMarshal.Cast(scratch))); + + private static void VerifyProductionTileSelectsFilterIntraMode( + int filterIntraModeValue, + int bitDepth, + TileWriterFactory createWriter, + FilterPrediction predictFilter) + where TSample : unmanaged, IBinaryInteger + { + const int Width = 16; + const int Height = 16; + const int QIndex = 1; + const int TargetX = 8; + const int TargetY = 8; + const Av1TransformSize TransformSize = Av1TransformSize.Size8x8; + Av1FilterIntraMode filterIntraMode = (Av1FilterIntraMode)filterIntraModeValue; + int sampleScale = 1 << (bitDepth - 8); + ObuColorConfig colorConfig = new() + { + IsMonochrome = true, + SubSamplingX = true, + SubSamplingY = true, + BitDepth = (Av1BitDepth)((bitDepth - 8) / 2) + }; + + using Av1EncoderFrameBuffer pilotSource = new( + Configuration.Default, + Width, + Height, + bitDepth, + Av1ColorFormat.Yuv400, + 1, + 1); + + using Av1EncoderFrameBuffer pilotReconstruction = new( + Configuration.Default, + Width, + Height, + bitDepth, + Av1ColorFormat.Yuv400, + 1, + 1); + + Buffer2DRegion pilotLuma = pilotSource.Frame.CodedView.GetPlane(Av1Plane.Y); + for (int y = 0; y < pilotLuma.Height; y++) + { + Span row = pilotLuma.DangerousGetRowSpan(y); + for (int x = 0; x < row.Length; x++) + { + row[x] = TSample.CreateChecked( + (64 + (((x * 71) + (y * 109) + (((x ^ y) & 3) * 37)) & 127)) * sampleScale); + } + } + + ClearPlane(pilotReconstruction.Luma); + using Av1EncoderModeInfoBuffer pilotModeInfo = new(Configuration.Default, Width, Height, disallow4x4AllFrames: true); + Av1PictureControlSet pilotTemplate = CreatePicture(pilotModeInfo, colorConfig, use128x128Superblock: false, QIndex); + pilotTemplate.Sequence.SequenceHeader.EnableFilterIntra = true; + using Av1EncoderPictureBuffer pilotPicture = new( + Configuration.Default, + pilotTemplate.Sequence.SequenceHeader, + pilotTemplate.Parent.FrameHeader, + Width, + Height); + + using Av1EncoderCoefficientBuffer pilotCoefficients = new( + Configuration.Default, + pilotTemplate.Sequence.SequenceHeader, + Width, + Height); + + using Av1EncoderSuperblockWorkspace pilotSuperblockWorkspace = new(Configuration.Default); + using Av1EncoderBlockWorkspace pilotBlockWorkspace = new(Configuration.Default); + using Av1IntraTileWriter pilotWriter = createWriter( + pilotSource.Frame, + pilotReconstruction.Frame, + pilotPicture.Picture, + pilotCoefficients, + pilotSuperblockWorkspace, + pilotBlockWorkspace); + + Buffer2DRegion reconstructedLuma = pilotReconstruction.Frame.CodedView.GetPlane(Av1Plane.Y); + Span aboveStorage = stackalloc TSample[9]; + Span above = aboveStorage[1..]; + Span left = stackalloc TSample[8]; + ReadOnlySpan reconstructedAbove = reconstructedLuma.DangerousGetRowSpan(TargetY - 1); + aboveStorage[0] = reconstructedAbove[TargetX - 1]; + reconstructedAbove.Slice(TargetX, 8).CopyTo(above); + for (int row = 0; row < 8; row++) + { + left[row] = reconstructedLuma.DangerousGetRowSpan(TargetY + row)[TargetX - 1]; + } + + Span target = stackalloc TSample[TransformSize.GetSize2d()]; + Span filterScratch = stackalloc TSample[Av1FilterIntraPredictorBase.ScratchLength]; + predictFilter(filterIntraMode, target, above, left, bitDepth, filterScratch); + + using Av1EncoderFrameBuffer source = new( + Configuration.Default, + Width, + Height, + bitDepth, + Av1ColorFormat.Yuv400, + 1, + 1); + + using Av1EncoderFrameBuffer reconstruction = new( + Configuration.Default, + Width, + Height, + bitDepth, + Av1ColorFormat.Yuv400, + 1, + 1); + + Buffer2DRegion sourceLuma = source.Frame.CodedView.GetPlane(Av1Plane.Y); + for (int y = 0; y < pilotLuma.Height; y++) + { + pilotLuma.DangerousGetRowSpan(y).CopyTo(sourceLuma.DangerousGetRowSpan(y)); + } + + for (int row = 0; row < 8; row++) + { + target.Slice(row * 8, 8).CopyTo(sourceLuma.DangerousGetRowSpan(TargetY + row).Slice(TargetX, 8)); + } + + ClearPlane(reconstruction.Luma); + using Av1EncoderModeInfoBuffer modeInfo = new(Configuration.Default, Width, Height, disallow4x4AllFrames: true); + Av1PictureControlSet pictureTemplate = CreatePicture(modeInfo, colorConfig, use128x128Superblock: false, QIndex); + pictureTemplate.Sequence.SequenceHeader.EnableFilterIntra = true; + using Av1EncoderPictureBuffer picture = new( + Configuration.Default, + pictureTemplate.Sequence.SequenceHeader, + pictureTemplate.Parent.FrameHeader, + Width, + Height); + + using Av1EncoderCoefficientBuffer coefficients = new( + Configuration.Default, + pictureTemplate.Sequence.SequenceHeader, + Width, + Height); + + using Av1EncoderSuperblockWorkspace superblockWorkspace = new(Configuration.Default); + using Av1EncoderBlockWorkspace blockWorkspace = new(Configuration.Default); + using Av1IntraTileWriter tileWriter = createWriter( + source.Frame, + reconstruction.Frame, + picture.Picture, + coefficients, + superblockWorkspace, + blockWorkspace); + + Buffer2DRegion actualLuma = reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y); + Assert.Equal(above, actualLuma.DangerousGetRowSpan(TargetY - 1).Slice(TargetX, 8)); + for (int row = 0; row < 8; row++) + { + Assert.Equal(left[row], actualLuma.DangerousGetRowSpan(TargetY + row)[TargetX - 1]); + Assert.Equal( + target.Slice(row * 8, 8), + actualLuma.DangerousGetRowSpan(TargetY + row).Slice(TargetX, 8)); + } + + ref Av1MacroBlockModeInfo targetBlock = ref picture.Picture.GetMacroBlockModeInfo(new Point(2, 2)); + Assert.Equal(Av1PredictionMode.DC, targetBlock.Block.Mode); + Assert.Equal(filterIntraMode, superblockWorkspace.FinalBlocks[3].FilterIntraMode); + + int targetTransformIndex = (3 * TransformSize.GetSize2d()) / + Av1EncoderCoefficientBuffer.TransformBlockUnitCoefficientCount; + + Av1EncoderTransformBlockState targetState = + coefficients.GetTransformBlockSpan(0, Av1Plane.Y)[targetTransformIndex]; + + Assert.Equal((ushort)0, targetState.EndOfBlock); + Assert.Equal(Av1TransformType.DctDct, targetState.TransformType); + Assert.NotEqual(0, pilotWriter.GetTileData(0).Length); + Assert.NotEqual(0, tileWriter.GetTileData(0).Length); + } + [Fact] public void ProductionDirectionalModesConsumeAvailableExtendedEdges() { @@ -1604,6 +1846,15 @@ public class Av1IntraSuperblockEncoderTests Av1EncoderBlockWorkspace blockWorkspace) where TSample : unmanaged; + private delegate void FilterPrediction( + Av1FilterIntraMode mode, + Span destination, + ReadOnlySpan above, + ReadOnlySpan left, + int bitDepth, + Span scratch) + where TSample : unmanaged; + private static void ClearPlane(Buffer2D plane) where TSample : unmanaged {