diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 116bb50eb3..c6708c6933 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -6,11 +6,282 @@ Complete a production-quality, fully managed AV1 codec and its bounded AVIF/HEIF This plan is the authoritative delivery checklist. A source file, unit test, build, self-roundtrip, or local implementation is not completion evidence by itself. +Checkpoint handling: work stays in the existing checkout. Do not create or use worktrees; the user reported a crash. +Commit completed, verified features regularly, with native reference material and temporary integration excluded. +The probability-storage checkpoint is `a658a2cb7`; adaptive syntax and retained palette tokens are `d8f6a1de3`. +The final supporting entropy, mode-grid, frame-buffer, and intra-copy run passed 2,085 tests with zero failures. +The encoder deblocking comparison covered 102,390 samples in 12 streams at 8/10/12 bits and 400/420/422/444: +maximum difference zero, differing samples zero, and samples exceeding one zero. These are same-bitstream +reconstruction checks; they do not establish parity between separately configured encoders or performance acceptance. + +Acceptance criteria are separate for encoding and decoding: + +- Encoder parity permits at most one component unit per sample when comparing separately encoded results + from identical source samples and explicitly reconciled settings. Report maximum errors and counts exceeding one. +- Decoder output must be byte exact against the reference for the same bitstream, precision, and output conversion. + No one-unit tolerance applies to decoding. Report every differing sample and byte count; the required count is zero. +- Historical decoder reports of zero samples exceeding one are insufficient by themselves. A recorded maximum + error of zero establishes sample equality only for the stated comparison scope, not complete decoder correctness. + +## Active production milestone: encoder motion search + +The persistent goal remains active. Complete the integrated encoder motion-search path, including configuration, +allocation geometry, rate costs, candidate and winner state, full-pixel search, fractional refinement, and production +verification before advancing. The following dependency result does not close that milestone. + +- The user resolved the speed-setting question: replace the invented Effort 0-10 scale with the reference's + cpu-used setting as an enum. Preserve values 0-9 and the native direction, with higher values selecting faster + encoding. Do not reverse-map or alias effort values. Use descriptive PascalCase names; the independent-frame + coding mode is named IntraOnly in managed code. `HeifEncodingSpeed` now defines Level0-Level9. + The public Speed property now passes these values unchanged into sequence motion search. The old Effort + property still controls unmigrated intra and global-motion policies; replacing those policies remains required. +- The non-realtime reference default is zero (`av1/av1_cx_iface.c:255-256`); validation permits 0-9 outside + realtime mode (`:781-782`). The value is assigned directly to encoder speed (`:1401`). +- Frame storage now takes an explicit luma border. Still images retain 64 samples; the byte and high-bit-depth + sequence owners use the selected superblock width plus 32. Source, reconstruction, and reference owners use the + same geometry. Native evidence: `av1/encoder/encoder_utils.h:1051-1064`, + `av1/av1_cx_iface.c:3280-3282,3513-3531`, and `av1/encoder/encoder.c:2684-2694`. + Managed changes are in `Av1EncoderFrame.GetPlaneBufferSize`, `Av1EncoderFrameBuffer`, and both sequence + constructors in `Av1FrameEncoder`. Motion consumers read the physical border from the retained plane view. +- Release .NET 11 build passed with zero errors and 1,009 warnings. The focused frame, superblock, intra-copy, + and transform run passed all 275 cases in 21.1628 seconds. Existing edge-replication and exact-owner tests now + include 96- and 160-sample borders while retaining every 64-sample expectation. + Evidence: `D:\GitHub\ynse01\av1-takeover-20260905\inter-frame-border-r1.trx`. +- At the border-only checkpoint, search still used the old effort-dependent controller. The production + integration recorded below supersedes that controller. No separate-encoder parity measurement has been run. +- The 12 regenerated moving-color sequence streams match optimized native decoding at all 21,348 samples: + maximum error zero, differing samples zero, and counts above one zero. This is same-stream evidence only. +- `Av1MotionSearchSettings` now resolves the motion-policy dependency from the enum, coding mode, dimensions, + quantizer, boosted-frame role, and content classification. The production integration below consumes these settings. + The reference initialization and override order is in `speed_features.c:2345-2377,2709-2739,2670-2685`, + with independent-frame policies at `:345-618`, sequence policies at `:1097-1516`, dimension policies at + `:169-342,713-1089`, and quantizer-selected patterns at `:2999-3050`. Block-size pattern changes are in + `motion_search_facade.h:97-144`; mesh patterns are in `speed_features.c:25-46`. Verification and consumption + by the full motion-search path remain required. This type is not a completed encoder architecture. +- The settings/border dependency run passed 312/312 tests in 20.7714 seconds under Release .NET 11 + (`motion-settings-border-r1.trx`). These assertions check policy boundaries and allocation geometry only. +- Inter motion rates now use worker-lifetime tables capturing the tile distributions at superblock entry. + `Av1MotionVectorCosts.cs:35-204` builds all signed component costs by magnitude recurrence; + `Av1EncoderBlockWorkspace` appends exactly 131,072 integer elements (512 KiB) only for inter workers. + The sequence owner retains this storage across frames; still-image workers retain their previous allocation. + Native evidence: `block.h:763-793`, `encoder_alloc.h:57-75`, `encodemv.c:125-250`, `rd.c:687-705`, + and `encodeframe_utils.c:1663-1675`. Per-superblock refresh is connected; speed-dependent row/set refresh + remains to be integrated with the new configuration controller. +- Inter-mode rate calculation now applies the missing rounded 108/128 weight to motion-vector rate alone. + Evidence: `mcomp.c:306-312`, `rd.h:46`, `motion_search_facade.c:535-542`; managed owning method is + `Av1IntraSuperblockEncoder.ReferenceModeDecision.GetInterModeRate`. Search still requires the separate + SAD and variance domains; this mode-rate correction does not reconcile the old search controller. +- After correcting a member-order analyzer failure, the Release .NET 11 build passed with zero errors and + 1,009 warnings. `motion-costs-r1.trx` passed 132/132 tests in 11.5927 seconds. New tests compare every + representable component at integer, quarter-, and eighth-sample precision against independent syntax + traversal, before and after adaptation, and verify snapshot retention and exact allocation ownership. + No separate-encoder parity or performance conclusion follows from this verification. +- Frame-relative search bounds are now integrated into the existing inter full-pixel and fractional path. + `Av1MotionVector.cs:115-168` computes distinct-prediction limits, reference-centered range intersections, + inward full-pixel rounding, and reserved-endpoint exclusion using `Rectangle` with exclusive upper edges. + Native evidence: `mcomp.h:203-228,341-357` and `mcomp.c:233-264`. The old search-order policy remains open. +- Final focused verification after the motion-state edits passed 185/185 cases in 12.7853 seconds + (`motion-state-final-r1.trx`), following a successful Release .NET 11 build with zero errors and 1,009 warnings. + The 12 freshly regenerated sequence streams again matched native decoding at all 21,348 samples, maximum + error zero, differing samples zero, and samples exceeding one zero. This does not establish separate-encoder parity. +- Corrected a further numerical defect in high-bit-depth full-pixel intra-copy search. The shared SAD operators + deliberately return unnormalized differences; the controller now truncates each complete SAD by 2 or 4 bits + before adding eight-bit-domain motion cost. Native evidence: `encoder_utils.h:157-172`; managed evidence: + `Av1IntraBlockCopySearchIndex.cs:689-691,797-893,895-1003,1005-1027`. Initial candidates, diamond candidates, + mesh batches, and tails use the same normalization. The existing raw sample-operator contract is preserved. +- `PixelSearchNormalizesSadBeforeComparingMotionRate` verifies the actual search winner at 10 and 12 bits. + Its two candidates deliberately reverse their SAD-plus-rate ordering if normalization is omitted; the test + asserts that arrangement before invoking production search. The final focused run passed 138/138 cases in + 12.5530 seconds (`motion-sad-normalization-r1.trx`), after a successful Release .NET 11 build with zero errors + and 1,009 warnings. Fresh native comparison again reported maximum error zero across all 21,348 samples. + Full search policy, inter SAD/variance domains, and separate-encoder parity remain unresolved. + +- The user has authorized regular commits after completed features. Inspect every staged diff and exclude + temporary native integration and generated artifacts. Checkpoint `d78734dc7` contains the verified high-depth + intra-copy SAD correction only; its regression tests remain with the pending explicit-border constructor changes. +- Rectangular SAD and residual-moment traversals now extend the existing residual operators without allocating + a residual plane. They cover full and alternate rows, independent strides, vector-width tails, and wide squared + totals for 128x128 twelve-bit blocks. Source contracts: `aom_dsp/variance.c:268-356` and + `av1/encoder/encoder_utils.h:157-172`. The fixed 8x8 entry points are preserved. The new rectangular entry + points are not yet consumed by the production motion controller and do not establish larger inter-block support. +- After correcting six comment-spacing analyzer errors, Release .NET 11 built with zero errors and 1,009 warnings. + `rectangular-motion-metrics-r1.trx` passed 289/289 cases in 24.4180 seconds, including scalar comparisons + across hardware paths, both signs at maximum precision, full-row and alternate-row costs, frame ownership, + intra-copy, and sequence encoding. The 12 newly regenerated streams again matched native decoding across + 21,348 samples with maximum error zero. No encoder parity or performance conclusion follows. +- Full-pixel and fractional controller comparison now includes `mcomp.c:1005-1294,1311-1920,2482-3366`. + Full-pixel SAD rates use an integer-rounded reference vector, whereas variance and fractional costs retain + the subpixel reference (`mcomp.c:317-386`). Search retains winner variance, SSE, motion cost, a second + candidate, and the five-position neighboring cost list; these are inputs to subsequent candidate refinement. + The regular fractional tree uses directional ties, a selected diagonal, and conditional second-level probes; + the two pruned trees have different first-step and follow-up decisions. These paths remain to be integrated. + +- The complete single-reference full-pixel controller is now implemented in + `Av1MotionSearchBase.FullPixelSearch.Search` (`Av1MotionSearchBase.cs:146-243`), with diamond restarts, + pattern walks, mesh passes, alternate-row reliability fallback, separate SAD/variance rates, and retained + winner/second-candidate/neighborhood state. Its byte and word operators consume the existing rectangular + residual traversal. It is not yet called by the production frame encoder: fractional search, native-valued + configuration plumbing, frame/block starting-candidate state, and final winner refinement remain open. +- `Av1MotionSearchSites.Configure` (`Av1MotionSearchSites.cs:63-153`) retains the six distinct site shapes and + stride-relative offsets in the existing worker owner. Each configuration uses 794 integers, matching the + eight-byte site and 22-stage/17-slot layout in `mcomp_structs.h:42-54`; six configurations add 18.61 KiB. + Fast diamond variants share the big-diamond shape. Configuration is rebuilt only after a stride change. + Pixel kernels are vectorized, but arbitrary three/four-candidate batches still require integration. +- Independent native verification calls the official `av1_make_default_fullpel_ms_params` and + `av1_full_pixel_search` from the optimized static library. Temporary adapter sources, binaries, and results + are outside the repository in `D:\GitHub\ynse01\av1-takeover-20260905`. Native source rows are copied into + 32-byte-aligned rows with identical pixel contents because optimized eight-bit SAD performs aligned source + loads (`aom_dsp/x86/sad4d_sse2.asm:152,165`); the managed inputs intentionally have odd strides. + The adapter first required a command-quoting correction, then exposed an eight-bit source-alignment access + violation. The aligned-row contract was corrected before accepting comparison results. +- The first valid comparison found two hexagon disagreements. Source tracing identified the interior + four-site-group path: `mcomp.c:1061-1076,1130-1147` passes a remainder count to the scalar helper whose + loop bound is exclusive (`:947-967`). Interior six-site stages therefore omit their trailing two sites; + the boundary path includes them. The managed controller now preserves that observed decision path. + No claim that extra candidate evaluation improves the encoder is made. +- Final Release .NET 11 build passed with zero errors and 1,009 warnings. + `full-pixel-controller-final-r1.trx` passed 315/315 cases in 20.7391 seconds. The freshly exported + 324 native comparisons cover all nine search methods, 16x16 and 64x64 blocks, 8/10/12-bit samples, + exact/perturbed predictions, fractional spatial references, clamped starts, mesh-triggering error, and + alternate-row fallback. Winner, second candidate, variance, squared error, motion cost, and five neighboring + costs all matched exactly: zero mismatched cases, zero differing fields, and maximum field difference zero. + Evidence: `motion-reference-comparison.json`. This verifies this controller subset, not encoder parity, + complete motion search, the full codec, or performance. +- Fractional refinement now implements the regular, pruned, and more-pruned trees in + `Av1MotionSearchBase.FractionalSearch.Search` (`Av1MotionSearchBase.Fractional.cs:136-275`). + It preserves integer winner statistics, strict directional ties, conditional second-level probes, quadratic + half-sample selection, precision stops, and the three retained centers used to terminate duplicate paths. + Native evidence: `mcomp.c:2615-2897,2971-3000,3026-3350`. The new controller is not yet wired into frame encoding; + scaled references, compound prediction, and frame/block candidate coordination remain open. +- Search interpolation reuses the existing direct SIMD convolution routines through + `Av1TranslationalInterPredictor.PredictForSearch` (`Av1TranslationalInterPredictor.Search.cs:25-239`). + Each Q7 pass rounds and clips to sample precision. The second pass consumes the clipped first pass, unlike + the final reconstruction path. The packed prediction aliases consumed rows of the 128-column intermediate. + Native evidence: `reconinter_enc.c:462-595` and `common/filter.h:271-279`. Required worker capacity is + 17,408 samples (17 KiB for bytes, 34 KiB for words), from `encoder.c:982-992`; production ownership plumbing + remains in progress. This does not enlarge the existing ten-buffer 8x8 inter workspace. +- After correcting ten argument-formatting analyzer errors and one test brace-spacing warning, Release .NET 11 + built with zero errors and 1,009 warnings. `fractional-controller-final-r1.trx` passed 329/329 focused tests + in 23.6178 seconds. Expanded native verification matched all 486 full-pixel cases and all 1,944 fractional + cases exactly, including 8x8, 16x16, and 64x64 blocks at 8/10/12 bits. Fractional coverage includes all three + policies, all three tap selections, four precision stops, retained/recomputed starting statistics, absent/present + integer costs, and repeated-path termination. All 18 published fractional fields had maximum difference zero + and mismatch count zero (`fractional-reference-comparison.json`, outside the repository). + These comparisons establish the stated unscaled search subset only; encoder parity and performance remain unproven. +- Checkpoint `e9babd55a` commits only the rectangular error-metric API and its independent scalar/SIMD tests + after the final focused run. The staged diff was inspected and contained no temporary native material. +- Production caller comparison identifies the next required integration dependencies: + `Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:1248-1385` still uses effort-derived radius and + one eight-direction pass per stage, while `motion_search_facade.c:151-283` derives stages from frame motion + history, spatial-reference magnitude, configured site radii, and the caller's search-range limit. + Frame history is initialized/consumed in `encoder.c:2010-2042`, updated from encoded new vectors in + `encodemv.c:268-272` and `bitstream.c:3953-3967`, and combined with spatial context from `rd.c:1173-1203`. + This is missing control/state functionality, not a proposed performance heuristic. +- `SelectInterBlock` still gathers candidates in advance (`ReferenceModeDecision.cs:555-596`) and lacks the + coordinated TPL/start-candidate and DRL-result state in `motion_search_facade.c:47-118,174-258,344-392,499-534`. + The reference selects up to two weighted full-pixel starts and retains the second winner for fractional + refinement (`:294-323,403-438`). Under the slower policy, the two refined winners are compared using + estimated transform rate-distortion (`:439-481`), so selecting solely by search variance would be incorrect. +- The required transform estimator is `tx_search.c:3142-3265`: DCT transforms, regular quantization, + coefficient rate, transform-domain distortion, and skip/non-skip decisions with retained entropy contexts. + Existing `Av1ForwardQuantizer.QuantizeLossy` (`Av1ForwardQuantizer.cs:29-59`) provides fast quantization + only. The regular quantizer and transform estimator are now implemented as described below; their consumption + by the production motion controller remains required instead of the current reconstruction-SSE cost. + Parameter evidence: `av1_quantize.c:582-631`; arithmetic evidence: + `aom_dsp/quantize.c:109-169,262-316`; transform-error scaling: `tx_search.c:1115-1154`. + +- Checkpoint `d5d4d74b1` adds regular quantization through the existing closed-generic quantizer traversal: + `Av1ForwardQuantizer.Regular.cs:25-175` and its byte/high-bit-depth semantic operators. It preserves + zero-bin rounding, sharpness, signed reciprocal correction, byte-source magnitude saturation, dequantization, + and scan end positions. No per-transform allocation or coefficient copy is introduced. +- Checkpoint `869340d68` adds `Av1TransformBlockEncoder.EstimateInterTransform` + (`Av1TransformBlockEncoder.cs:1256-1408`), which composes + DCT/reversible transforms, regular/lossless quantization, coefficient rate, local entropy-edge updates, + whole-block skip selection, and partial-estimate termination. It visits multiple transforms within prediction + blocks up to 128x128 and reuses the caller's coefficient workspace. Only two 32-byte local context arrays are + added. `GetTransformError` widens SIMD lanes before squaring, rounds high-depth totals before transform + scaling, and uses the scale-zero/one/two factors of one-quarter/one/four. The scaling constant is one + (`av1/common/idct.h:34`), not two; normalization is in `rdopt.c:789-806`. +- Source reconciliation confirms this estimator uses no quantization matrices or coefficient optimization: + `tx_search.c:3199-3205` calls `av1_setup_quant`, whose matrix pointers are null (`encodemb.c:478-496`). + This is specific to the winner estimate and does not resolve missing optimization/matrix policy elsewhere. + The estimate retains coefficient-skip rate for all-empty blocks and excludes the block skip-header rate + from returned statistics, while including that rate in the decision (`tx_search.c:3236-3265`). +- Independent temporary tooling calls the complete optimized `av1_estimate_txfm_yrd`, with default tile + probabilities, explicit matching rate multiplier/header costs, source-minus-prediction inputs, bit depth, + quantizer, sharpness, block/transform geometry, and incoming coefficient contexts. A first comparison exposed + invalid test-generated sign contexts packed above bit six instead of bit three. Input generation was corrected; + production rate arithmetic was unchanged. Regular-quantizer comparison separately reconciles native + column-major coefficients with the established managed raster layout before invoking native quantization. +- Hardware exports now retain separate results for each actual vector width. This exposed an inert .NET 11 + `EnableAVX512F` test switch; `FeatureTestRunner` now maps that request to `EnableAVX512` on .NET 11. + All four executed paths (scalar, 128, 256, 512) independently match optimized native results: 5,184 regular + quantizer cases with zero differing quantized/dequantized coefficients or end positions, and 1,248 complete + transform estimates with zero differences in rate, distortion, original energy, skip selection, and final cost. + Earlier reports that only requested DisableAVX512F do not establish execution of the 256-bit path on .NET 11. +- Final Release .NET 11 build passed with zero errors and 1,009 existing warnings. + `transform-estimate-final-r2.trx` passed 379/379 focused cases in 29.1507 seconds, including the shared + feature runner, frame/superblock paths, coefficient entropy, transform blocks, and motion search. + Evidence remains outside the repository in `D:\GitHub\ynse01\av1-takeover-20260905`: + `regular-quantization-comparison.json` and `transform-estimate-comparison.json`. + At the transform-estimator checkpoint, the production encoder still called the old effort-dependent search. + The subsequent production integration below connects frame/block coordination, DRL state, storage, and configuration. + No benchmark, encoder parity, or full-codec completion claim follows from these dependency results. + +- Checkpoint `105d76954` adds `Av1MotionSearchBase.SingleReference.cs`, which coordinates spatial/temporal starts, + duplicate-start history, frame/block step selection, full-pixel winners, DRL pruning, fractional refinement, + final interpolation and transform-RD selection, and retained results across three differential-reference choices. + It borrows source/reference planes and caller-owned arithmetic buffers; it adds no per-candidate owner. + `CollectStartingCandidates` retains partial temporal analysis, groups full-sample positions into rounded + eight-sample cells, weights the spatial start, and orders completed temporal votes by descending weight. + Reference control-flow evidence is `motion_search_facade.c:47-118,120-544`; normal coding disables the optional + full-pixel cost list (`speed_features.c:2360`, `encoder.h:4240-4244`), so the controller passes an empty span. + Full-result DRL pruning resolves from `speed_features.c:781,890,971,997`, including the inclusive 480-line boundary. +- Temporary tooling now calls the actual optimized `av1_single_motion_search`, its final predictor, and its + transform estimator. With reconciled block inputs, cost tables, speed policy, frame/block step parameters, + reference choices, temporal analysis, and header costs, all 2,646 decisions match exactly across 8/10/12-bit + samples, 8/16/64-square blocks, seven speed levels, integer/fractional motion, and incomplete temporal analysis. + All compared vector components, syntax rates, skip decisions, start counts, and retained full-search values + have maximum difference zero and zero mismatches (`single-motion-comparison.json`, temporary directory above). + Final .NET 11 Release VSTest run `single-motion-final-r3.trx` passed 467/467 focused cases in 33.9823 seconds; + the final build had zero errors and 1,009 existing warnings. The complete full-pixel and fractional comparisons + also match exactly in all 486 and 1,944 cases respectively, with zero maximum errors and zero mismatch counts. + This run used the host SIMD configuration; it does not independently establish each disabled-intrinsic path. + The complete staged checkpoint was reviewed before committing; it contains 18 managed source/test files and + excludes temporary native source, integration, libraries, executables, builds, and generated comparison data. +- Production integration now replaces FindInterMotionVector and its SSE/radius loop in + Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:490-1022. NEWMV uses the coordinated controller with + frame-relative bounds, live reference-stack rates, local coefficient contexts, predictor filters, and retained DRL results. + The existing LAST-frame candidate path now follows nearest/new/near/global ordering and no longer gates + nearest, near, or new motion on Effort. Reference order: rdopt.c:111-143. Reference-dependent range reduction + follows rdopt.c:1240-1271; frame/spatial steps follow encoder.c:2010-2042 and rd.c:1130-1203. + Prediction storage borrows the inactive intra workspace beyond the retained inter candidates, with 17,408 + samples available and no additional owner. Byte and ushort block operators use the existing motion operator family. +- Av1TileEncoder.Encode:246-303 resolves frame policy and initializes retained motion history. Actual packed + NEWMV syntax updates the absolute whole-sample maximum in Av1TileWriter.WriteModesBlock. Trials and + inherited/global vectors do not contribute. Reference evidence: encodemv.c:251-273 and bitstream.c:3953-3967. + Source classification is now carried separately from palette/intra-copy syntax eligibility; the detector + includes the eight-sample-aligned source extent. Reference evidence: encoder.c:2047-2110,2445-2484 and + aom_scale/generic/yv12config.c:247-248. Existing Effort-based screen-tool eligibility remains a deviation. +- Release .NET 11 build passed with zero errors and 1,009 existing warnings. The final focused run + motion-production-r2.trx passed 494/494 cases in 36.0448 seconds. Thirty-nine regenerated streams exercise + all ten speed levels at 8/10/12 bits, three-frame 4:2:0 prediction, and additional 4:2:2/4:4:4 cases. + Optimized native decoding and managed decoding match byte-for-byte for all 117 frames and 86,859 samples: + maximum error zero, differing samples zero, samples exceeding one zero (production-motion-decoder-comparison.json). + The first run stopped at an old Effort-0 fixture that assumed GLOBALMV was the only available inter candidate. + The fixture now supplies an explicit entropy history favoring GLOBALMV and checks its cost against every + competing mode; all preexisting expected skip/rate/distortion/coefficient/pixel assertions remain intact. +- This remains the same active production milestone. Scaled references, production temporal analysis, + broader block/reference support, complete mode pruning and ordering, and speed-dependent entropy-cost refresh + remain open. Still-image and intra/global-motion Effort policies are not migrated. The production decoder + comparisons establish same-bitstream equality only, not separate-encoder parity, performance, or codec completion. + ## Takeover audit: 2026-09-05 The earlier checked boxes and measurements below are historical checkpoint reports, not accepted conclusions about -the current encoder or complete decoder. The fresh production-path audit is still in progress. No benchmark has -been run during this investigation, and the complete 738-file upstream diff has not yet received a line-by-line audit. +the current encoder or complete decoder. The fresh production-path audit is still in progress. The first decoder +measurement after source-led corrections is recorded below for 2026-09-07; no encoder benchmark has been run during +the takeover. The complete 738-file upstream diff has not yet received a line-by-line audit. ### Reference and worktree evidence @@ -219,7 +490,8 @@ Further quantization, distortion, and final-packing source comparison on 2026-09 - `Av1ForwardQuantizer.cs:88-236` and `Av1ForwardQuantizer.Operator.cs:98-316` implement the no-matrix fast quantizers. The rounding factor 64 agrees with `av1/encoder/av1_quantize.c:609-651` at sharpness zero; the regular quantizer's factor 48 is not evidence of a fast-quantizer rounding defect. The managed path - does not implement the reference's sharpness adjustment, regular quantizer, or quantization-matrix policy. + still lacks the complete quantizer-selection/optimization and quantization-matrix policy. Regular quantization + with sharpness is now implemented for the motion-winner estimator; its production integration remains open. - `Av1TransformBlockEncoder.cs:1175-1225` always ends lossy coefficient selection at fast quantization. Reference `av1/encoder/tx_search.c:2064-2391` derives trellis eligibility from segment, evaluation stage, normalized residual energy, and transformed SATD; chooses fast or regular quantization; measures coefficient @@ -345,24 +617,566 @@ Frame/block RD and decoder filter follow-up after `aa2ecf690`: Retained-state and cost-policy follow-up after `ef8b1a823`: +Implementation in progress on 2026-09-06: symbol and tile traversal now support separate write and CDF-update +operations. `Av1TileEncoder.cs:239-381` completes block analysis across all tiles before a second traversal +packs their bytes. `Av1TileWriter.BlockEncoding.cs:70-151` derives partitions from the completed mode grid +and loads retained prediction parameters. Palette tokens, selected motion references, and coefficient contexts +have production writers and consumers. `Av1PictureControlSet.cs:127-154` resets entropy edges without clearing +those decisions. Final frame filter selection and the complete reference search/rate controller remain missing. + +Allocation evidence for the picture-state extension: + +- `av1/encoder/encodeframe.c:1413-1430` allocates palette tokens only when screen-content tools are allowed, + using `MAX_SB_SIZE_LOG2`, rather than the selected frame superblock size. `tokenize.h:42-46,105-135` stores + one byte per token and reserves `ceil(width/128) * ceil(height/128) * 128 * 128 * min(2, planeCount)` bytes. + The managed extension uses this maximum-superblock rounding. Color token storage is 128 KiB + at 256x256 and 15.9375 MiB at 3840x2160 for color; monochrome requires half those amounts. +- `Av1EncoderPictureBuffer.cs:75-333,368-390` already owns a combined picture-state allocation and disposes it + on constructor failure and normal disposal. New regions are non-owning slices of that same allocation; + they add no allocator call, per-candidate rent, or independently disposable owner. All byte-length products + and sums use checked arithmetic. Existing fixed-geometry sequence reuse retains the allocation across frames. +- Final prediction parameters use a separate eight-byte `Av1EncoderBlockStruct` region; the frequently read + `Av1MacroBlockModeInfo` entry remains eight bytes. Block palette entries use the existing 50-byte + `Av1EncoderPaletteInfo` and the existing mode-allocation count. + Motion contexts use that same count only when motion-vector state is allocated. Native frame extensions use + mode-allocation geometry too (`encoder_alloc.h:36-57`) and retain four usable candidate entries, weights, + count, global vectors, and mode context (`block.h:245-259`). The 28-byte compact managed context covers the + currently implemented single-reference/IBC syntax; compound and complete reference-controller parity remain open. +- Palette token capacity bounds all block tokens because coded blocks partition each superblock, each active + palette plane emits at most one token per luma sample, and there are at most two palette planes. Token writes + must stay in their assigned superblock region. Existing coefficient storage uses the same raster superblock + decomposition (`Av1EncoderCoefficientBuffer.cs:32-59`). + - `Av1EncoderTransformBlockState.cs:12-43` uses four bytes for EOB and byte-sized transform type, leaving one padding byte. Reference `av1/encoder/encodetxb.c:593-625,714-730` records the neighboring skip context in bits 0-3 and DC-sign context in bits 4-5 beside each EOB. Its packer consumes those retained contexts (`encodetxb.c:296-306,410-421`). The managed structure can represent that state without increasing its size, - but storing it alone would not implement deferred packing; no unused state was added. + and both analysis and packing now use the spare byte. The broader verification run below covers these paths. - Reference palette-token allocation is conditional on non-statistics coding with screen-content tools allowed (`av1/encoder/encodeframe.c:1413-1430`). `tokenize.h:105-135` reserves up to two full-resolution planes in maximum-superblock-rounded storage. Tokens retain the selected context and color-order rank, including the first raw index (`tokenize.c:174-225,264-278`, `bitstream.c:353-368`); keeping only palette colors is insufficient. -- `Av1EncoderPictureBuffer.Reset` clears the complete mode grid and packed state (`:344-363`), so it cannot - be reused as the boundary between analysis and packing. Selected prediction fields remain in the reusable - `Av1EncoderBlockStruct` workspace, while `Av1EncoderBlockModeInfo` retains only its smaller neighbor subset. - These lifetimes must be reconciled together with palette tokens, selected MV state, and coefficient contexts. +- `Av1EncoderPictureBuffer.Reset` still clears the complete mode grid and packed state between frames. + The separate `ResetEntropyContexts` boundary preserves retained syntax between analysis and packing. - Native cost defaults are explicit controls as well as speed features: `av1/av1_cx_iface.c:391-394,550-553` differs between default and realtime configurations. `rd.c:724-758,824-851` combines controls with speed policy and initializes frame costs; `encodeframe_utils.c:1629-1689` suppresses block refresh when CDF updates are disabled. No new managed effort mapping or isolated cost-refresh threshold was introduced. +Verification of the staging change: + +- The focused frame/sequence, intra-superblock, coefficient/entropy, and HEIF encoder run passed 2,411 cases + (`encoder-staging-focused-r3.trx`). Native decoding matched 12 regenerated color sequences (21,348 samples), + eight partition streams (16,640 samples), and 23 palette streams (985 samples), with maximum error zero and + zero differing samples. These are same-bitstream/reconstruction checks, not separate-encoder parity. +- The first broad run stopped on the old packed-state allocation-size assertion. The ownership test now checks + the additional fixed-width regions, exact pointer offsets, conditional allocation, clean storage, and return + of the same two owners. The existing compact-mode eight-byte assertion remains unchanged. +- All 13 storage/token checks pass (`encoder-staging-storage-tokens-r3.trx`). The new token test poisons the + color map after analysis, verifies unchanged packed bytes and matching adaptive costs, and compares analysis + output with an empty range coder. It covers both palette planes with CDF updates enabled and disabled. +- Intermediate failures exposed an unconditional four-entry read from a shorter candidate-weight span and an + initial expansion of compact mode storage; both production defects were fixed. Test-development failures + involved size arithmetic and unsupported InlineArray value equality; clean storage now uses byte-span checks. +- Final Release/.NET 11 build: zero errors, 1,009 existing warnings. The broad run passed all 9,379 cases + (`encoder-staging-production-r2.trx`, 2.6931 minutes), including the final storage and token tests. + No benchmark or claim of complete codec correctness, reference-controller parity, or performance improvement. + +Further filter-controller source comparison on 2026-09-06: + +- `Av1LoopFilterBase.cs:75-265` now owns the shared boundary traversal and level arithmetic. Decoder and encoder + operators supply their existing mode state; both call the same byte/high-bit-depth deblocking operators. + `Av1LoopFilterEncoder.cs:24-140` consumes the retained encoder grid and component planes without allocating a + second mode map or copying samples. `Av1TileEncoder` applies the signaled levels after analysis and before + packing. Automatic level selection is still missing; production frame configuration still leaves levels zero. +- Native `av1/common/av1_loopfilter.c:197-215` selects lossless 4x4, luma block/variable-inter transforms, and + maximum chroma transforms. The encoder currently retains a uniform luma transform per block. Variable inter + transform trees and superblock filter-delta encoding remain unresolved capabilities, not new rejection rules. +- The shared decoder traversal passed 26 focused tests (`shared-loop-filter-r1.trx`). Restoration and film-grain + native reference checks covered 8,500,087 samples with maximum error zero and zero differing samples. + New encoder reconstruction tests cover 8/10/12-bit monochrome and 4:2:0/4:2:2/4:4:4 at 33x137, distinct + component/direction levels, sharpness, and reference deltas. Their final verification is still in progress. +- All 12 encoder-filtering cases pass (`encoder-loop-filter-r2.trx`). Optimized native decoding of their + exported bitstreams matches retained reconstruction byte-for-byte: 102,390 samples, maximum error zero, + zero differing samples, and zero errors above one. These remain same-bitstream comparisons. Subsequent + shared-traversal cleanup retains each resolved block mode through level derivation, avoiding another grid + lookup and mode-structure copy. Final verification after that cleanup passed 138 focused cases + (`loop-filter-and-sequence-final-r1.trx`, 9.9613 seconds) and all 9,429 explicitly selected AV1, HEIF encoder, + and sequence-parser cases (`av1-sequence-final-r1.trx`, 2.6884 minutes). The Release .NET 11 build had zero + errors and 1,009 existing warnings. The 12 regenerated filtering streams again matched native decoding + exactly across 102,390 samples. This verifies the selected-level application, not automatic level selection. +- The six restoration and nine film-grain reference streams were decoded again with the optimized native + executable after the final run: 8,500,087 samples, maximum error zero and zero differing samples. The managed + conformance path uses `Av1ReconstructionConformanceTests.AssertNativePlanesEqual` (`:4053-4164`) and + `AssertSampleEqual` (`:4187-4284`): it compares every native sample, checks the complete reference length, and + fails on any unequal value. These assertions use no one-unit tolerance. Native reference-file validation and + managed comparison to those same files form bounded decoder evidence, not a separate-encoder comparison. +- `av1/encoder/picklpf.c:62-209,348-401` searches filter levels from previous-frame levels, caches per-level SSE, + applies directional bias and step reduction, restores the complete coded plane after each candidate, and + searches combined luma before separate luma directions and chroma. The trial buffer is reused by the encoder. +- `picklpf.c:211-347,403-467` also depends on explicit filter controls, tuning/sharpness, speed-selected search + methods, frame layers, and retained reference levels. Those policies cannot be replaced with a fixed quantizer + formula or a newly invented effort threshold. `speed_features.c:2545-2548` supplies the full-image search + defaults; speed-specific overrides still need to be reconciled with the complete encoder configuration. +- `encoder.c:2868-2923,3786-3815` places level selection and deblocking before CDEF/restoration and final + packing. `picklpf.c:348-367` retains a bordered unfiltered frame and reuses it across trials/frames. This + owner has not been added yet; the current application stage requires no additional frame buffer. +- `encoder_utils.h:1051-1064` uses 64-pixel borders for non-resized all-intra, selected-superblock-width plus + 32 for inter sequences, and a separate resizing border. Managed `Av1EncoderFrame.cs:28` still fixes the + border at 64 for both roles. This is an allocation/layout deviation; no performance claim follows from it. +- At this historical checkpoint public Effort remained 0-10 and the configuration question was pending. + The user subsequently required the native cpu-used setting as an enum; see the active milestone above. + +Sequence recovery failure found during broader verification on 2026-09-06: + +- `shared-loop-filter-production-r1.trx` stopped after 9,373 passes on + `HeifSequenceParserTests.DecodeSkipsInvalidAv1SampleWhenImageDataErrorsAreIgnored`. Earlier 9,379-case runs + selected AV1 plus HEIF encoder tests, not sequence parser tests. They did not establish this behavior. +- The failure preceded codec dispatch: `HeifDecoderCore.FindSequencePrimaryItem` rejected the usable track + because the synthetic file had no still primary item. Current [libavif track selection](https://github.com/AOMediaCodec/libavif/blob/main/src/read.c#L5754-L5869) + selects track configuration and samples independently of primary-item availability. The + [AVIF sequence specification](https://aomediacodec.github.io/av1-avif/#avif-image-sequence-brands) still describes + MIAF's image-item requirement for conforming files; accepting a usable track is decoder tolerance, not a claim + that the synthetic file conforms. Production sequence output continues to include its primary image item. +- Identification and decoding now use the first timed sample as the animated root when no still item is + available. Existing independent primary roots retain their presentation, metadata, and frame behavior. + No expected output or failure assertion was weakened. The existing recovery test now also checks every + retained RGBA pixel against the valid standalone image and asserts animated-root metadata. +- The initial targeted parser/encoder run passed 119 cases (`sequence-primary-recovery-r1.trx`). After the final + pixel assertions, both the 138-case focused run and the 9,429-case AV1/encoder/sequence-parser run passed. + The recovery regression now establishes the retained pixels and animated-root metadata as well as frame count. + +Motion-controller follow-up: + +- `Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:1275-1412` bounds an effort-scaled staged search by + a fixed 64-pixel border. `:1426-1496` uses prediction SSE and full mode/vector RD cost for both integer and + fractional search. Native `mcomp.c:72-240,317-386` carries frame-relative MV limits, search sites, mesh policy, + separate SAD/error-per-bit scaling, and subpixel stop/iteration policy. Enlarging the allocation alone would + not correct that controller. The active milestone now includes the allocation correction as a dependency; + the complete controller replacement remains required. +- Roslyn references for `Av1ForwardTransformer.GetHadamard8x8Cost` currently resolve only to two tests. The + retained primitive is not on the production search path; its tests do not establish encoder pruning behavior. + +Mode and transform controller follow-up on 2026-09-06: + +- Current `Av1IntraSuperblockEncoder.ModeDecision.cs:613-625,708-726,743-753` completes luma and chroma intra + selection before inter evaluation. Native `rdopt.c:6196-6333` evaluates ordered inter candidates while retaining + reference costs, skip costs, prediction SSE, motion candidates, and estimated transform results. It then runs + motion refinement and deferred transform search (`:6338-6358`), gates and evaluates intra (`:6367-6375`), and + refines winners (`:6383-6389`). Reordering one call without those retained dependencies would not reproduce + this controller. +- Native `rdopt.c:3746-3892` restores each winner's mode and references, selects winner-stage transform policy, + recomputes luma/chroma transforms, revisits skip, replaces the old transform rates, and retains an improved + result. Managed `EncodeBlock` has no corresponding refinement phase. This is missing functionality, separate + from numerical defects in individual transforms. +- `Av1TransformBlockEncoder.cs:1175-1225` always ends lossy coefficient decisions at fast quantization. Native + `tx_search.c:2084-2191,2216-2235` selects trellis by segment, evaluation stage, normalized residual MSE and + transform SATD, then optimizes before distortion evaluation and coefficient-rate pruning. Final intra encoding + also optimizes before inverse reconstruction (`encodemb.c:825-886`) and only then publishes entropy context + (`:912-945`). Adding an optimizer after managed reconstruction or packing would be the wrong integration point. +- Native `rdopt_utils.h:607-709` distinguishes default, candidate, and winner evaluation, changes transform and + coefficient policy together, and invalidates cached RD records when the stage changes. `speed_features.c:2810-2846` + derives those staged parameters from the profile. GOOD speed zero already enables partition, reference, and + transform pruning (`:1113-1172`); full enumeration is not its baseline architecture. +- GOOD's filter selection remains full-image at baseline, uses coarse inter search from cpu-used 3 + (`speed_features.c:1364-1365`), and disables independent luma-direction search from 4 (`:1425`). ALLINTRA's + later quantizer-based choice is a different profile decision. Public Effort mapping remains unresolved. + +Entropy-cost and frame-controller source trace on 2026-09-06: + +- `Av1SymbolEncoder.cs:1444-1595` measures coefficients directly from its live distributions. Native + `rd.c:602-684` fills coefficient snapshots, including base-level decrement costs and cumulative base-range + costs plus decrement deltas. `txb_rdopt_utils.h:112-162,226-238` uses those deltas when comparing a coefficient + with its one-level reduction. A copied probability graph or an optimizer attached after reconstruction would + not reproduce this cost architecture. +- Native mode snapshots cover partition, luma/chroma, palette, transform, angle, segment, reference, and motion + syntax together (`rd.c:82-322`). Frame initialization reconciles codec controls with speed policy and fills + the required cost families (`:723-851`); superblock refresh respects disabled adaptation and each family's + update boundary (`encodeframe_utils.c:1621-1692`). The managed live-CDF queries have no equivalent boundary. +- Current still-frame ownership is in `Av1FrameEncoder.cs:604-775`; sequence ownership is in `:1493-1921`. + Both call the tile encoder through `:977-1049`, which currently contains analysis, selected deblocking, and + packing. The sequence then extends borders and swaps reconstruction/reference owners. The existing source, + reconstruction, coefficient, picture, and operation workspaces must be retained when adding the missing + frame-level decisions; replacing them with decoder state or extra frame copies is not required. +- Before the sparse follow-up below, the decoder inverse factory forwarded every non-DC lossy case into full transforms + (`Av1InverseTransformerFactory.cs:47-60,98-112`). The 256-bit driver uses eight Int32 lanes and traverses the + complete height, including the uncoded half of 64-point inputs (`Av1Inverse2dTransformer.cs:385-524`). + Native low-bit-depth AVX2 selects separate sparse row/column functions and limits transformed row batches from + EOB bounds (`av1/common/x86/av1_inv_txfm_avx2.c:1611-1708`; + `av1_inv_txfm_ssse3.h:171-220`). Default, horizontal-identity, and vertical-identity scans have different + bounds. This is an established dispatch/traversal gap; its timing contribution is unmeasured on the current tree. + +Sparse inverse-transform follow-up on 2026-09-06: + +- Native high-bit-depth dispatch selects separate axis kernels for one, eight, sixteen, and thirty-two supported + inputs (`av1/common/x86/highbd_inv_txfm_avx2.c:4081-4179`). Four-point axes fall back to their complete + narrower kernels (`:4214-4230`; `highbd_inv_txfm_sse4.c:5503-5715`). High-bit-depth mixed identity paths + retain a complete identity axis and use conservative support on the other axis (`:5226-5361`); the low-bit-depth + paths use the row/column scan bounds in both dimensions (`av1_inv_txfm_avx2.c:1796-1889`). +- All 36 stored managed coefficient-scan arrays were compared with their native arrays after converting native + column-major indices into managed raster indices: 5,936 entries, zero differences. Native diagonal support + tables and identity bounds are in `av1_inv_txfm_ssse3.h:96-219`. The larger-transform matrix-scan aliases in + the managed dispatch table occur outside the legal transform sets; they are not evidence of a reachable decoder + numerical defect. `Av1ScanOrder.InverseScan` currently has no production references. +- `Av1Transform2dFlipConfiguration.cs:382-502` now carries EOB-derived support bounds into the factory dispatch. + `Av1Inverse2dTransformer.cs:167-458` selects closed generic sparse DCT/ADST operators for both axes. + Thirteen operator variants cover DCT lengths 8/16/32/64 and ADST lengths 8/16, including the coded 32-input + network for 64-point DCTs. Zero branches are removed while preserving surviving rotations, rounding, and + stage clamps. DC DCT variants retain their cosine scaling and uniform inverse-stage clamp. +- Each operator declares its input prefix in the existing operator interface. The scalar, 128-bit, and 256-bit + traversals (`Av1Inverse2dTransformer.cs:515-896`) process only potentially nonzero row batches, initialize every + input consumed by a complete second-axis operator, and continue writing the entire active destination. + Existing caller-owned workspace and independently strided prediction/output contracts are retained. +- Release .NET 11 build after the final relevant edit: zero errors and 1,009 existing warnings. The focused + inverse-transform suite passed 996/996 cases (`sparse-transform-focused-r1.trx`). New checks cover every + permitted scan prefix, sparse scalar/128-bit/256-bit operators against complete scalar arithmetic, poisoned + unused inputs and scratch, and production reconstruction at EOB support transitions. These are managed + arithmetic and dispatch checks. The final native-backed reconstruction suite passed 93/93 cases + (`sparse-transform-reconstruction-r1.trx`). These fixtures cover the existing production reconstruction corpus; + they do not independently enumerate every sparse arithmetic input or establish encoder parity. +- Remaining architecture gaps include packed 16-bit low-bit-depth arithmetic, native intermediate layout/sizing, + and further SIMD coverage. The existing full raster workspace is not the native compact intermediate layout. + No performance improvement or complete decoder/encoder parity is claimed, and no benchmark has been run. + +Inverse-transform storage follow-up on 2026-09-06: + +- The decoder previously reserved the forward-transform maximum: 11,520 Int32 values, or 45 KiB. + `Av1TransformWorkspace.cs:49-64` now distinguishes inverse storage: two axis vectors and one raster, + with a maximum of 5,120 Int32 values, or 20 KiB. This removes 25 KiB from each block decoder's + existing contiguous workspace owner; it adds no owner or allocator request. +- All 75 inverse scalar/128-bit/256-bit operator overloads consume their input before writing stage scratch. + The traversal now aliases those lifetimes and sizes vectors from the active dimensions + (`Av1Inverse2dTransformer.cs:533-544,698-709,842-853`). Native high-bit-depth traversal also uses + in-place axis storage with local kernel stages (`av1/common/x86/highbd_inv_txfm_avx2.c:4112,4138-4146`). + The managed raster layout and Int32 low-bit-depth arithmetic still differ; the allocation reduction is + not a claim of complete native storage or SIMD parity. +- Operator parity checks now exercise aliased input/stage storage, and the production sparse reconstruction + checks use the exact inverse workspace requirement. The owner test checks its reduced exact size. + After the final edit, Release .NET 11 built with zero errors and 1,009 existing warnings; Roslynk reported + zero compiler errors. The focused inverse/owner run passed 1,000/1,000 in 6.2844 seconds + (`inverse-storage-focused-r1.trx`). The broader AV1, HEIF encoder, and sequence-parser run then passed + 9,908/9,908 in 2.7773 minutes (`inverse-storage-production-r1.trx`), including the reconstruction fixtures. + These are correctness checks, not codec timing measurements or separate-encoder acceptance. + +Film-grain presentation storage follow-up on 2026-09-06: + +- The two presentation-copy branches previously allocated the full 288-sample prediction border + (`Av1Decoder.cs:899-906,952-959` before this change). Native output creation requests even visible dimensions, + no border, and 16-byte row alignment (`av1/av1_dx_iface.c:768-795`; `aom/src/aom_image.c:128-186,217-224`). + The synthesized output is separate from the ungrained reference. This is a demonstrated allocation-role + deviation, not an unmeasured claim that a particular SIMD kernel is faster. +- `Av1FrameBuffer.cs:255-259,562-612` now creates a presentation layout with those dimensions and row strides + in the existing single owner. Monochrome continues to omit unused chroma planes. The byte count describes + managed plane storage, not native allocator alignment overhead or total process memory. For a 100x60 + eight-bit 4:2:0 presentation, the calculated plane storage changes from 660 KiB to 9.84375 KiB. +- `CopyVisibleTo` preserves destination origins, strides, and allocation capacity (`Av1FrameBuffer.cs:276-325`) + while copying the active picture. Both newly decoded and shown-existing grained frames use the compact + layout; retained references keep their original bordered storage. Color conversion consumes independently + located rows through `Av1PlanarSampleBuffer.cs:100-131`, so no additional pixel-conversion buffer was added. +- Removing the border exposed empty overlap rectangles that the old padding had made addressable. + The new overlap-edge regression failed with an out-of-range span at `Av1FilmGrainDecoder.cs:525` + (`grain-presentation-overlap-red.trx`). Native calls carry zero height/width for these regions and its sample + loops do no work (`grain_synthesis.c:1133-1166,1252-1305`). The managed region owner now forms sample spans + only for nonempty vertical strips/interiors, while preserving all boundary blending and outgoing state. +- Tests cover exact owner size/return, separate origins and strides, untouched padding, and 144 grain cases + at clipped overlap edges across 8/10/12-bit monochrome and 4:2:0/4:2:2/4:4:4. These compare compact output + with bordered managed output; they are not independent native vectors for every edge case. The reference + ownership test now compares actual visible samples instead of comparing differently sized whole buffers. +- After the final relevant edit, Release .NET 11 built with zero errors and 1,009 existing warnings; + Roslynk reported zero compiler errors. The focused frame/reference run passed 55/55 in 3.0082 seconds + (`grain-presentation-overlap-fixed.trx`). The AV1, HEIF encoder, and sequence-parser run passed 9,932/9,932 + in 2.8026 minutes (`grain-presentation-production-final.trx`), including the existing exact native-plane + grain fixtures and hardware fallbacks. Their retained references were previously verified against the same + optimized native revision: 3,113,847 samples, maximum error 0, zero nonzero errors. No new native arithmetic + result, codec timing, or separate-encoder parity is claimed. No benchmark was run. + +Active-frame allocation validation follow-up on 2026-09-06: + +- The decoder rejected a valid small frame when its sequence maximum would require an oversized contiguous + allocation. `Av1Decoder.ValidateSequence` and the frame-buffer constructor both validated sequence maxima + even though `ReadTile` already allocates the active upscaled width and height. Native ordinary frame setup + reads the size override and allocates that active frame (`av1/decoder/decodeframe.c:1995-2049`). The separate + maximum-sized neutral-reference allocation at `:4854-4904` is corruption recovery, not ordinary frame setup. +- The regression preserves the original 4x4 Orange fixture's tile bytes and writes a full sequence header + declaring 65,536x65,536 maxima with an explicit 4x4 frame override. Before correction it failed at the + sequence-capacity check (`active-frame-size-red.trx`). The optimized native decoder accepts both the original + and expanded-maximum payloads and produces identical output: 24 samples per payload, maximum error 0, + zero differing samples and bytes. The reference build has `CONFIG_SIZE_LIMIT=0`. +- `Av1FrameBuffer.cs:231-250` now validates the requested allocation dimensions. `Av1Decoder.ReadTile:767-774` + performs the same active-frame capacity check before allocating syntax state. Sequence-only validation + continues to reconcile codec configuration and color declarations. Requests that actually exceed the + contiguous allocation limit remain rejected; no allocation limit or malformed-input assertion was weakened. +- Final Release .NET 11 build: zero errors and 1,009 existing warnings; Roslynk: zero compiler errors. + Header/lifecycle/frame-buffer verification passed 80/80 in 2.9384 seconds (`active-frame-size-focused.trx`). + The final reconstruction suite passed 93/93 in 1.9015 minutes (`active-frame-size-reconstruction.trx`). + Fresh optimized-native comparison matches the corrected managed output for both payloads on every byte; + per-plane counts are in temporary `active-frame-size-comparison.json`. This is bounded decoder evidence, + not separate-encoder parity or a performance result. + +Restoration boundary ownership and precision follow-up on 2026-09-06: + +- Architectural deviation: boundary rows were allocated in each frame's completion operation and always + widened to ushort. Native `av1/common/alloccommon.c:299-348` retains above/below buffers in decoder state, + allocates all sequence planes when restoration is used, aligns each row to 32 samples after adding four + samples on each horizontal side, and replaces owners when their physical byte length changes. Stripe + allocation uses the mode-info-aligned luma height for every plane. Native decoder destruction releases + these buffers (`av1/av1_dx_iface.c:122-137`, `av1/common/alloccommon.c:350-366`). +- `Av1LoopRestorationBoundary.cs:81-149` now owns those buffers for the decoder session, with physical byte + storage at both sample precisions and the corresponding aligned layout. The frame/header arguments are + borrowed by each save operation; no reconstructed frame or header remains referenced by boundary storage. + Its save/copy methods at `:360-445` preserve byte or ushort samples directly, including super-resolution, + and replicate horizontal context. The allocation-failure path retains successfully acquired owners for + normal session disposal. `Av1Decoder` owns/disposes this object and passes it through `Av1FrameDecoder`. +- Fourteen focused cases passed in 2.6187 seconds (`restoration-boundary-focused.trx`): all bit-depth/chroma + combinations, exact contents across multiple stripes, same-size reuse, growth/shrink replacement, and + exactly-once returns after initial/replacement allocation failure. Release .NET 11 build: zero errors, + 1,009 existing warnings; Roslynk: zero compiler errors. Reconstruction and ownership verification passed + 162/162 in 1.8952 minutes (`restoration-boundary-reconstruction.trx`). A further independent-plane regression + reuses one decoder across 8/8/10/12/8-bit fixtures, exercising unchanged size, changed size, changed precision + at identical byte size, and return to byte samples. After that final test edit, the Release build again had + zero errors and 1,009 warnings; all 15 focused boundary/sequence cases passed in 3.6690 seconds + (`restoration-boundary-sequence-final.trx`). Fresh optimized native decoding matches the six retained + restoration/super-resolution fixtures: 5,386,240 samples, maximum error 0, zero nonzero errors and zero + errors above one. Managed tests independently compare current output to those same native reference planes. +- At this checkpoint, `Av1LoopRestorationDecoder.DecodePlane:102-233` still rented a per-plane output and + filter scratch, copied each bordered processing block into ushort storage, and narrowed filtered 8-bit + output. The later physical-sample correction below replaces those block copies. Native + `restoration.c:252-383` temporarily replaces and restores stripe + boundary rows in the source. Its destination frame persists with a 32-sample border + (`restoration.c:1069-1138`, `restoration.h:30`); its shared filter scratch and line-save buffers persist + (`alloccommon.c:299-310`). Correcting boundary ownership does not resolve these other paths. +- Native self-guided scratch is not just its two output arrays: `restoration.c:718-862` also uses local + source and intermediate arrays, while `:863-898` places its two outputs 161,588 integers apart. + `restoration.h:74-92` reserves about 1.233 MiB for those two outputs despite processing 64-sample chunks. + The optimized AVX2 path differs from scalar scratch ownership: `x86/selfguided_avx2.c:547-637` allocates + four aligned intermediate/integral planes per filter call and frees them afterward; the coefficient and + integral phases correspond to managed `Av1SelfGuidedFilter.Operations.cs:33-117`. Managed scratch combines + these with the two outputs in caller-owned storage sized to the actual processing block. Their complete + layouts and lifetimes must remain distinct in the comparison. No performance benefit has been measured, + and no benchmark was run. + +Wiener traversal follow-up on 2026-09-06: + +- SIMD coverage/architecture finding: the previous `Av1WienerFilter` reduced an eight-tap SIMD dot product + to one horizontal output and performed every vertical dot product scalarly. Native optimized filtering + processes independent columns together in both passes (`av1/common/x86/highbd_wiener_convolve_avx2.c:28-245`, + `av1/common/x86/wiener_convolve_avx2.c:43-242`). Native scalar equations and intermediate clipping were + traced in `av1/common/convolve.c:1350-1519`. No numerical defect was established in the prior arithmetic. +- `Av1WienerFilter.cs:57-116` now sends both passes through one closed generic traversal + (`Av1WienerFilter.Operator.cs:172-329`). The semantic `WienerOperator` at `.WienerOperator.cs:17-162` + implements symmetric tap arithmetic at scalar/128/256/512 widths. The driver owns row/column access, + complete vector bounds, width selection, coefficient broadcasts, and one scalar remainder. + The operator widens before symmetric additions and retains signed 32-bit sums through rounding/clipping. +- The implicit center contribution is incorporated into each kernel before filtering. First-pass bias, + intermediate clipping, the 12-bit first-pass shift of five, and final bias removal remain at their + original stages. The source and intermediate layouts and requested scratch size are unchanged. + Portable lane order remains sequential; it does not reproduce x86-specific lane permutations. + Wider dispatch is not evidence of an end-to-end performance improvement. +- New focused verification uses a separate complete eight-slot, 64-bit scalar calculation across legal + coefficient extrema, mixed signs, both pass orderings, 8/10/12-bit precision, vector boundary widths, + odd heights, clipping, output padding and scratch sentinels. The first focused run failed its explicit + intermediate-clipping coverage assertion: random/checkerboard inputs had not reached that boundary. + No compared sample had mismatched before the assertion. The failing remote test process subsequently + timed out after 60 seconds (`wiener-vector-focused.trx`); its failure-reporting log also contains an + access-denied message. The identified child process IDs were absent after the failed test completed. + The test now adds sparse impulses and inverse impulses to force both clipping endpoints and runs the + matrix in the VSTest host before the restricted hardware child runs. No assertion or expected output was + weakened. The corrected matrix passed (`wiener-vector-coverage-fixed.trx`, 3.2988 seconds total test time). + Release .NET 11 build: zero errors, 1,009 existing warnings. Final reconstruction/ownership/row-filter + verification passed 164/164 in 1.8839 minutes (`wiener-vector-reconstruction.trx`), including the six + restoration/super-resolution streams compared to byte-exact native planes. Their fresh optimized-native + agreement remains 5,386,240 samples with maximum error 0 and zero differing samples. The filter matrix + runs the current hardware path before restricted hardware configurations and scalar fallback; unavailable + host instruction sets cannot be established by feature-disable runs. No performance comparison was run. + At this checkpoint, byte-source integration, restoration output/scratch lifetime, and per-block widening + remained open. The subsequent correction below addresses sample storage and stripe dataflow; this + traversal verification does not establish encoder parity. + +Restoration physical-sample and stripe dataflow correction on 2026-09-06: + +- Architectural deviation: the prior controller materialized every bordered processing block as ushort + samples, then copied/narrowed each result into a per-plane output. Native + `av1/common/restoration.c:252-383,985-1051` filters the original physical sample storage after saving and + replacing only three context rows above/below each stripe, then restores those rows before another unit + reads them. The fixed six-row save area is 4.594 KiB + (`av1/common/restoration.h:211-219`), irrespective of sample precision. +- `Av1LoopRestorationDecoder.cs:77-110` now selects byte/ushort once for the complete frame traversal. + `:120-236` extends the existing reconstruction border, allocates output at physical sample precision, + and copies the completed plane back after all units consume its original samples. + `:259-402` saves/replaces/restores the bounded stripe context and gives both filters direct strided + views of the source and destination. It no longer materializes bordered blocks or widens frame rows. + `Av1LoopRestorationBoundary.cs:354-365` exposes the saved rows including horizontal context. +- Both filter families retain their existing arithmetic scales and scratch representation while carrying + the selected physical sample type through source reads and output writes. Shared register operations in + `Av1RestorationSampleOperations.cs:20-253` load only the samples owned by each batch, widen unsigned values + in registers, and narrow only after clipping. No numerical mismatch had been established in the prior + ushort arithmetic; this corrects storage and traversal rather than changing the filter equations. +- Release .NET 11 build: zero errors, 1,009 existing warnings, 31.83 seconds. Independent Wiener full-kernel + and self-guided direct-window tests now include byte-backed sources/destinations, row padding and outer + sentinels. Both passed with their available hardware and scalar configurations + (`restoration-physical-kernels.trx`, 4.0156 seconds). Final reconstruction, ownership, and filter + verification passed 165/165 in 1.9481 minutes (`restoration-physical-reconstruction.trx`). The six + restoration/super-resolution native-plane fixtures are byte exact with the edited managed decoder. + Fresh optimized libaom decoding independently matches those retained planes across 5,386,240 samples: + maximum error 0, nonzero errors 0, errors above one 0 (`restoration-comparison.json`). +- At this checkpoint, output, line-save, and arithmetic workspace still rented per active plane. The later + output-frame correction below addresses destination lifetime/layout; scratch remains unresolved against + `alloccommon.c:299-310`. The distinct + optimized self-guided temporary allocations described above must not be conflated with its shared output + workspace. No performance comparison or encoder parity claim follows from this change. + +Restoration output-frame lifetime follow-up on 2026-09-06: + +- Architectural deviation: restored output was rented and returned for every active plane. Native + `av1/common/restoration.c:1069-1138` retains the destination frame in decoder state, uses a 32-sample + border, and copies only restored planes back after all filtering. Its allocation grows only when the + required byte capacity increases; new storage is cleared, while reuse does not clear or copy previous + samples (`aom_scale/generic/yv12config.c:59-267`). +- `Av1FrameBuffer.cs:281-332` now creates/resizes restoration output using its existing single-owner + plane abstraction. The shared layout calculation at `:628-685` retains eight-sample coded alignment, + 32-sample luma row alignment, subsampled chroma strides, and the restoration border. Growth releases the + previous allocation before renting the new one. A failed rent leaves the target empty and recoverable + on a subsequent resize; no second complete output frame is retained during growth. +- `Av1Decoder` owns this output independently of reference/presentation frames and disposes it with the + session. `Av1FrameDecoder.CompleteFrame` prepares it after super-resolution and before restoration. + `Av1LoopRestorationDecoder.DecodeFrame:101-139` publishes only active restored planes after all units, + preserving its source until filtering finishes. The output no longer rents per plane. +- Monochrome continues to allocate only luma through the established managed frame abstraction; the native + generic destination allocator reserves chroma even in this case. No unused monochrome chroma owner was + introduced. This is an explicit sizing difference, not an assertion that allocation layouts are identical. + Line-save and filter arithmetic scratch still need their separate lifetime comparison. +- Two Release builds stopped on StyleCop enum ordering before tests: first an enum after methods, then a + constructor after the enum. The enum is now between constructors and properties; Roslyn's analyzer pass + reports zero errors. Fourteen focused layout/ownership/sequence-change cases passed + (`restoration-output-ownership.trx`, 3.6731 seconds), including unchanged-layout view reuse, growth, + shrinkage, byte/ushort transitions, clean new allocations, preserved reused storage, and recovery after + allocation failure with exactly-once returns. Two new assertion-style warnings were corrected without + changing their checks. Final Release .NET 11 build: zero errors, 1,009 existing warnings, 13.67 seconds. + Broader reconstruction/ownership/filter verification after that final edit passed 178/178 in + 1.9329 minutes (`restoration-output-final.trx`), including byte-exact native restoration fixtures. + +Restoration stripe-save lifetime follow-up on 2026-09-06: + +- Native `av1/common/alloccommon.c:299-310,350-366` retains one line-save allocation for the decoder session. + `restoration.h:211-219` specifies six rows of 392 ushort slots, 4.594 KiB total, even for byte samples. + The managed controller previously included that storage in every active plane's temporary Wiener owner. +- `Av1LoopRestorationBoundary` now owns that fixed save area alongside its preserved boundary rows, + initializes ownership before saving deblocked rows, shares it with every restoration plane, and returns it + at session disposal. The per-plane Wiener owner now contains only its convolution intermediate. +- Existing exact allocation/content tests include the additional session owner, its fixed size, and reuse + across dimension changes. Failure cases now target the first save-area allocation and the initial/replacement + below-row allocations after it. Exactly-once return assertions remain in place. Release .NET 11 build: + zero errors, 1,009 existing warnings, 33.80 seconds. All 51 frame-buffer and sequence-change cases passed + (`restoration-save-ownership.trx`, 3.8032 seconds). Final reconstruction/ownership/filter verification + passed 179/179 (`restoration-save-final.trx`, 1.9193 minutes). + +Self-guided work-row alignment correction on 2026-09-06: + +- The managed integral/coefficient rows reserved multiples of sixteen integers, although the widest + implemented arithmetic batch contains eight. The optimized reference uses eight-integer row alignment + after its border and row-separation columns (`av1/common/x86/selfguided_avx2.c:547-637`). + `Av1SelfGuidedFilter.GetBufferStride` now retains that eight-lane alignment. At 64x64 this changes the + combined managed scratch request from 138.5 KiB to 129.625 KiB. The native shared output and per-call + temporary allocations remain distinct from this combined managed workspace; their lifetimes are unresolved. +- The independent direct-window matrix now includes 32x32 and 64x64 processing blocks and poisons scratch + inside two outer sentinels for both physical sample types. An initial Release build stopped on SA1515 + because a comment followed an expression-bodied method declaration without the required spacing. + The explanation now sits inside the method body. The corrected Release .NET 11 build passed with zero + errors and 1,009 existing warnings (35.64 seconds). The expanded filter matrix passed + (`restoration-alignment-kernel.trx`, 3.2735 seconds), followed by 179/179 reconstruction/ownership/filter + cases (`restoration-alignment-final.trx`, 1.9455 minutes). No benchmark or encoder parity check was run. + +Additional production-path source checks on 2026-09-06: + +- Restoration stage lifetime follow-up: `Av1LoopRestorationDecoder.cs:26-139` now follows the existing + session-owned CDEF stage pattern. It owns the restored output and grows/reuses the two managed arithmetic + workspaces; borrowed source/header/unit state stays in call parameters. `Av1FrameDecoder.CompleteFrame` + invokes the retained stage, and `Av1Decoder` disposes it with the session. Native persistent self-guided + outputs are allocated in `av1/common/alloccommon.c:299-305` and freed at `:350-366`. Native AVX2 temporary + integral/coefficient allocations (`x86/selfguided_avx2.c:547-637`) and Wiener stack intermediates + (`x86/wiener_convolve_avx2.c:43-242`, `x86/highbd_wiener_convolve_avx2.c:28-245`) remain a separate layout + and lifetime difference: managed code retains these bounded workspaces to avoid block-level rents and + oversized stack storage. This is not a claim of identical native allocation layout or measured speed. + The sequence regression first failed its early-return assertion (`restoration-scratch-before.trx`). + After correction it requires one retained 4,544-ushort workspace and one 33,184-integer workspace across + all five 8/8/10/12/8-bit frames, byte-exact native-plane output, and exactly-once disposal. Six additional + cases reject each initial/growing output or scratch rent and verify retry, unchanged-input copying, + steady-capacity reuse, and balanced returns. All seven passed (`restoration-scratch-failure.trx`, 3.5712 + seconds); final reconstruction/ownership/filter verification passed 185/185 (`restoration-scratch-final.trx`, + 1.9700 minutes). Final Release .NET 11 build: zero errors, 1,009 existing warnings, 8.39 seconds. + Roslynk reports zero compiler errors. No benchmark or separate-encoder parity check was run. + +- Frame-header serialization is already after analysis: both `Av1TileEncoder` sample-type constructors complete + `Encode` before `Av1FrameEncoder` calls the OBU writer; `GetTileData:239-244` only exposes retained bytes. + The fact that `ObuWriter.WriteFrameObu:116-142` writes header scratch before querying tile lengths therefore + does not establish another ordering defect. Automatic filter selection remains missing as documented above. +- Tile-list decoding remains missing functionality. `ObuReader.cs:388-391` rejects it, while native + `av1/decoder/obu.c:479-591,1073-1092` supports its camera-header/external-reference path when normal-only + tile mode is disabled. The optimized reference has `CONFIG_NORMAL_TILE_MODE=0`. The existing rejection test + proves the current restriction, not implementation of that mode; its container/API integration is unresolved. + +CDEF controller and storage follow-up on 2026-09-06: + +- Encoder CDEF remains missing. `Av1FrameEncoder.cs:372-408` disables it, and `Av1TileEncoder.cs:261-270` + currently runs only selected deblocking between analysis and packing. Native `encoder.c:2777-2825` preserves + restoration boundaries, searches CDEF, applies it, performs super-resolution, preserves the later boundaries, + and searches restoration. Enabling the sequence flag or calling a filter primitive does not implement that path. +- The complete CDEF search was read in `av1/encoder/pickcdef.c` and `pickcdef.h`. Search excludes wholly skipped + units and groups 128-wide/high blocks (`pickcdef.h:173-218`, `pickcdef.c:523-647`). Distortion covers only the + listed non-skipped blocks; high-bit-depth squared error is shifted after accumulation + (`pickcdef.c:236-261,397-518`). Luma directions are reused across strengths and chroma + (`:543-617`). Joint strength selection includes per-unit index bits and frame strength literals; distortion is + scaled by 16 before RD comparison (`:897-973`), followed by per-unit selections, adaptive decisions, and strength + remapping (`:976-1099`). These dependencies remain absent from production encoding. +- Search distortion arrays and unit indices are allocated per search and freed afterward + (`pickcdef.c:655-680,1099`), while the search-context object persists (`:891-894`). Application buffers have a + different lifetime: `av1/common/alloccommon.c:192-278` retains them, replaces changed sizes, and releases them + when CDEF is disabled. The decoder invokes that boundary after its final tile (`decodeframe.c:5396-5402`). + These two allocation lifetimes must not be conflated when integrating encoder search and filtering. +- The managed application stage previously rented its entire scratch region inside each `DecodeFrame` + (`Av1CdefDecoder.cs:145-157` before this edit). It now belongs to `Av1Decoder` and accepts frame inputs + per call, retaining no frame references. Equal storage requirements reuse the owner, resizing returns the old + allocation, and disabling CDEF or disposing the decoder releases it. Filter arithmetic and traversal are unchanged. +- Remaining application differences are explicit: managed scratch still uses a 68-sample stride and two-column + border, two saved top-row slots, and block-local sample-width dispatch. Native application uses its aligned + source/line/column layouts and a separately captured bottom border + (`av1/common/cdef.c:161-253,369-421`; `alloccommon.c:209-231`). This lifetime correction does not establish + matching sizing, complete shared encoder/decoder traversal, SIMD coverage, or a measured timing improvement. +- The production reuse test decodes 8-bit 4:2:0 twice, then 10/12-bit 4:4:4, then returns to the smaller frame. + It checks live allocator identities, exact-once returns on resize/disposal, and every sample against native + references. All 11 focused cases passed after the final edit (`cdef-lifetime-focused-r1.trx`, 13.4479 seconds). + Release .NET 11 compiled with zero errors and 1,009 existing warnings. The final wider reconstruction run passed + all 93 cases (`cdef-lifetime-reconstruction-r1.trx`, 1.8393 minutes), including reference frames, restoration, + film grain, intrinsic fallbacks, and constrained allocators. Roslynk reports zero compiler errors. +- Fresh optimized-native decoding of the three CDEF streams matches 3,219,456 reference samples exactly: + maximum error zero, zero differing samples, zero exceeding one, and zero differing bytes. The comparison script, + output, and `cdef-comparison.json` remain outside the repository. No benchmark or encoder-parity claim follows. +- The remaining quantizer-dependent profile policy was read in `speed_features.c:2892-3134`. It changes motion + search, partition eligibility, coefficient optimization, winner transforms, and restoration-unit size using frame + role, dimensions, screen-content state, and qindex together. Public Effort mapping is still unresolved; copying + any one threshold would not establish the required encoder policy. + +Decoder intra-block-copy traversal correction on 2026-09-06: + +- Before this edit, `Av1BlockDecoder.cs:1211-1277` rebuilt the source coordinates and dispatched prediction for + every transform. Native `av1/decoder/decodeframe.c:681-693,852-879,971-973` predicts the complete block planes + before residual traversal. Plane dimensions are at least four samples (`av1/common/av1_common_int.h:1343-1355`); + displacement validity covers the full source rectangle and decoded-region delay + (`av1/common/mvref_common.h:279-337`). This establishes the traversal boundary without a timing hypothesis. +- Prediction now runs once for each participating block plane, before the transform loop, using the existing + byte/high-bit-depth predictor and frame spans. Transform reconstruction and CfL publication keep their order. + No new owner, buffer, copy of a frame, search policy, or rejection rule was introduced. +- The native-fixture regression now requires actual IBC blocks with multiple luma transforms at every tested + precision, in addition to comparing every sample. All 25 focused cases passed after the final edit + (`ibc-plane-prediction-r1.trx`, 12.6576 seconds), including half-sample chroma production tests and the official + extreme-displacement sequence. Release .NET 11 compiled with zero errors and 1,009 existing test warnings. +- Fresh optimized-native comparison of the three 512x256 4:4:4 streams and the two-frame official sequence + matched all 7,400,448 retained reference samples: maximum error zero, zero differing samples, and zero differing + bytes. The local comparison script, extracted payloads, and report are outside the repository. Final wider + reconstruction verification passed all 92 cases (`ibc-plane-reconstruction-r1.trx`, 1.8239 minutes), including + native precision, presentation, intrinsic tiers, sequence references, and constrained allocator cases. + No encoder-parity or measured-performance claim follows from this change. + Decoder coefficient-stage correction after `b305e6e89`, verified on 2026-09-05: - The source trace established an architectural deviation: `Av1SymbolDecoder.cs:1415-1459` published a @@ -414,6 +1228,271 @@ Decoder coefficient-stage correction after `b305e6e89`, verified on 2026-09-05: different buffer lifetimes (`:3244-3277`). Those traversal, clearing, and worker-lifetime differences remain open; this checkpoint does not claim that changing coefficient representation completes them. +Decoder CDF storage correction in progress on 2026-09-07: + +- `Av1Distribution.cs:39,274-303,328-335,395-422` stored sixteen inverse cumulative thresholds as unsigned + 32-bit values. Their complete domain is 0 through 32768. Native `aom_dsp/prob.h:29,110-137` uses unsigned + 16-bit storage and wider arithmetic during adaptation. The managed field and copy views now use the existing + `InlineArray16`; initialization narrows after inverse conversion. The indexer and update arithmetic + retain their wider types. The threshold region is 32 bytes instead of 64 bytes per distribution. + Alphabet size, counters, object identity, copying boundaries, and CDF adaptation are unchanged. +- This follows a current-tree decoder measurement after the residual-skip verification below, using the existing + benchmark and optimized native adapter. Both paths include construction, parsing, reconstruction, RGB48 output + allocation/conversion, and disposal. The native path uses the same ImageSharp color converter, so packed output + equality does not independently validate that converter. Both inputs passed exact packed-sample comparisons. +- Before the CDF-width edit, the three-iteration Short job measured 13.739 ms managed versus 5.133 ms native + for `libavif-kodim23-8b.bit`, and 24.848 ms versus 9.263 ms for `libavif-cosmos1650-10b.bit`. + The approximately 2.68x gap remains open. Managed allocation counts were 933.1 KiB and 1014.38 KiB; + native allocations are not included in the managed counter. No historical speedup is established by this run. + CPU model lookup and power-plan configuration were denied by the environment. The baseline report is + `D:\GitHub\ynse01\av1-takeover-20260905\decoder-current-short\20260907-011627`. + BenchmarkDotNet 0.15.8 used the in-process toolchain, .NET 11 preview 7, x64, one coding thread, three warmups, + and three measured iterations. Production assembly SHA-256 was + `4264B2B47188E8B5B15B24E55D9B5CDA468C407D9F48EE02F928DB977BBFDE5A`, matching the verified test assembly. +- The complete CDF object graph still differs from the flat native frame context (`av1/common/entropymode.h:71` + onward). `Av1FrameEntropyContexts.cs:31-37,99-121` retains three active graphs plus independently owned + reference snapshots, and `Av1FrameEntropyContext.cs:140-207` constructs each distribution separately. + Reducing threshold width does not resolve graph shape, cost-table policy, or the remaining codec controller. +- Decoder coefficient and transform-state storage is already bounded to one active superblock + (`Av1FrameInfo.cs:250-338,651-698`, `Av1SuperblockInfo.cs:50-94`). Reference-state transfer retains motion and + segmentation data, not that scratch (`Av1FrameInfo.MotionField.cs:491-581`). Native single-thread decode also + uses one superblock buffer (`decoder.h:45-108`, `decodeframe.c:2504-2524,2794`), retained in worker state. + Managed frame-owned allocation lifetime remains different; no per-reference coefficient-copy defect was found. +- Release .NET 11 builds after the CDF-width edit with zero errors and zero reported warnings (24.25 seconds). + All 2,298 focused entropy cases pass (`cdf-width-focused.trx`, 4.6192 seconds), followed by all 9,997 AV1 and + selected HEIF cases (`cdf-width-production.trx`, 2.9274 minutes). Roslynk reports zero compiler/analyzer errors. + No tests or expected output were changed. Fresh optimized native checks preserve exact agreement across + 8,521,435 established restoration, film-grain, and regenerated moving-color samples: maximum error zero, + no differing samples, and zero errors above one. The retained-reference comparison and direct regenerated-output + comparison boundaries are unchanged from the residual-skip checkpoint below. +- The post-edit Dry and Short benchmark jobs both pass exact packed output for the two existing inputs. + Across 2,494,464 RGB48 components, maximum error is zero, with no differing samples or errors above one. + Both decoders use the shared ImageSharp converter. Independently re-decoding those inputs also reproduces their + 1,904,640 retained native plane samples exactly (`decoder-benchmark-inputs.json`). + Current managed/native means are 13.296/5.099 ms for Kodak and 25.419/9.448 ms for Cosmos. + Managed allocations are 715.97/797.24 KiB, about 217 KiB less per input than before the storage correction. + Three iterations and the recorded timing variance do not establish a speedup or close the performance gate. + Full results and error intervals: `decoder-current-short\20260907-012459` in the temporary takeover directory. + The post-edit production assembly SHA-256 is + `8469641ACE0351349694D766BB9395DD14800CA3BDFF30E78EF74A0C9D1CE7F5` in both the test and benchmark outputs. +- Further CDF-layout comparison: native `av1/common/entropymode.h:71-167` embeds differently sized alphabets, + their zero sentinels, and their observation counters into the frame context. `decoder/decoder.c:110-116` + allocates its base/default contexts once with 32-byte alignment. Tile completion copies the selected state, + resets counters, and publishes independent retained-frame state (`decoder/decodeframe.c:5488-5507`). + Managed `Av1SymbolReader.cs:112-123` and `Av1SymbolWriter.cs:139-145` hold live distribution aliases; + `Av1FrameEntropyContext.cs:545-690` copies/reset counts through those stable objects. Tests explicitly check + independent snapshot adaptation (`Av1EntropyTests.cs:70-119`, `Av1MotionModeEntropyTests.cs:98-194`). + Flattening storage must preserve those aliases and independent counters together. No distribution-type, + frame-ownership, or CDF-layout refactor beyond the verified width correction has been applied. + +Encoder residual-skip correction in progress on 2026-09-07: + +- The inter block decision required an empty quantized transform on every plane before considering prediction-only + reconstruction. This excludes valid lower-cost skip decisions with nonzero coefficients. The starting path was + `Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:1230-1266`; its IBC caller has the same restriction at + `:327-356`. Native `av1/encoder/tx_search.c:3875-3898` forces skip for an all-empty result and otherwise + compares residual syntax plus distortion against prediction-only SSE for lossy blocks, including ties. + Both inter and IBC use that decision (`av1/encoder/rdopt.c:1734,3490`). +- The bounded inter correction now measures prediction-only SSE before transform scratch is reused, normalizes + high-depth error with rounding, retains four fractional distortion bits, compares costs without shared prediction + syntax, and publishes prediction samples with cleared coefficients and default transform state when skip wins + (`EvaluateInterCandidate:1020-1243`, `EvaluateInterPlane:1547-1724` in the current file). + IBC now uses the same residual decision at `SelectIntraBlockCopy:318-367`. No search order, effort mapping, partition policy, + reference role, quantizer, buffer allocation, or encoded-sample acceptance threshold changed. +- The new regression initially failed at qindex 64 and perturbation amplitude 4: skip residual cost 938,631 versus + coded residual cost 1,001,301, with nonzero coefficients in every legal transform + (`residual-skip-red.trx`). The full quantizer/perturbation matrix remains and now runs at 8, 10, and 12 bits. + Its oracle computes prediction SSE and rounded costs explicitly, and checks selected rate, distortion, cost, + cleared state, and byte-exact retained prediction. Transform arithmetic and entropy primitives are shared with + production, so this establishes the block-decision regression only, not independent encoder parity. +- Automatic approval review rejected the initial combined inter/IBC replacement as too broad, and separately + rejected narrowing the test to one pinned input. Neither rejected edit was applied. The safer inter-only patch + passed a content check and was accepted; the original test matrix was retained and expanded to all three depths. + The expanded test initially failed compilation because the generic span assertion selected an incompatible + overload (CS9244); comparing the same complete sample spans as bytes resolved that without changing expectations. +- The inter-only focused encoder set passes 251/251 (`residual-skip-inter-focused.trx`, 21.9169 seconds). + After the precision-test expansion, Release .NET 11 builds with zero errors and 1,009 existing warnings; + all three precision cases pass (`residual-skip-precision.trx`, 2.8044 seconds). + Twelve regenerated moving-color streams match optimized native decoding across 21,348 samples, maximum + error zero, no differing samples, and zero errors above one. Final verification is recorded below. + No benchmark or separate-encoder parity measurement has been run. +- A separate IBC regression now reaches production pixel candidate search using completed reconstructed blocks + before the target superblock. After correcting its coefficient-buffer width, it failed at qindex 64 and amplitude 4: + skip residual cost 940,455 versus coded residual cost 1,005,028, with nonzero coefficients in every legal transform + (`residual-skip-ibc-setup-corrected-red.trx`). The subsequent IBC-only correction was accepted by automatic review. + The first post-fix rate check then exposed a missing target-grid mapping in the test setup; adding the same + `MapModeInfoBlock` operation used by production traversal corrected the displacement lookup. No expected value + or tolerance was relaxed. All six inter/IBC precision cases and the two existing full-RD/half-chroma IBC cases pass + (`residual-skip-reference-mapped.trx`, 8/8, 3.4124 seconds). + The final build has zero errors and 1,009 existing warnings (10.76 seconds); Roslynk reports zero compiler + and analyzer errors. All 9,997 AV1 and selected HEIF cases pass after the final C# edit + (`residual-skip-production.trx`, 2.7919 minutes). + Fresh optimized-native checks match all 8,521,435 established samples exactly: maximum error zero, + zero differing samples, and zero errors above one. The restoration and film-grain scripts re-decode the + retained references also exercised by the current managed tests; the moving-color script compares + regenerated managed output directly. These checks do not establish separately encoded output parity. + +Decoder coefficient reuse correction in progress on 2026-09-06: + +- Neighbor-context storage-width correction in progress on 2026-09-07: managed above/left entropy, partition, + and transform contexts used `int` entries. Native `av1/common/entropy.h:51-52,80` and + `av1/common/enums.h:168,522` define byte-sized entries. Partition masks require five bits, transform extents + reach 128, and coefficient contexts pack three saturated level bits with sign class 0/1/2, for a maximum of 23 + (`av1/common/txb_common.h:274-280`, `av1/decoder/decodetxb.c:316-319`). + The two existing context owners and their tile-reader/entropy-decoder spans now use bytes. + Coefficients and arithmetic retain their existing precision. Explicit casts occur only after these established bounds. + At 3840-pixel width with three planes, the above-context surface changes from 18.75 KiB to 4.6875 KiB; + owner count, region strides in entries, and allocation lifetime are unchanged. This is a storage-size correction, + not measured performance evidence or completion of the remaining context-lifetime work. + Existing coefficient regressions now derive packed edge contexts independently from original quantized values. + Their first run exposed an incorrect six-bit assumption in the newly added test and comments (expected 127, + actual 15). Direct inspection confirmed that both managed and native constants use three bits; the new expectations + and comments were corrected to that contract. Production packing arithmetic and pixel expectations were not changed. + All 181 focused coefficient-entropy, tiling, and frame-buffer cases now pass + (`byte-neighbor-contexts-focused-corrected.trx`, 5.0877 seconds). + Final Release .NET 11 build: zero errors, 1,009 existing warnings, 31.86 seconds. + All 9,991 AV1 and selected HEIF cases pass (`byte-neighbor-contexts-production.trx`, 2.8460 minutes). + Fresh optimized-native comparisons remain exact across the established 8,521,435-sample corpus: + maximum error zero, zero differing samples, and zero errors above one. + +- Above-context reset correction on 2026-09-07: `Av1ParseAboveNeighbor4x4Context.Clear` previously reset + only the visible tile width and used that same width for every coefficient plane. Native + `av1/common/av1_common_int.h:1594-1628` rounds the width to the current superblock extent and scales + chroma coefficient widths by horizontal subsampling. Managed above contexts are tile-relative + (`Av1TileReader.ParseTransformBlock:1282-1285`), so the equivalent reset starts at local offset zero. + The correction now clears that padded extent, preserving neutral contexts for nominal transform reads + beyond a clipped tile edge. A poisoned-context regression failed at the first padded transform entry + before the correction (expected 64, actual 128; `above-context-reset-before.trx`). + Eight cases cover both superblock sizes and monochrome/4:2:0/4:2:2/4:4:4; they also verify untouched + regions beyond each plane's required extent. All 66 focused tiling/compound cases pass + (`above-context-reset-focused.trx`, 5.6818 seconds). Final incremental Release .NET 11 build: + zero errors and zero reported warnings, 24.79 seconds. All 9,991 AV1 and selected HEIF cases pass + (`above-context-reset-production.trx`, 2.8410 minutes). Fresh optimized-native comparisons remain exact + across the established 8,521,435-sample corpus: maximum error and differing-sample count both zero. + This establishes the reset-state defect; a newly failing independently encoded tile fixture has not + been established. Decoder-wide conformance and performance claims remain unsupported. + +- Reconstruction workspace lifetime follow-up: native `av1/decoder/decoder.h:45-134` retains block scratch + on decoder worker state. `av1/decoder/decodeframe.c:3483-3518,5339-5357` allocates prediction scratch when + its physical sample capacity changes, and `av1/decoder/decoder.c:237` releases it at decoder destruction. + Managed `Av1BlockDecoder` previously rented its combined inverse/prediction/CfL owner in each frame's constructor + and returned it through `Av1FrameDecoder.Dispose`. The owner now belongs to `Av1Decoder.ReadTile:777-897` + and `Dispose:1061-1076`; frame/block reconstruction borrow `Memory` without an ownership flag or wrapper. + `Av1BlockDecoder.GetWorkspaceLength:138-150` preserves the existing layout and sequence-dependent capacity. + A larger requirement releases the old owner before renting the replacement; a failed rent leaves no stale owner. + Smaller subsequent frames reuse the larger capacity. Native and managed scratch layouts remain different: + native has separate motion, convolution, mask, and OBMC allocations, while managed prediction families share + one short-based working region. This lifetime correction does not establish identical allocation layout or speed. + Component callers now supply and dispose their own workspace; existing exact sizing and one-rent assertions remain. + Production checks cover initial/growth allocation rejection, retry, 64/128 superblock transitions, return to the smaller + size, exact constant lossless output, and balanced exactly-once returns. The independent restoration sequence also + requires the same reconstruction owner across repeated frames and 8/10/12-bit sequence changes. + Seven focused cases pass (`reconstruction-workspace-recovery.trx`, 5.5893 seconds). + Latest Release .NET 11 build: zero errors, 1,009 existing warnings, 8.44 seconds; Roslynk reports zero compiler errors. + All 9,983 AV1 and selected HEIF cases pass (`reconstruction-workspace-production.trx`, 2.8023 minutes). + Fresh optimized-native checks after that run remain exact across 8,521,435 samples: maximum error zero, + zero differing samples, and zero errors above one. The comparison corpus and retained-plane versus direct-plane + distinction are the same as the CfL follow-up below; no timing or separate-encoder parity claim follows. + Coefficient and transform-descriptor owners still belong to frame syntax storage, + and above/left parser contexts remain frame-local; those worker/common-state lifetime deviations remain open. + +- Inter/IBC CfL storage follow-up: `Av1BlockDecoder.EndBlock:1465-1503` now stores luma once after every + residual in the block has been reconstructed. Ordinary intra storage remains per transform. + `Av1TileReader.ParseBlock` invokes this completion phase after `Residual`; the pre-parsed component entry + point uses the same phase. The final parsed luma descriptor supplies the transform alignment for the visible + block extent. Evidence: native `av1/decoder/decodeframe.c:844-850,1037,1064-1124` and + `av1/common/cfl.c:406-436`. `Av1ChromaFromLumaContext.Store:94-194` shares its existing subsampling kernel + between transform dimensions and explicit completed-block dimensions; no additional sample storage was added. + Twelve new cases poison the CfL surface, prove it is untouched before completion, then independently calculate + Q3 samples and check untouched regions across 8/10/12 bits, 4:2:0/4:2:2, and visible/extended transform heights. + The first regression failed before the production correction (expected poison -32768, actual Q3 sample 20). + A test initially attempted to assign a private-set property; the inaccessible assignment was removed before + execution. Final Release .NET 11 build: zero errors, 1,009 existing warnings, 8.64 seconds. + All 58 focused tiling/compound cases pass (`cfl-block-store-focused.trx`, 5.5436 seconds). + All 9,981 AV1 and selected HEIF cases pass (`cfl-block-store-production.trx`, 2.8205 minutes). + Fresh optimized native comparisons cover 8,521,435 samples: 5,386,240 restoration samples, 3,113,847 film-grain + samples, and 21,348 samples from twelve regenerated moving-color streams. Maximum error, differing-sample count, + and count above one are all zero. Restoration/film-grain tests compare the edited decoder to the same retained + planes independently checked with aomdec; moving-color comparisons directly compare regenerated managed/native + planes. These checks do not establish encoder parity or a performance improvement. + +- Transform interleaving follow-up: `Av1TileReader.ParseBlock:959-1019` now publishes complete modes and + transform geometry before residual parsing, preserving its assigned map index in the borrowed partition state. + `Residual:1056-1208` calls reconstruction immediately after each transform's EOB is written. + `Av1BlockDecoder.BeginBlock:252-1275` predicts inter/IBC planes before residuals; + `DecodeTransform:1283-1457` performs ordinary intra prediction, inverse reconstruction, populated-prefix clearing, + and the existing luma-context update. Native ordering is in `av1/decoder/decodeframe.c:283-310,935-1037,1172-1228`. + The existing pre-parsed block API uses these same methods for component tests; production no longer walks + a second block/transform list. Source/frame ownership and coefficient capacities remain unchanged. + The test stub poisons unread EOB fields at block entry and checks during each transform callback that its + EOB is populated while all later transforms remain poisoned. It also requires every callback before the next block. + This verifies traversal timing independently of the final decoded output. + An initial build stopped on three blank lines before closing braces left by the method split; these were corrected. + Final Release .NET 11 build: zero errors, 1,009 existing warnings, 32.49 seconds. Forty-six tiling/compound checks + passed (`transform-interleaving-focused.trx`, 5.6221 seconds), followed by 9,969/9,969 AV1 and selected HEIF tests + (`transform-interleaving-production.trx`, 2.7762 minutes). Fresh optimized native decoding agrees with retained + restoration/film-grain planes across 8,500,087 samples; tests compare the edited managed decoder to those same planes. + Twelve regenerated moving-color streams add 21,348 directly compared managed/native samples. Combined maximum + error is zero, with zero differing samples and zero errors above one (`restoration-comparison.json`, + `film-grain-comparison.json`, `decoder-comparison.json` in the temporary takeover directory). + At this checkpoint, inter-block CfL storage remained per luma transform, + while native `decodeframe.c:844-850,1037` stores once after the block's residuals. Native `cfl.c:421-436` rounds + the visible block extent to the last selected transform dimensions before storing. The follow-up above addresses + that storage phase. No performance or encoder + parity claim follows from these reconstruction checks. + +- Subsequent parsed-block traversal correction: native `av1/decoder/decodeframe.c:1172-1228,2746-2801` + reconstructs while walking parsed partitions. The managed production reader previously parsed every block + in a superblock, then `Av1FrameDecoder.DecodePartition` walked their records again. `Av1TileReader` now + begins reconstruction at superblock entry and invokes `IAv1FrameDecoder.DecodeBlock` immediately after + publishing each parsed record. `Av1FrameInfo.UpdateModeInfo` returns that existing stored record by reference; + reconstruction clears its borrowed palette-map views after consumption. No new buffer or copy was introduced. + The first build exposed the unchanged `IAv1FrameDecoder` contract; the interface and its existing test stub + were then updated together. A new total-count assertion initially mistook `FrameInfo.ModeInfoCount` capacity + for parsed records. It now sums the tile's actual per-superblock record counts without changing pixel expectations. + Assertions inside the callbacks require zero records at superblock entry and exactly the currently visited + number of published records at each block callback, proving that later blocks have not been parsed first. + All 17 tiling checks passed (`block-traversal-focused-fixed.trx`, 2.9108 seconds), followed by 9,969/9,969 AV1 + and selected HEIF tests (`block-traversal-production.trx`, 2.7630 minutes). Final Release .NET 11 build: + zero errors, 1,009 existing warnings, 8.90 seconds. Individual-transform parse/reconstruct interleaving remains + open at this checkpoint; `decodeframe.c:935-958,283-310` consumes each transform before reading the next. + These are reconstruction/traversal checks, not encoder parity or performance evidence. + +- Official main was rechecked through a live request to the official Gitiles endpoint and remains + `d565eec60f084421fa34fc0534b760c6452b6a6c`. The cached web response was older and was not used as revision evidence. + A fresh direct Gitiles request on 2026-09-07 confirms the same revision. The first request was blocked by + sandbox socket permissions; the allowed direct retry succeeded. The web tool again returned an older July + response, which was excluded from current-reference evidence. +- The preceding coefficient-stage checkpoint left a clearing-lifetime deviation open. Native + `av1/decoder/decodetxb.c:135-150,279-284` records the largest populated raster index independently of EOB; + `av1/decoder/decodeframe.c:154-164` clears that prefix after inverse reconstruction. EOB alone cannot bound + the prefix because the scan can visit a larger raster index earlier than its final nonzero symbol. +- The managed sign/dequantization loop now records that index in `Av1TransformInfo.MaximumCoefficientIndex`. + `Av1BlockDecoder.DecodeBlock` clears through it after reconstruction. `Av1TileReader.ReadTile` retains whole-plane + clearing only for syntax-only parsing, whose contract exposes coefficients without invoking reconstruction. + Initial allocator storage is already clean (`Av1FrameInfo.cs:289-292`); no additional coefficient buffer or owner + was introduced. Skipped and all-zero transforms reset their residual metadata at the owning parse boundaries. +- Focused entropy tests check the populated raster bound, including sparse input, and reuse a nonempty descriptor + for an all-zero transform. Existing tile reconstruction cases now assert that all three coefficient scratch planes + are zero afterward in both byte and high-bit-depth paths. These assertions supplement exact reference comparisons; + they do not establish full decoder correctness or a measured performance improvement. +- Release .NET 11 build passed with zero errors and the existing 1,009 warnings. Serialized Visual Studio VSTest + passed 125/125 focused entropy, tiling, and frame-buffer cases (`coefficient-clearing-focused.trx`), followed by + 9,375/9,375 AV1 and public HEIF encoder cases in 2.6811 minutes (`coefficient-clearing-production.trx`). + Fresh optimized-native checks matched all 8,500,087 retained restoration/film-grain samples and 21,348 regenerated + color-sequence samples: maximum error zero and zero differing samples. The retained fixtures also passed the + managed exact-plane assertions; this is bounded same-bitstream evidence, not separate-encoder parity. +- A subsequent ownership correction removes the second dequantization context constructed by `Av1TileReader`. + `Av1InverseQuantizer` had already constructed identical values, then discarded its own context when the reader + supplied the duplicate to `UpdateDequant`. The quantizer now retains and updates its original context. The frame + setup and conditional delta-Q updates still follow `decodeframe.c:1865-1906,1200-1217`. After this final production + edit, the Release .NET 11 rebuild passed with zero errors and 1,009 existing warnings; serialized Visual Studio + VSTest passed 222/222 focused cases including the independent reconstruction corpus in 1.8621 minutes + (`coefficient-lifetime-final.trx`). Roslynk reports zero compiler errors and the existing 34 compiler warnings. + No benchmark has run. +- Whole-superblock parsing before reconstruction and worker lifetimes remain unresolved. This change addresses + clearing ownership only; it does not claim completion of the encoder or decoder architecture. + Film-grain decoder source comparison after `ef8b1a823`: - The complete template generation, random state, autoregression, scaling interpolation, overlap traversal, @@ -818,8 +1897,10 @@ Motion-controller investigation continued after correction checkpoint `578ec34d9 - [ ] Reconcile frame configuration and encoder decision policy with the reference before isolated pruning changes. - [ ] Implement missing tools and complete reference, partition, motion, transform, coefficient, and winner decisions. - [ ] Compare separately encoded results from identical source samples with explicitly reconciled settings. - Report maximum absolute error and counts exceeding one for every decoded output component and every frame. - The acceptance limit is one component unit per sample; PSNR and average error cannot replace it. + Report maximum absolute error and counts exceeding one for every component and every frame. + Encoder parity permits at most one component unit per sample; PSNR and average error cannot replace it. +- [ ] Verify byte-exact decoder output against the reference for identical bitstreams and reconciled output conversion. + Report maximum error and counts of differing samples and bytes. Every count and maximum error must be zero. - [ ] Verify each final relevant edit with focused serialized Release .NET 11 Visual Studio VSTest and independent native production-output checks. Compilation and component tests do not close codec completeness. - [ ] Run equivalent end-to-end benchmarks only after the relevant source comparison justifies the next change. @@ -1417,13 +2498,13 @@ Previously verified algorithm checkpoints remain valuable evidence, but the fina - [x] Lossless inverse transform, loop filtering, CDEF, super-resolution, restoration, and film grain have focused checkpoint evidence. - [x] Retained references, CDF snapshots, segmentation maps, global motion, temporal motion fields, and dependent-frame lifecycle have been re-audited and verified against current libaom `main`. - [x] The 12-case all-intra profile matrix covers every valid 8, 10, and 12-bit monochrome, 4:2:0, 4:2:2, and 4:4:4 combination. Dependent-frame coverage is recorded separately above. -- [x] The exact current-tree native-plane matrix passes through the production decoder on net10.0 and net11.0. The normal-dispatch and FeatureTestRunner fallback methods pass 2 of 2 focused tests on each target. -- [x] The exact current-tree presentation matrix passes 12 of 12 cases through ImageSharp's established reference-image API on net10.0 and net11.0. +- [x] At this historical checkpoint, the native-plane matrix passed through the production decoder on net10.0 and net11.0. The normal-dispatch and FeatureTestRunner fallback methods passed 2 of 2 focused tests on each target. These results do not verify subsequently edited trees. +- [x] At this historical checkpoint, the presentation matrix passed 12 of 12 cases through ImageSharp's established reference-image API on net10.0 and net11.0. Current verification is recorded in the dated takeover entries above. - [x] Verify malformed/truncated data, frame IDs, reference slots, tile bounds, allocation limits, cancellation, and failure unwinding. - [x] Verify still items and bounded sequences from file, memory, non-seekable, and short-read streams. - [~] Verify ICC, CICP, alpha, grids, pixel aspect ratio, clean aperture, rotation, mirroring, metadata, and every presented sequence frame. Grid validation now requires the first cell to be at least 64 samples on both axes, enforces even output and cell dimensions along each subsampled AV1 chroma axis, and requires every cell to cover its row-major output region without exceeding the first cell's dimensions. This accepts the smaller right and bottom cells supported by the writer while also accepting uniform coded cells whose final row and column are cropped to the grid descriptor. Auxiliary alpha uses the same cropped overlap, so padded cells cannot write outside the final frame. Production regressions cover smaller right, bottom, and bottom-right color and alpha cells; Roslynk compiler and scoped analyzer diagnostics are clean, while runtime verification of the current tree remains pending. - [x] Complete the public AVIF format/API review so registered capabilities match implemented behavior. -- [x] Remove or reject every valid in-scope AV1 syntax branch that remains silently ignored or unsupported. +- [ ] Implement every valid in-scope AV1 syntax branch. Rejecting an unsupported valid branch does not complete it; tile-list decoding and its external-reference integration remain unresolved. Verified negative-path and frame-identifier gate evidence on 2026-08-31: @@ -1569,10 +2650,11 @@ Final decoder allocation, lifetime, precision, architecture, and test-validity a clipping, while high-bit-depth filtering writes directly to the native `ushort` destination. - [x] Reference-to-presentation copying now copies visible native rows only. Padding remains destination owned, and the ownership tests mutate a copied visible sample rather than unrelated padding. -- [x] The remaining decoder allocations and copies are either bounded scratch or required ownership - boundaries. Frame planes enforce their contiguous single-span invariant before allocation; palette, - transform, film-grain, super-resolution, color-conversion, and alpha workspaces remain bounded and - allocator owned. No per-block managed allocation remains in reconstruction. +- [~] The historical audit did not establish that every remaining decoder allocation or copy is required. + Frame planes use contiguous storage and the listed workspaces are allocator owned, but the takeover audit + has since found and corrected excess frame-local workspace lifetimes and wider-than-required neighbor + contexts. Remaining parser-context, coefficient/TU storage, dispatch, and output ownership paths still + require source comparison; see the current audit evidence above. - [x] Block reconstruction now uses one exact-size signed-short owner for inverse quantization, inverse transform, compound prediction, convolution, and chroma-from-luma scratch. Even-length slices provide the integer workspaces without another rent. Monochrome reserves no chroma coefficients, and 4:2:0, @@ -1580,9 +2662,10 @@ Final decoder allocation, lifetime, precision, architecture, and test-validity a constructor rents and their catch-all rollback path; exact allocation length, coefficient span length, and exactly-once return pass for all four layouts, with 549 adjacent reconstruction tests passing direct net11 VSTest in Release. -- [x] Valid unsupported tile-list syntax is rejected explicitly. Reserved and metadata OBUs are consumed - only after bounded framing and trailing-bit validation. Eight-, ten-, and twelve-bit reconstruction, - presentation, alpha, restoration, and film-grain paths retain native precision. +- [~] Tile-list decoding is not implemented. The explicit rejection test only establishes that restriction; + the camera-header/external-reference path and its container integration remain unresolved. Reserved and + metadata OBUs have bounded framing and trailing-bit checks. Existing precision tests cover represented + reconstruction and presentation paths, not complete AV1 feature support. - [x] Predictor traversal remains split into semantic readonly operator families. The planar sample adapter and transform-block context are value types, and Release construction sites use `default` without null-forgiving suppression. @@ -1670,16 +2753,16 @@ Encoder verification contract: - [~] SIMD-first RGB-to-native-plane conversion now feeds eight-bit and high-bit-depth bordered AV1 source frames directly, preserving ImageSharp's arbitrary packed-pixel input contract without an intermediate full-frame native-plane copy. - [~] Auxiliary-alpha encoding now follows the same packed-pixel conversion boundary without scanning pixel contents or cloning the image. Source alpha is converted through ImageSharp's 16-bit pixel contract, deinterleaved with descending Vector512, Vector256, Vector128, and scalar traversal through the shared vector-count helpers, then scaled and rounded once by the existing native-sample writer directly into the final bordered monochrome source frame. One operation-wide allocator owner provides the packed and planar row views; there is no frame-sized alpha staging allocation or second owner. Exact 12-bit precision, physical border extension, the single 12-bytes-per-pixel row rent, and balanced return pass through the production converter. The complete 47-case frame-encoder set passes direct foreground net11 Release VSTest, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts the generated 8-, 10-, and 12-bit monochrome payloads. AVIF auxiliary item properties, references, and public activation remain open. - [~] Forward transform families, transform workspace, and an allocation-free DC intra block boundary exist locally. For eight-bit and high-bit-depth samples, the composed boundary now follows current libaom's encoder order: predict into the reconstruction plane, subtract prediction from source, transform, quantize into separate qcoeff and dqcoeff storage, retain EOB and transform type, and inverse-transform only when EOB is nonzero so later blocks consume decoder-identical references. Prediction and subtraction retain their SIMD-first operators, independent source and reconstruction strides are preserved, and no frame-sized or per-block buffer is introduced. The block boundary consumes the real bordered encoder-plane regions and indexes their one-segment owner directly; this preserves physical row strides without a row copy and avoids the per-call enumerator allocation exposed by the initial array-only test. One reusable 61 KiB allocator owner supplies tightly packed residual, aligned transform-coefficient, dequantized-coefficient, and transform scratch spans across transform blocks; quantized coefficients write directly to the retained frame coefficient owner instead of being duplicated. A fixed 8x8 DC-intra superblock baseline now traverses the same recursive preorder and frame-edge pruning as the tile writer, gathers left references into that reusable block workspace, writes luma and chroma coefficient-owner slices in the writer's exact consumption order, and updates the caller-owned reconstruction planes for subsequent predictions. Stage-by-stage scalar-oracle, physical-border, retained-syntax, superblock-to-writer synchronization, high-bit-depth precision, and steady-state zero-allocation coverage passes 8 of 8 through direct net11 VSTest in Release. This is a legal fixed baseline, not complete partition or mode analysis. -- [~] The production tile writer walks raster superblocks, analyzes each immediately before entropy coding, reuses one decision workspace and one block workspace, and retains decoder-identical reconstructed references across each tile. Frames exceeding AV1's 4,096-sample tile-width or 4,096-by-2,304-sample tile-area limit now select the minimum uniform tile-column and tile-row logarithms used by current libaom. Every tile begins from the same normative frame probabilities, appends its independently finalized range-coded bytes to one bounded output allocation, and records only its offset and length in the picture-state owner. Closed byte and high-bit-depth operators feed the existing superblock boundary without runtime sample-type checks. Existing byte-exact and clipped-superblock tests cover traversal and coefficient indexing; a production 4,097-sample-wide lossless case crosses the first tile boundary and checks decoded pixels on both sides. Roslynk reports zero compiler errors; runtime and current-libaom verification of this multi-tile checkpoint remain pending. +- [~] Production coding now analyzes all tiles before packing. Analysis retains modes, coefficients, palette tokens, motion contexts, and reconstruction; selected deblocking runs before entropy edges are reset for packing. The current frame controller still does not select deblocking levels or select/apply CDEF and restoration. The dated takeover checks above establish only their stated reconstruction and syntax behavior. - [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. One operation-ready frame owner now rents the aligned Y, U, and V storage contiguously, exposes non-owning `Buffer2D` plane views, and returns the rent exactly once. A 4K 4:2:0 frame occupies about 13.0 MiB at 8-bit or 26.0 MiB at 10/12-bit; source and reconstruction therefore remain distinct frame owners rather than adding a full-frame copy. The corrected tests use this real ownership path and verify the exact 54 KiB 64x64 4:2:0 rent. The frame-encoder operation now instantiates matching source and reconstruction owners with ordinary `using` lifetimes and converts packed pixels directly into the source owner before extension. - [~] Temporal delimiter, sequence header, frame header, combined-frame tile-group writing, uniform multi-tile layout, and reduced and non-reduced frame operations now exist locally. The remaining codec-tool and verification work is tracked below. - [~] Implement superblock and partition analysis for every permitted block size and partition. Efforts zero through eight deliberately split every in-frame node to 8x8 blocks. Effort nine performs recursive live rate-distortion selection at complete 8x8 and 16x16 nodes, while effort ten extends the same search to complete 32x32, 64x64, and 128x128 nodes. Candidate order matches current libaom: `PARTITION_NONE`, `PARTITION_SPLIT`, `PARTITION_HORZ`, `PARTITION_VERT`, the four asymmetric partitions, then `PARTITION_HORZ_4` and `PARTITION_VERT_4`; the two 1-to-4 partitions are excluded at 128x128 as required by current libaom. Invalid chroma geometries are excluded before evaluation. Each candidate saves and restores the exact partition, coefficient, transform, and palette neighbor edges in one aligned block-workspace owner; trials neither allocate nor copy probability state. Recursive split trials publish each selected child's decoded mode, transform, coefficient, and palette contexts before evaluating its next sibling. Coefficient contexts are published per retained transform rather than broadcasting the first transform over an entire partition leaf. Large luma and chroma leaves are evaluated as bounded-64, raster-ordered transform tiles in the existing aligned workspace, and each winning plane is copied to retained storage once. Production picture state retains the compact 8x8 mode allocation below effort nine and explicitly selects 4x4 allocation granularity when sub-8x8 partitions are enabled. Effort-dependent pruning remains. - [~] Implement intra mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. Live luma search now covers all 13 zero-angle base modes and all six nonzero adjustments for each of the eight directional modes. Joint spatial chroma search covers the same 61 candidates, combines both chroma planes in one rate-distortion decision, and preserves the winning shared angle adjustment. Chroma-from-luma now searches the complete signed alpha alphabet from reconstructed luma and retains its joint U/V syntax. Filter-intra now searches all five predictors after ordinary luma modes. Palette entropy, retained state, production syntax, luma and paired chroma palette selection, screen-content activation, and joint intra-block-copy mode selection exist, but their full reference decision policy and separate-encoder parity remain unverified. - [~] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. The sequence encoder retains the preceding reconstruction and, from effort six, searches a bounded full-pixel frame translation against that LAST_FRAME reference. Candidate discovery uses the existing SIMD-first squared-error kernels over a central analysis window, validates the winner over the complete coded luma plane, and charges its exact uncompressed-header bit count in the inter-frame rate-distortion domain. Pure translation is signaled as an identity-scale rotation/zoom model, matching current libaom's workaround for the AV1 translation-only axis defect. Each 8x8 inter-frame block first retains the complete intra candidate, then compares NEARESTMV, all three legal NEARMV dynamic-list entries, GLOBALMV, and all three legal NEWMV dynamic-list entries against it with live intra/inter, single-reference, mode, DRL, differential-vector, skip, transform, coefficient, and distortion costs. The initial NEWMV search retains the reference's cheap prediction-error stage, but every surviving mode now owns a complete transform, coefficient, skip, and distortion evaluation before mode selection; selected and candidate workspace views exchange ownership only on strict improvement. Inter trials remain in the existing shared workspace until they strictly beat the intra result, so losing trials require no backup buffer or copy. The tile writer emits the matching DRL path and normative context-selected `LAST_FRAME` reference tree instead of forcing every block through a segmentation feature. The selected DRL index reuses the filter-intra byte because those block syntax branches are mutually exclusive, preserving the existing packed state size. Packed short vectors reuse the existing picture-state owner. Effort six keeps a full-pixel fast path; effort seven refines each selected NEWMV through half- and quarter-pixel eight-tap prediction; effort eight adds the final eighth-pixel stage. The frame header advertises the matching precision, and the shared motion-vector entropy path emits fractional and high-precision symbols only when that precision permits them. Roslynk reports no compiler or scoped analyzer diagnostics, but runtime verification is pending. Additional retained reference pictures, compound prediction, and the remaining inter tools remain. - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. High-bit-depth paths widen before multiplying instead of applying the eight-bit coefficient clamp. Lossless blocks use the AV1 4x4 Walsh-Hadamard transform, exact lossless quantization and dequantization, four-by-four-only transform syntax, and non-skipped residual coding. Transform search and coefficient optimization remain. -- [~] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. The complete luma and joint chroma candidate sets, including chroma-from-luma, filter-intra, palette, and intra-block copy, now perform live rate-distortion selection. Public quality mapping and effort tiers through exhaustive uniform luma mode/transform search are implemented. Effort nine adds exact recursive 8x8 and 16x16 partition rate-distortion selection, and effort ten extends it through 128x128; effort-dependent pruning and the remaining sequence searches remain. -- [~] Frame effort now progressively expands the available current search: zero is DC-only, one adds every zero-angle spatial mode, two adds every legal directional adjustment, three refines the preliminary luma winner's transform type, four adds filter-intra and chroma-from-luma, and five adds adaptive palette and intra-block-copy analysis. Lower tiers do not signal unavailable sequence or frame tools, and tiers below five skip the whole-frame screen-content scan. Effort six enables `TX_MODE_SELECT` and compares the winning ordinary spatial or filter-intra luma mode as one 8x8 transform against four raster-ordered 4x4 transforms; each luma palette candidate owns that size comparison from effort six onward. Effort seven searches every legal 8x8 transform type inside every ordinary spatial candidate rather than refining only the preliminary winner. Effort eight also performs the 8x8-versus-four-4x4 comparison inside every ordinary spatial and filter-intra candidate, matching current libaom's per-candidate uniform-transform ownership. Effort nine additionally searches every legal partition at complete 8x8 and 16x16 nodes in current-libaom order, and effort ten extends that recursive search through 128x128. Prediction and residual construction run once per mode and are reused across its legal transform types. A 128x128 leaf evaluates four 64x64 luma transforms and as many as sixteen 32x32 transforms per 4:4:4 chroma plane, retaining sparse state at coefficient-area offsets. Residual emission follows AV1's bounded-region order, completing Y, U, and V for each 64x64 luma region before advancing. Every 4x4 transform searches all legal types with live coefficient contexts and reconstructed intra references. The search reuses the aligned block workspace, preserves only global improvements, and performs no per-block, per-partition, or per-transform rent. Non-skipped intra-block copy writes and costs the current-libaom unsplit variable-transform root; skipped intra-block copy emits no transform-partition symbol. Effort-dependent model and transform pruning remain. Decoder-visible production cases inspect the emitted restrictions and frame state and decode the produced streams, including real effort-nine streams selecting sub-8x8 and 8x16 rectangular blocks. The complete non-HEVC HEIF/AV1 namespace passes 9,077 of 9,077 through one foreground net11 Release VSTest run. The last independently built `aomdec`, from the then-current `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` snapshot, accepts the previously generated effort-eight and effort-ten payloads as well as the existing palette and intra-block-copy payloads. The affected encoder, partition, and workspace surface passes 139 of 139 through one foreground net11 Release VSTest run. The net11 Release build and Roslynk compiler and analyzer passes report zero errors. -- [~] Encoder rate accounting converts the entropy writer's live inverse cumulative distributions into current-libaom fixed-point symbol costs without allocating or duplicating probability state. Read-only luma-mode, directional-delta, filter-intra, chroma-mode, block-skip, transform-size, transform-block-skip, and complete transform-coefficient queries share the exact distributions mutated by the subsequent entropy write. Complete coefficient costing follows current libaom's optimized shape: it returns immediately for an empty transform, uses the EOB-specific base-range context, fuses magnitude, sign, base-range, and Golomb accounting into one reverse traversal, and combines repeated full base-range chunks instead of replaying each emitted symbol. Tile-lifetime level and context scratch is reused, the one-coefficient path neither clears nor initializes the forward-neighbor level map, and steady-state queries allocate nothing. Transform-size writing and costing share one subdivision-depth calculation, while shared closed symbol operations keep the writer and cost mappings for transform skip, transform type, and EOB syntax identical without forcing the estimator through the writer's slower two-pass coefficient traversal. The current-libaom fixed-point RD combiner preserves 64-bit distortion and rounds the weighted 1/512-bit rate at the required boundary. Its key-frame multiplier follows libaom's squared DC-quantizer formula and exact 10/12-bit normalization. Live final-block selection evaluates all 61 legal 8x8 luma candidates: the 13 zero-angle base modes in current-libaom order, followed by six nonzero adjustments for each directional mode. Joint chroma selection evaluates the equivalent 61 spatial candidates, combines U and V distortion plus coefficient rate, and charges one live chroma-mode and shared-angle symbol over the actual subsampled 4x4, 4x8, or 8x8 geometry. Chroma-from-luma subsamples the reconstructed luma block once into fixed-stride Q3 stack scratch, subtracts the rounded mean, evaluates all 33 signed alpha values independently for each plane with complete transform RD, and combines the cached plane results across all 1,088 valid joint pairs with one live sign cost and the conditional U/V magnitude costs. This is the allocation-free equivalent of current libaom's exhaustive 33-value path: it requires 66 evaluation transforms rather than transforming every joint pair, preserves DC-before-CfL-before-spatial tie order, and fixes the implicit chroma transform to DCT-DCT. Filter-intra follows ordinary luma candidates, searches all five predictors in syntax order, and evaluates every legal transform while reusing one prepared prediction and source residual per filter mode. Every candidate includes its live mode, angle, filter mode, alpha, and coefficient rate plus normalized pixel-domain distortion. The corrected prepared reference edges retain the common-corner prefix and width-plus-height extent required by rectangular directional prediction. A shared encoder/decoder availability calculation selects reconstructed top-right and bottom-left extensions according to tile, frame, superblock, and block reconstruction order; unavailable extensions repeat the nearest coded endpoint. Missing top or left edges retain current libaom's perpendicular-sample and bit-depth-midpoint rules. Directional prediction applies the AV1 three-degree adjustment step and reuses transform workspace for zone-three transposition before the transform overwrites it, keeping candidate evaluation allocation-free. The winning luma and chroma signed adjustments are retained in the packed final-block state consumed by the tile writer. The tile writer invokes these reusable workspace-backed selectors after mapping current neighbors and immediately before writing each block, so later decisions see reconstructed samples, coefficient contexts, and CDF updates from every preceding block. Block skip is read only after the callback has combined every coded plane. Luma and chroma candidate scratch is partitioned from the encoder's single aligned reusable block workspace; transform-size search uses that owner for four retained 4x4 transform states, local coefficient contexts, and the compact trial reconstruction needed to preserve the best result. No candidate path rents a buffer per block or per transform. Only a newly winning candidate is copied into retained frame storage. Production fixtures force every luma base predictor, both extreme adjustments in all three directional zones, available top-right and bottom-left extensions, high-bit-depth adjustment propagation, exact signed luma and chroma angle-rate terms, joint U/V decisions, packed chroma state, and 4:2:0, 4:2:2, and 4:4:4 transform geometry. The CfL fixtures derive target chroma from a pilot production encode's actual reconstructed luma through an independent scalar Q3 oracle and prove exact positive/negative alpha syntax plus zero-residual DCT-DCT reconstruction for all three subsampling geometries at 8, 10, and 12 bits. The stable fixed-DC traversal comparison uses neutral samples for which both the baseline and live search select DC with non-skip coefficient syntax, instead of relying on textured content to happen to select the baseline mode. Luma palette selection now evaluates dominant-color and one-dimensional K-means candidates for every legal size, snaps near-cache colors with the reference threshold and tie order, removes duplicate snapped colors, extends boundary maps from active samples, and performs complete transform rate-distortion search. Ordinary DC and filter-intra candidates pay the palette-disabled symbol whenever screen-content syntax is enabled. The exact net11 Release rebuild reports 1,992 test-project warnings and zero errors, all 58 intra-superblock cases pass, all 8,935 AVIF cases pass, and all 230 HEIF cases pass. Remaining mode decision work includes transform-size coverage for filter-intra and palette, broader joint mode/transform refinement, and effort-dependent pruning. Ordinary intra blocks now remain non-skipped even when all transforms are empty; inter and intra-block-copy mode selection own their distinct skip-transform RD decisions. +- [~] Complete reference-led rate-distortion selection remains open. The managed implementation evaluates spatial, filter-intra, palette, chroma-from-luma, and intra-block-copy candidates, but its effort thresholds, partition policy, intra-before-inter ordering, transform pruning, coefficient optimization, and winner refinement are not reconciled with the reference. Evaluating additional candidates and passing the existing tests do not establish encoder parity. +- [~] The current Effort 0–10 behavior is an unreconciled implementation policy. Its DC-only lower tier and progressive activation of directions, transforms, screen-content tools, and larger partitions were not derived from the complete native profile controller. Existing effort tests verify that current behavior only. GOOD/ALLINTRA cpu-used mapping and the dependent frame-size, quantizer, frame-role, and staged search rules remain unresolved; the native default is not exhaustive enumeration. +- [~] Encoder rate accounting uses live adaptive CDFs and shared symbol-cost operations. Native coding also uses staged coefficient/mode/motion cost snapshots whose refresh policy must be reproduced. The corrected fixed-point arithmetic and bounded syntax tests do not establish that current candidate costs, ordering, transform choices, or reconstruction match the complete native encoder. Deferred packing is implemented; staged pruning, coefficient optimization before inverse reconstruction, and final winner refinement remain open. - [~] The tile writer now publishes one packed coefficient context per covered 4x4 edge unit and derives luma/chroma skip plus DC-sign contexts from the complete transform edges using current-libaom units. Partition, transform, and coefficient neighbor state retains only the above and left context regions used by current libaom; the unused third top-left region, its granularity state, and its unused sentinel are removed. One picture owner packs segmentation, every tile's partition, luma, chroma, and transform edges, CDEF state, preceding quantizer, and encoded payload bounds into one clean byte allocation with typed non-owning views; together with the separately typed packed mode-information owner, the complete picture state uses two allocator rents rather than seven. Each encoded tile has independent neighbor and probability state while sharing the bounded output owner. Earlier aligned-length and balanced-return coverage exists; runtime allocation verification of the current multi-tile layout remains pending. - [~] Encoder mode information now uses a frame-owned integer alias grid over a packed 8-byte value allocation, matching current libaom's `mi_grid_base` and `mi_alloc` relationship without a managed object or reference per 4x4 entry. The visible dimensions are aligned to eight luma samples, the grid stride and allocated row count are aligned to 32 mode-information units, and optional 8x8 allocation granularity reduces the value store in both dimensions exactly as current libaom does. One clean ImageSharp byte owner contains both independently typed regions, reducing libaom's two allocation lifetimes to one without a copy. At 4K, the 4x4 layout occupies about 6.0 MiB in total; the 8x8 layout occupies about 3.0 MiB. Exact geometry, clean allocation, typed lengths, aligned mapping, untouched row padding, and exactly-once return pass 4 of 4 direct net11 VSTest cases in Release. Every coded 4x4 cell covered by square, rectangular, or clipped edge blocks maps to its owning allocation entry before context-dependent symbols are written. Packed syntax, relative neighbor lookup, full block mapping, writer traversal, entropy, and OBU coverage pass 1,947 of 1,947 direct net11 VSTest cases in Release; complete mode decision still remains. - [~] The superblock decision and palette-map workspace uses one reusable 40.3 KiB ImageSharp allocator owner. Its aligned 8.3 KiB decision region contains 1,024 explicitly packed 8-byte final-block entries and the 341 preorder partition bytes required by a complete 128x128-through-8x8 quadtree; its remaining 32 KiB contains the fixed 128x128 luma and chroma palette maps. This combines storage held separately by libaom; exact lifetime and size reconciliation remains part of the fresh allocation audit, and fewer owners alone does not establish an improvement. Palette colors have their own current-block value and are copied only to the picture edges that later blocks can reference, so enabling palette mode does not add 50 bytes to every final-block entry. Construction and the explicit per-superblock reset initialize every syntax field, including the nonzero sentinel that disables filter-intra prediction; pooled quantizer, prediction, partition, and current-palette bytes cannot leak into the next decision pass. Roslynk reports zero compiler errors for the current one-owner refactor; runtime allocation verification remains pending. @@ -1699,7 +2782,7 @@ Encoder verification contract: - [~] Luma mode selection now evaluates each of its 61 mode-and-angle candidates with the mode-derived default transform used by current libaom's fast intra path. It then refines only the winning mode across all seven transform types permitted by the 8x8 intra set in transform-enum order. This removes the fixed DCT-DCT limitation while avoiding a 61-by-7 expansion; each trial includes live transform-type and coefficient rate, reconstructed pixel-domain distortion, and the existing aligned reusable block workspace. Eighteen exact-prediction production cases prove DCT-DCT wins equal-cost ties in reference order even when the first pass used a different default, while the 72x72 textured traversal proves a non-DCT transform with nonzero coefficients reaches retained syntax. Current-main `aomdec` accepts all 29 regenerated payloads. Special-mode transform-size coverage, full partition search, and broader effort-dependent joint mode/transform search remain. - [x] Chroma-from-luma mode decision now reuses the decoder's SIMD-first 4:2:0, 4:2:2, and 4:4:4 reconstructed-luma preparation and prediction kernels for both byte and high-bit-depth encoder operators. The constant DC predictor for each chroma plane is computed once and its sample refills every alpha candidate, matching libaom's per-plane DC cache instead of rebuilding the same edge average 33 times. Each block uses 512 bytes of fixed stack scratch for the maximum 8-row predictor surface plus 792 bytes for complete U/V rate and distortion tables; no allocator owner, managed object, frame copy, or persistent buffer was added. Live probability costs exactly mirror current libaom's joint-sign ownership and conditional magnitude symbols. Nine production cases independently derive exact CfL targets from decoder-visible reconstructed luma at 8, 10, and 12 bits, and three entropy cases cover two nonzero signs plus each single-zero-plane form. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, all 8,959 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. - [x] Filter-intra mode decision now runs after ordinary luma modes in current-libaom order, evaluates all five recursive predictors, and refines each predictor across every legal 8x8 transform in transform-enum order. Strictly-better replacement preserves ordinary-mode and filter-mode tie order. Each filter prediction and its source residual are prepared once and reused across transform candidates, avoiding repeated recursive prediction while retaining SIMD-first predictor and subtraction operators. The stack cost is 192 bytes for eight-bit samples or 256 bytes for high-bit-depth samples; no allocator owner or managed buffer was added. Fifteen production cases force every filter mode at 8, 10, and 12 bits and prove retained filter syntax, zero-residual reconstruction, and the DCT-DCT equal-cost transform tie. The decoded-frame MD5 values selected by this checkpoint are `d7d68803763b95827483f14515281d3a` for the 8x8 10-bit gradient, `3f7e34d44c65d7797ad26b5cd4c35bf4` for the 8x8 12-bit gradient, and `9985f05790d2c9f5f28723ef86d5b89b`, `2ba2f1d0fcfef60394a5175553c7cb8b`, and `6aa7a2ed0dbf76ad2ec0c222585272d0` for the odd 4:2:0, 4:2:2, and 4:4:4 gradients. The exact net11 Release rebuild remains at 1,005 warnings and zero errors, 18 focused filter-intra, predictor-reference, syntax-cost, and allocation cases pass, all 8,974 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. -- [x] Empty-transform block skip now compares the complete live rate of the two decoder-identical syntax choices after luma and every coded chroma plane have been selected. Current libaom forces all-intra blocks to non-skip; this encoder retains that behavior for every non-empty block and for equal-cost empty blocks, but emits block skip when its adapted context cost is strictly lower than non-skip plus all empty-transform coefficient costs. Costing and writing share the same above-and-left skip-context calculation, and the coefficient estimator returns after the transform-block-skip symbol without reading coefficient storage. This adds no allocation, copy, or persistent state. A focused adapted-CDF regression proves both outcomes through the production decision helper, the two production all-zero fixtures still prove the default real block path, the exact net11 Release rebuild remains at 1,005 warnings and zero errors, all 8,975 HEIF/AV1 tests pass, and current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 29 regenerated payloads. +- [~] The historical ordinary-intra empty-transform skip experiment was superseded by the source-led correction above. Production ordinary-intra blocks now remain non-skipped, including all-zero transforms. The obsolete helper and its direct test remain after automatic review rejected their removal; they have no production caller. The earlier 8,975-case report and 29 decodable payloads established neither reference-controller parity nor current-tree correctness. - [~] Palette entropy coding now mirrors current libaom's adaptive luma-mode, chroma-mode, palette-size, and spatial color-index distributions, together with its truncated-binary uniform code used by palette colors. The complete mutable palette probability graph is created once on first palette search or write, so the current palette-disabled frame path retains zero palette allocations. Three focused regressions cover every legal 2-through-8 color alphabet and every defined mode, size, and color-index context; all 1,928 entropy cases and all 8,978 HEIF/AV1 cases pass direct net11 Release VSTest. The exact Release rebuild remains at 1,005 warnings and zero errors. This checkpoint adds the exact entropy foundation only: palette candidate generation, retained color and index storage, mode decision, map tokenization, and production syntax remain incomplete, and no generated payload changed. - [~] Luma and chroma palette-color coding now matches current libaom's neighbor-cache flags, sorted delta representation, wrapped V-plane deltas, strict delta-versus-raw V selection, and fixed-point color-rate model at 8, 10, and 12 bits. Encoder costing and emission use only fixed stack spans, including explicitly initialized cache-membership state, and steady-state color costing allocates zero managed bytes. The decoder consumes the same bounded color-syntax primitive after the tile reader derives its neighbor cache, removing duplicated color parsing without changing retained palette ownership. Nine focused syntax, exact palette decode, constrained-allocation, truncation, presentation, and allocation cases pass; all 1,933 entropy cases and all 8,983 HEIF/AV1 cases pass direct net11 Release VSTest. The exact Release rebuild remains at 1,005 warnings and zero errors. Retained encoder palette colors, neighbor caches, color-index maps, candidate generation, and production palette selection remain incomplete, and the compact 8-byte frame mode entries were not enlarged. - [~] Palette color-index map coding now shares the exact current-libaom neighbor weights, stable color ordering, five context classes, first-index uniform code, and diagonal wavefront between encoder costing, encoder writing, and decoder parsing. The decoder's stack-allocated context scores are explicitly cleared before accumulation, removing an invalid dependency on uninitialized stack contents. Costing and writing use a closed generic operation while the shared driver owns traversal and context derivation, so the semantic operations remain independent of map layout and tail handling. The path adds no retained state or per-call managed allocation. Its allocation regression now runs one complete unmeasured hot-path window before measuring an independent 1,000-call steady-state window, so tiered-runtime transitions cannot make the full parallel suite report a one-time allocation as a recurring operation cost. Twelve focused map, exact palette decode, padding, trailing-bit, and allocation cases pass; all 1,941 entropy cases and all 8,991 HEIF/AV1 cases pass direct net11 Release VSTest. The exact Release rebuild remains at 1,005 warnings and zero errors. Production payloads remain unchanged because palette selection is still disabled; retained colors, neighbor caches, index-map storage, candidate generation, and production palette mode decision remain incomplete. @@ -1709,7 +2792,7 @@ Encoder verification contract: - [~] Paired chroma palette clustering now preserves current libaom's squared two-component distance, first-centroid tie order, independently rounded U/V means, paired deterministic empty-cluster replacement, preceding-state retention on increased distortion, and 50-iteration limit. The source planes remain separate, with Vector512, Vector256, Vector128, then scalar dispatch through ImageSharp's shared vector-count helpers. An end-to-end improvement over the native implementation has not been established. Three independent tests cover exact paired convergence, midpoint initialization, 12-bit distance and index parity, untouched destination bounds, and every intrinsic tier. The exact Release test-project build reports 1,992 baseline warnings and zero errors; the focused three-case set, complete 8,934-case AVIF set, and complete 230-case HEIF set pass direct foreground net11 Release VSTest. Roslynk reports zero compiler errors and no touched-file analyzer warnings. Candidate integration and production activation remain in the open chroma-palette checkpoint. - [~] Live paired chroma palette selection now follows current libaom's complete 2-through-8 color-size search, U-plane neighbor-cache snapping, stable U-ordered color pairs, shared U/V index map, implicit DCT-DCT transform, and strict rate-distortion winner replacement. It omits the reference's early header-cost pruning, keeps planar U/V source data separate, and reuses the existing prediction, residual, transform, quantization, and reconstruction operators. Omitting pruning is an unresolved decision-policy deviation, not an established improvement. The production tile regression proves both palette-mode probability branches, exact paired colors and indices, coefficient-free reconstruction, and nonempty syntax. The complete 58-case intra-superblock set, 8,935-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains the next checkpoint. - [x] Production screen-content activation now matches current libaom's default good-quality detector: it scans only complete 16x16 luma blocks, normalizes palette samples to eight bits, admits 2-through-4-color blocks, and uses the reference's strict greater-than-ten-percent frame-area threshold. The same pass accumulates centered sums and squared sums at native precision, applies libaom's exact 10-bit and 12-bit variance rounding, and enables intra-block copy only when positive rounded per-pixel variance exceeds its strict one-twelfth frame-area threshold. A 256-bit stack bitset and fifth-color early exit replace libaom's larger per-block histogram without a second source scan or allocation. The adaptive sequence flag remains enabled and both frame flags are fixed before picture-state allocation. Focused regressions prove strict palette-threshold equality, high-bit-depth normalization, the exact variance rounding boundary, five-color rejection, emitted frame-header activation, actual production IBC selection, and production decode. The exact Release test-project build reports 1,992 baseline warnings and zero errors; all 9,242 non-HEVC HEIF/AV1 cases pass direct foreground net11 Release VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 31 payloads regenerated by the current test tree, including an actual IBC-coded 328x16 stream with decoded MD5 `677435e5af39c930af1178f91c34af6a`. Roslynk reports zero compiler errors and no touched-file analyzer warnings. -- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Paths above the speed-zero screen-content threshold continue through libaom's 256-pixel, one-pixel-step exhaustive mesh. Four adjacent byte or high-bit-depth candidates share each SIMD source load, strict row-major tie ordering is retained, and the final legal tail column remains searchable where libaom's current four-wide remainder loop omits it. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. Production mode decision now derives the same spatial displacement reference used by the writer, deduplicates hash and full-pixel finalists in search order, and evaluates every surviving vector through complete luma and chroma transform RD. This differs from libaom's preliminary-error pruning by permitting a hash and pixel finalist from the same search region to compete using final syntax and reconstruction costs. It is an unresolved controller deviation, with no established quality or performance improvement. Prediction is prepared once per plane and vector, including integer or half-sample chroma phase, then reused across every legal inter transform without an allocator rent or frame copy. The joint comparison includes the live intra-block-copy flag, discounted displacement rate, skip flag, coefficient syntax, and normalized Y/U/V distortion; an empty transform alternative can win only when its complete skip cost is strictly lower, while conventional intra and earlier vectors retain tie precedence. Winning reconstruction, coefficients, transform state, DC modes, cleared palette/filter/CfL state, and displacement are copied once into the existing retained stores. Production regressions force the path at 8 and 12 bits and force 4:2:0 horizontal half-sample chroma with an unaligned reference. The former bulk local workspace occupied 2.75 KiB for byte samples or 3.375 KiB for high-bit-depth samples. Prediction, candidate, winning reconstruction, residual, and coefficient scratch now occupy one naturally aligned 3.125 KiB extension of the existing frame-reused block-workspace owner, matching libaom's reusable macroblock-scratch lifetime without adding an allocation; only the 128-byte reference, weight, and finalist arrays remain on the stack. The net11 Release solution build reports zero errors; all 2,082 focused transform, entropy, intra-block-copy, intra-superblock, and frame-encoder cases and all 9,242 non-HEVC HEIF/AV1 cases pass through direct foreground VSTest, with tiered compilation disabled only for the full allocation-sensitive suite. Adaptive production activation is complete, and the emitted frame flag remains authoritative for the complete frame rather than being invalidated after tile coding. +- [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Paths above the speed-zero screen-content threshold continue through libaom's 256-pixel, one-pixel-step exhaustive mesh. Four adjacent byte or high-bit-depth candidates share each SIMD source load, strict row-major tie ordering is retained, and the final legal tail column remains searchable where libaom's current four-wide remainder loop omits it. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. Production mode decision now derives the same spatial displacement reference used by the writer, deduplicates hash and full-pixel finalists in search order, and evaluates every surviving vector through complete luma and chroma transform RD. This differs from libaom's preliminary-error pruning by permitting a hash and pixel finalist from the same search region to compete using final syntax and reconstruction costs. It is an unresolved controller deviation, with no established quality or performance improvement. Prediction is prepared once per plane and vector, including integer or half-sample chroma phase, then reused across every legal inter transform without an allocator rent or frame copy. The joint comparison includes the live intra-block-copy flag, discounted displacement rate, skip flag, coefficient syntax, and normalized Y/U/V distortion. The historical implementation required an empty transform and a strictly lower skip cost; the 2026-09-07 residual-skip correction replaces that rule with prediction-only residual RD, including ties. Conventional intra and earlier vectors still retain mode-selection tie precedence. Winning reconstruction, coefficients, transform state, DC modes, cleared palette/filter/CfL state, and displacement are copied once into the existing retained stores. Production regressions force the path at 8 and 12 bits and force 4:2:0 horizontal half-sample chroma with an unaligned reference. The former bulk local workspace occupied 2.75 KiB for byte samples or 3.375 KiB for high-bit-depth samples. Prediction, candidate, winning reconstruction, residual, and coefficient scratch now occupy one naturally aligned 3.125 KiB extension of the existing frame-reused block-workspace owner, matching libaom's reusable macroblock-scratch lifetime without adding an allocation; only the 128-byte reference, weight, and finalist arrays remain on the stack. The net11 Release solution build reports zero errors; all 2,082 focused transform, entropy, intra-block-copy, intra-superblock, and frame-encoder cases and all 9,242 non-HEVC HEIF/AV1 cases pass through direct foreground VSTest, with tiered compilation disabled only for the full allocation-sensitive suite. The frame flag remains authoritative through tile coding. Complete adaptive search activation and reference-controller parity are unverified. - [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips. - [x] Combined-frame OBU output now counts the byte-aligned frame and tile-group headers, non-final tile-size fields, and owned tile payloads before emitting the OBU size. It retains only the small allocator-owned header scratch and writes each entropy-coded tile span directly from its detached owner, removing the second file-sized allocator rent and complete-payload copy. A 64 KiB regression proves exactly one sub-payload-sized byte rent with a balanced return and verifies the exact streamed tile tail; the existing two-tile round trip proves size-prefix and ordering parity. The focused writer and production-frame set passes 32 of 32 direct net11 VSTest cases, current-main `aomdec` accepts all 29 generated native-format payloads, and the complete HEIF/AV1 namespace passes 8,860 of 8,860 cases with zero failures or skips. - [ ] The earlier fixed-block skip checkpoint was not equivalent to libaom's ordinary-intra policy. @@ -1720,7 +2803,7 @@ Encoder verification contract: - [x] Operation-wide allocation tracking now exercises a real 64x64 12-bit 4:4:4 frame through packed-pixel conversion, both native frame owners, picture and coefficient state, reusable block workspaces, entropy coding, OBU framing, and a non-seekable destination. It proves exactly one 60 KiB tile-output reservation from current libaom's all-intra 2.5x rule and balanced exactly-once returns for every tracked allocation before the operation completes. The focused ownership case passes 1 of 1 and the complete HEIF/AV1 namespace passes 8,863 of 8,863 direct net11 VSTest cases with zero failures or skips. - [x] HEIF box offsets are now counted from the start of the encoded file instead of reading `Stream.Position`. This preserves ISO BMFF file-relative `iloc` offsets when the destination begins at a nonzero position and permits non-seekable output. Decoder item extents and image-sequence chunk offsets now resolve from that same file origin rather than the backing stream origin. Real legacy-JPEG HEIF round trips cover non-seekable output and a prefixed destination, while current-position AV1 decode covers both a still item and a five-frame sequence. All 96 encoder/decoder cases and all 38 sequence-parser cases pass direct net11 Release VSTest; the Release build remains at the established 1,005-warning baseline with zero errors. -- [~] Current-libaom source comparison now drives uniform luma transform ownership at each effort boundary. Effort six retains the cheaper winner-only size decision for ordinary spatial and filter-intra modes, while every palette candidate already owns its size decision. Effort seven evaluates every legal 8x8 transform type for every ordinary spatial candidate. Effort eight and above make transform size part of every ordinary spatial and filter-intra candidate's rate-distortion result, so an 8x8-only preliminary comparison cannot discard the mode that wins with four 4x4 transforms. Prediction and subtraction are prepared once per mode and reused across transform types, matching the reference separation between prediction and transform search. Each 4x4 transform searches every legal type with coefficient contexts derived from retained transform edges and preceding trial blocks, while reconstructed top-right and bottom-left references follow production coding order. Palette prediction uses non-owning subregions of the retained color map, and filter-intra rebuilds each recursive prediction from reconstructed edges. The existing aligned block-workspace owner retains prediction, residual, coefficients, contexts, compact reconstruction, and four final states; no allocator rent, managed array, best-candidate re-transform, or full-block intermediate copy was added. Dense decision points now document scratch lifetime, enumeration tie order, global-winner publication, raster reconstruction dependencies, and the deliberate lower-effort shortcut. The packed encoder transform edges initialize to 64, matching libaom and the ImageSharp decoder before a coded neighbor publishes its size, and variable transform syntax remains gated to blocks larger than 4x4. The focused Release verification passes 13 of 13 cases across efforts zero through eight and ten, palette split selection, and transform-size selection. The complete non-HEVC HEIF/AV1 namespace passed 9,301 of 9,301 cases at that checkpoint. The `aomdec` built from the then-current `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` snapshot accepts the generated effort-eight and effort-ten streams. Partition search and effort-dependent pruning remain. +- [~] Uniform luma transform search currently changes ownership and enumeration at managed effort thresholds. Those thresholds and winner-only shortcuts are not validated native policy. The implementation reuses prediction/residual workspace and retains trial coefficient contexts and reconstruction, but candidate-stage versus winner-stage transform policy and cache invalidation still need the complete reference controller. Historical decodability and component-test results do not establish encoder parity or performance. - [x] Intra-block-copy transform search now prepares motion compensation and subtraction once per plane, alternates the existing candidate and selected work buffers whenever a transform improves, and performs at most one final normalization copy into the caller-owned selected span. This matches current libaom's pointer-swap ownership without adding an allocation or a third reconstruction buffer. Inline documentation now records the scratch lifetime, strict transform tie order, skip-rate replacement, unsplit transform-root syntax, joint-plane winner retention, and final publication boundary. The focused Release encoder and intra-block-copy set passes 17 of 17 cases, the complete non-HEVC HEIF/AV1 namespace passes 9,301 of 9,301 cases with zero failures or skips, and current-main `aomdec` accepts the regenerated effort-five and effort-six intra-block-copy streams. @@ -1752,7 +2835,7 @@ Encoder exit gate: - [ ] Current-main libaom accepts the payloads regenerated from the final tree. - [ ] Reverify lossless output at public pixel and native-plane precision for 8-, 10-, and 12-bit output. -- [ ] Separately encoded lossy outputs differ by no more than one unit at every decoded output sample with reconciled settings; report maxima and counts exceeding one. +- [ ] Encoder parity: separately encoded lossy results differ by no more than one component unit per sample with reconciled settings; report maxima and counts exceeding one. - [ ] Record equivalent end-to-end absolute timing, output size, quality, and allocation evidence after the source audit. - [ ] 8, 10, and 12-bit monochrome, 4:2:0, 4:2:2, and 4:4:4 outputs pass. - [ ] Alpha, grids, metadata, color profiles, transforms, and bounded sequences pass. diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderBlockWorkspace.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderBlockWorkspace.cs index efd96ea31f..f56182d144 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderBlockWorkspace.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderBlockWorkspace.cs @@ -39,6 +39,7 @@ internal sealed class Av1EncoderBlockWorkspace : IDisposable private const int ResidualStorageLength = MaximumResidualCount / 2; private const int MotionSearchSiteCount = 6; + private const int MotionSearchPredictionSampleCount = 128 * (128 + 8); private const int MotionSearchSiteStorageOffset = StorageLength + Av1MotionVectorCosts.StorageLength; private const int TransformCoefficientOffset = ResidualStorageLength; private const int DequantizedCoefficientOffset = TransformCoefficientOffset + MaximumCoefficientCount; @@ -80,9 +81,12 @@ internal sealed class Av1EncoderBlockWorkspace : IDisposable InterPredictionCoefficientStorageLength; private const int ModeDecisionStorageLength = Av1EncoderModeDecisionWorkspace.StorageLength; - private const int SharedModeDecisionStorageLength = ModeDecisionStorageLength > InterPredictionStorageLength + private const int InterSearchStorageLength = + InterPredictionStorageLength + (MotionSearchPredictionSampleCount * sizeof(ushort) / sizeof(int)); + + private const int SharedModeDecisionStorageLength = ModeDecisionStorageLength > InterSearchStorageLength ? ModeDecisionStorageLength - : InterPredictionStorageLength; + : InterSearchStorageLength; private const int PartitionContextStorageOffset = InterPredictionSampleStorageOffset + SharedModeDecisionStorageLength; @@ -186,6 +190,20 @@ internal sealed class Av1EncoderBlockWorkspace : IDisposable public Av1MotionVectorCosts GetMotionVectorCosts(Av1MotionVectorPrecision precision) => new(this.owner.Memory.Span.Slice(StorageLength, Av1MotionVectorCosts.StorageLength), precision); + /// + /// Borrows prediction samples for motion search while retaining the selected inter reconstruction. + /// + /// The frame's unsigned sample storage type. + /// The reusable search prediction span. + public Span GetMotionSearchPrediction() + where TSample : unmanaged + { + // Intra trials have finished before inter search starts. Reuse their storage beyond the live inter + // candidate buffers; the extra eight rows accommodate separable filtering of a 128x128 prediction. + int offset = InterPredictionSampleStorageOffset + InterPredictionStorageLength; + return MemoryMarshal.Cast(this.owner.Memory.Span[offset..])[..MotionSearchPredictionSampleCount]; + } + /// /// Gets the retained full-pixel search geometry for the reference plane's current stride. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs index 3f63afff57..d511deb143 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs @@ -184,22 +184,23 @@ internal readonly struct Av1EncoderFrame Av1Math.AlignPowerOf2(height, CodedDimensionAlignmentLog2)); /// - /// Calculates the physical dimensions required for an all-intra component plane. + /// Calculates the physical dimensions required for a bordered component plane. /// /// The visible luma width. /// The visible luma height. /// The plane's horizontal subsampling shift. /// The plane's vertical subsampling shift. + /// The border width and height in luma samples. /// The physical plane dimensions, including its complete border and row padding. - public static Size GetPlaneBufferSize(int width, int height, int subsamplingX, int subsamplingY) + public static Size GetPlaneBufferSize(int width, int height, int subsamplingX, int subsamplingY, int lumaBorder) { Size codedSize = GetCodedSize(width, height); - // libaom aligns the complete luma row before deriving a subsampled plane's stride. + // Align the complete luma row before deriving a subsampled plane's stride. // Aligning chroma independently would produce a different physical layout for narrow or odd-sized frames. - int lumaStride = Av1Math.AlignPowerOf2(codedSize.Width + (2 * LumaBorder), LumaStrideAlignmentLog2); + int lumaStride = Av1Math.AlignPowerOf2(codedSize.Width + (2 * lumaBorder), LumaStrideAlignmentLog2); int planeStride = lumaStride >> subsamplingX; - int planeBorderHeight = LumaBorder >> subsamplingY; + int planeBorderHeight = lumaBorder >> subsamplingY; return new Size(planeStride, (codedSize.Height >> subsamplingY) + (2 * planeBorderHeight)); } diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrameBuffer.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrameBuffer.cs index 1742b94ed0..78bfe364ca 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrameBuffer.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrameBuffer.cs @@ -15,7 +15,7 @@ internal sealed class Av1EncoderFrameBuffer : IDisposable where TSample : unmanaged { /// - /// The byte boundary used by libaom for SIMD-accessible component planes. + /// The byte boundary used for SIMD-accessible component planes. /// private const int PlaneAlignmentBytes = 32; @@ -34,6 +34,7 @@ internal sealed class Av1EncoderFrameBuffer : IDisposable /// The native luma and chroma sampling layout. /// The horizontal chroma position in half-luma-sample units. /// The vertical chroma position in half-luma-sample units. + /// The border width and height in luma samples. public Av1EncoderFrameBuffer( Configuration configuration, int width, @@ -41,16 +42,17 @@ internal sealed class Av1EncoderFrameBuffer : IDisposable int bitDepth, Av1ColorFormat colorFormat, int chromaPositionX, - int chromaPositionY) + int chromaPositionY, + int lumaBorder) { int subsamplingX = colorFormat is Av1ColorFormat.Yuv420 or Av1ColorFormat.Yuv422 ? 1 : 0; int subsamplingY = colorFormat == Av1ColorFormat.Yuv420 ? 1 : 0; Size codedSize = Av1EncoderFrame.GetCodedSize(width, height); - Size lumaSize = Av1EncoderFrame.GetPlaneBufferSize(width, height, 0, 0); + Size lumaSize = Av1EncoderFrame.GetPlaneBufferSize(width, height, 0, 0, lumaBorder); int lumaElementCount = checked(lumaSize.Width * lumaSize.Height); Size chromaSize = colorFormat == Av1ColorFormat.Yuv400 ? Size.Empty - : Av1EncoderFrame.GetPlaneBufferSize(width, height, subsamplingX, subsamplingY); + : Av1EncoderFrame.GetPlaneBufferSize(width, height, subsamplingX, subsamplingY, lumaBorder); int chromaElementCount = checked(chromaSize.Width * chromaSize.Height); int planeAlignment = Math.Max(PlaneAlignmentBytes / Unsafe.SizeOf(), 1); @@ -60,8 +62,8 @@ internal sealed class Av1EncoderFrameBuffer : IDisposable ? lumaElementCount : checked(chromaRedOffset + chromaElementCount); - // Libaom keeps the three component planes in one 32-byte-aligned frame allocation. The non-owning - // Buffer2D views preserve ImageSharp's row API without introducing separate plane rents or copies. + // Component planes share one frame allocation; their offsets preserve the 32-byte plane alignment. + // Non-owning Buffer2D views expose rows without introducing separate plane rents or copies. IMemoryOwner owner = configuration.MemoryAllocator.Allocate(storageLength); Memory storage = owner.Memory; Buffer2D luma = Buffer2D.WrapMemory( @@ -72,8 +74,8 @@ internal sealed class Av1EncoderFrameBuffer : IDisposable this.Luma = luma; Buffer2DRegion lumaRegion = luma.GetRegion( - Av1EncoderFrame.LumaBorder, - Av1EncoderFrame.LumaBorder, + lumaBorder, + lumaBorder, codedSize.Width, codedSize.Height); @@ -94,8 +96,8 @@ internal sealed class Av1EncoderFrameBuffer : IDisposable this.ChromaBlue = chromaBlue; this.ChromaRed = chromaRed; - int chromaBorderX = Av1EncoderFrame.LumaBorder >> subsamplingX; - int chromaBorderY = Av1EncoderFrame.LumaBorder >> subsamplingY; + int chromaBorderX = lumaBorder >> subsamplingX; + int chromaBorderY = lumaBorder >> subsamplingY; int codedChromaWidth = codedSize.Width >> subsamplingX; int codedChromaHeight = codedSize.Height >> subsamplingY; chromaBlueRegion = chromaBlue.GetRegion( diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs index 69ef8c6f70..8da318fb82 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1FrameEncoder.cs @@ -272,8 +272,9 @@ internal static class Av1FrameEncoder int height, ObuColorConfig colorConfig, int qIndex, - int effort) - => CreateSequenceEncoder(configuration, width, height, colorConfig, qIndex, effort, false); + int effort, + HeifEncodingSpeed speed) + => CreateSequenceEncoder(configuration, width, height, colorConfig, qIndex, effort, speed, false); /// /// Creates an encoder that retains reconstructed alpha frames for prediction by later samples in the sequence. @@ -284,8 +285,9 @@ internal static class Av1FrameEncoder int height, ObuColorConfig colorConfig, int qIndex, - int effort) - => CreateSequenceEncoder(configuration, width, height, colorConfig, qIndex, effort, true); + int effort, + HeifEncodingSpeed speed) + => CreateSequenceEncoder(configuration, width, height, colorConfig, qIndex, effort, speed, true); private static ObuSequenceHeader Encode( Configuration configuration, @@ -359,14 +361,15 @@ internal static class Av1FrameEncoder ObuColorConfig colorConfig, int qIndex, int effort, + HeifEncodingSpeed speed, bool encodeAlpha) { if (colorConfig.BitDepth == Av1BitDepth.EightBit) { - return new ByteSequenceEncoder(configuration, width, height, colorConfig, qIndex, effort, encodeAlpha); + return new ByteSequenceEncoder(configuration, width, height, colorConfig, qIndex, effort, speed, encodeAlpha); } - return new HighBitDepthSequenceEncoder(configuration, width, height, colorConfig, qIndex, effort, encodeAlpha); + return new HighBitDepthSequenceEncoder(configuration, width, height, colorConfig, qIndex, effort, speed, encodeAlpha); } private static ObuSequenceHeader CreateSequenceHeader( @@ -622,7 +625,8 @@ internal static class Av1FrameEncoder ByteSampleBitDepth, colorFormat, chromaPositionX: CenteredChromaSamplePosition, - chromaPositionY: CenteredChromaSamplePosition); + chromaPositionY: CenteredChromaSamplePosition, + lumaBorder: Av1EncoderFrame.LumaBorder); using Av1EncoderFrameBuffer reconstruction = new( configuration, @@ -631,7 +635,8 @@ internal static class Av1FrameEncoder ByteSampleBitDepth, colorFormat, chromaPositionX: CenteredChromaSamplePosition, - chromaPositionY: CenteredChromaSamplePosition); + chromaPositionY: CenteredChromaSamplePosition, + lumaBorder: Av1EncoderFrame.LumaBorder); using Av1EncoderCoefficientBuffer coefficients = new( configuration, @@ -651,7 +656,7 @@ internal static class Av1FrameEncoder using ObuWriter obuWriter = new(configuration); - PrepareFrame( + bool isScreenContent = PrepareFrame( configuration, image, sourceRectangle, @@ -670,6 +675,7 @@ internal static class Av1FrameEncoder source.Frame.Height, disallow4x4AllFrames: !frameHeader.CodedLossless && effort < 9); + picture.Picture.Parent.IsScreenContent = isScreenContent; Encode( obuWriter, stream, @@ -709,7 +715,8 @@ internal static class Av1FrameEncoder bitDepth, colorFormat, chromaPositionX: CenteredChromaSamplePosition, - chromaPositionY: CenteredChromaSamplePosition); + chromaPositionY: CenteredChromaSamplePosition, + lumaBorder: Av1EncoderFrame.LumaBorder); using Av1EncoderFrameBuffer reconstruction = new( configuration, @@ -718,7 +725,8 @@ internal static class Av1FrameEncoder bitDepth, colorFormat, chromaPositionX: CenteredChromaSamplePosition, - chromaPositionY: CenteredChromaSamplePosition); + chromaPositionY: CenteredChromaSamplePosition, + lumaBorder: Av1EncoderFrame.LumaBorder); using Av1EncoderCoefficientBuffer coefficients = new( configuration, @@ -738,7 +746,7 @@ internal static class Av1FrameEncoder using ObuWriter obuWriter = new(configuration); - PrepareFrame( + bool isScreenContent = PrepareFrame( configuration, image, sourceRectangle, @@ -757,6 +765,7 @@ internal static class Av1FrameEncoder source.Frame.Height, disallow4x4AllFrames: !frameHeader.CodedLossless && effort < 9); + picture.Picture.Parent.IsScreenContent = isScreenContent; Encode( obuWriter, stream, @@ -777,7 +786,7 @@ internal static class Av1FrameEncoder /// /// Converts one source frame and resolves every content-dependent coding tool before picture-state allocation. /// - private static void PrepareFrame( + private static bool PrepareFrame( Configuration configuration, ImageFrame image, Rectangle sourceRectangle, @@ -797,7 +806,7 @@ internal static class Av1FrameEncoder sequenceHeader.ColorConfig, encodeAlpha); - ConfigureFrameTools( + return ConfigureFrameTools( source, reference, sequenceHeader, @@ -808,7 +817,7 @@ internal static class Av1FrameEncoder /// /// Converts one sequence sample through its retained row workspace before resolving frame coding tools. /// - private static void PrepareFrame( + private static bool PrepareFrame( Configuration configuration, ImageFrame image, Rectangle sourceRectangle, @@ -827,7 +836,7 @@ internal static class Av1FrameEncoder source, conversionWorkspace); - ConfigureFrameTools( + return ConfigureFrameTools( source, reference, sequenceHeader, @@ -838,7 +847,7 @@ internal static class Av1FrameEncoder /// /// Resolves the eight-bit frame tools whose syntax depends on the converted source samples. /// - private static void ConfigureFrameTools( + private static bool ConfigureFrameTools( Av1EncoderFrame source, Av1EncoderFrame reference, ObuSequenceHeader sequenceHeader, @@ -852,32 +861,28 @@ internal static class Av1FrameEncoder sequenceHeader.ColorConfig.BitDepth, effort); - bool allowScreenContentTools = false; - bool allowIntraBlockCopy = false; - if (effort >= 5) - { - // Lower effort levels never search palette or intra-block-copy modes, so scanning the complete - // luma plane cannot affect their bitstream decisions. - Av1ScreenContentDetector.Detect( - source, - out allowScreenContentTools, - out allowIntraBlockCopy); - } + bool isScreenContent = Av1ScreenContentDetector.Detect( + source, + out bool allowScreenContentTools, + out bool allowIntraBlockCopy); - frameHeader.AllowScreenContentTools = allowScreenContentTools; + frameHeader.AllowScreenContentTools = effort >= 5 && allowScreenContentTools; // The current intra-block-copy search owns one 8x8 transform. Lossless coding requires reversible // 4x4 transforms, so palette remains available while this incompatible candidate is omitted. frameHeader.AllowIntraBlockCopy = frameHeader.IsIntra && !frameHeader.CodedLossless && + frameHeader.AllowScreenContentTools && allowIntraBlockCopy; + + return isScreenContent; } /// /// Converts one high-bit-depth source frame and resolves every content-dependent coding tool before picture-state allocation. /// - private static void PrepareFrame( + private static bool PrepareFrame( Configuration configuration, ImageFrame image, Rectangle sourceRectangle, @@ -897,7 +902,7 @@ internal static class Av1FrameEncoder sequenceHeader.ColorConfig, encodeAlpha); - ConfigureFrameTools( + return ConfigureFrameTools( source, reference, sequenceHeader, @@ -908,7 +913,7 @@ internal static class Av1FrameEncoder /// /// Converts one high-bit-depth sequence sample through retained row storage before resolving frame coding tools. /// - private static void PrepareFrame( + private static bool PrepareFrame( Configuration configuration, ImageFrame image, Rectangle sourceRectangle, @@ -927,7 +932,7 @@ internal static class Av1FrameEncoder source, conversionWorkspace); - ConfigureFrameTools( + return ConfigureFrameTools( source, reference, sequenceHeader, @@ -938,7 +943,7 @@ internal static class Av1FrameEncoder /// /// Resolves the high-bit-depth frame tools whose syntax depends on the converted source samples. /// - private static void ConfigureFrameTools( + private static bool ConfigureFrameTools( Av1EncoderFrame source, Av1EncoderFrame reference, ObuSequenceHeader sequenceHeader, @@ -952,26 +957,22 @@ internal static class Av1FrameEncoder sequenceHeader.ColorConfig.BitDepth, effort); - bool allowScreenContentTools = false; - bool allowIntraBlockCopy = false; - if (effort >= 5) - { - // Lower effort levels never search palette or intra-block-copy modes, so scanning the complete - // luma plane cannot affect their bitstream decisions. - Av1ScreenContentDetector.Detect( - source, - out allowScreenContentTools, - out allowIntraBlockCopy); - } + bool isScreenContent = Av1ScreenContentDetector.Detect( + source, + out bool allowScreenContentTools, + out bool allowIntraBlockCopy); - frameHeader.AllowScreenContentTools = allowScreenContentTools; + frameHeader.AllowScreenContentTools = effort >= 5 && allowScreenContentTools; // The current intra-block-copy search owns one 8x8 transform. Lossless coding requires reversible // 4x4 transforms, so palette remains available while this incompatible candidate is omitted. frameHeader.AllowIntraBlockCopy = frameHeader.IsIntra && !frameHeader.CodedLossless && + frameHeader.AllowScreenContentTools && allowIntraBlockCopy; + + return isScreenContent; } private static void Encode( @@ -1166,7 +1167,7 @@ internal static class Av1FrameEncoder int effortShift = effort - MinimumGlobalMotionSearchEffort; int searchRadius = Math.Min( MinimumGlobalMotionSearchRadius << effortShift, - Av1EncoderFrame.LumaBorder); + Math.Min(referenceLuma.Bounds.X, referenceLuma.Bounds.Y)); Point bestOffset = default; long bestAnalysisError = GetGlobalMotionSquaredError( @@ -1499,6 +1500,7 @@ internal static class Av1FrameEncoder ObuColorConfig colorConfig, int qIndex, int effort, + HeifEncodingSpeed speed, bool encodeAlpha, bool usesHighBitDepth) { @@ -1548,10 +1550,11 @@ internal static class Av1FrameEncoder width, height); + this.PictureBuffer.Picture.Parent.EncodingSpeed = speed; this.SuperblockWorkspace = new Av1EncoderSuperblockWorkspace(configuration); this.TileWorkspace = new Av1EncoderTileWorkspace(this.FrameHeader, this.SuperblockWorkspace); - this.BlockWorkspace = new Av1EncoderBlockWorkspace(configuration); + this.BlockWorkspace = new Av1EncoderBlockWorkspace(configuration, allocateInterMotionCosts: true); // Tile probabilities adapt within a sample, while error-resilient frame headers prohibit carrying // those updates into the next sample. The retained encoder is therefore reset before each frame. @@ -1692,6 +1695,7 @@ internal static class Av1FrameEncoder ObuColorConfig colorConfig, int qIndex, int effort, + HeifEncodingSpeed speed, bool encodeAlpha) : base( configuration, @@ -1700,12 +1704,17 @@ internal static class Av1FrameEncoder colorConfig, qIndex, effort, + speed, encodeAlpha, usesHighBitDepth: false) { try { Av1ColorFormat colorFormat = colorConfig.GetColorFormat(); + + // Inter prediction needs a complete superblock beyond the image plus interpolation and alignment margins. + int lumaBorder = (this.SequenceHeader.Use128x128Superblock ? 128 : 64) + 32; + this.source = new( configuration, width, @@ -1713,7 +1722,8 @@ internal static class Av1FrameEncoder ByteSampleBitDepth, colorFormat, CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + CenteredChromaSamplePosition, + lumaBorder); this.reference = new( configuration, @@ -1722,7 +1732,8 @@ internal static class Av1FrameEncoder ByteSampleBitDepth, colorFormat, CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + CenteredChromaSamplePosition, + lumaBorder); this.reconstruction = new( configuration, @@ -1731,7 +1742,8 @@ internal static class Av1FrameEncoder ByteSampleBitDepth, colorFormat, CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + CenteredChromaSamplePosition, + lumaBorder); } catch { @@ -1764,7 +1776,7 @@ internal static class Av1FrameEncoder Rectangle sourceRectangle = new(0, 0, image.Width, image.Height); this.SymbolEncoder.Reset(); - PrepareFrame( + bool isScreenContent = PrepareFrame( this.Configuration, image, sourceRectangle, @@ -1776,6 +1788,7 @@ internal static class Av1FrameEncoder this.ConversionWorkspace); this.PictureBuffer.Reset(frameHeader); + this.PictureBuffer.Picture.Parent.IsScreenContent = isScreenContent; Encode( this.ObuWriter, stream, @@ -1812,6 +1825,7 @@ internal static class Av1FrameEncoder ObuColorConfig colorConfig, int qIndex, int effort, + HeifEncodingSpeed speed, bool encodeAlpha) : base( configuration, @@ -1820,6 +1834,7 @@ internal static class Av1FrameEncoder colorConfig, qIndex, effort, + speed, encodeAlpha, usesHighBitDepth: true) { @@ -1827,6 +1842,10 @@ internal static class Av1FrameEncoder { int bitDepth = colorConfig.BitDepth.GetBitCount(); Av1ColorFormat colorFormat = colorConfig.GetColorFormat(); + + // Inter prediction needs a complete superblock beyond the image plus interpolation and alignment margins. + int lumaBorder = (this.SequenceHeader.Use128x128Superblock ? 128 : 64) + 32; + this.source = new( configuration, width, @@ -1834,7 +1853,8 @@ internal static class Av1FrameEncoder bitDepth, colorFormat, CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + CenteredChromaSamplePosition, + lumaBorder); this.reference = new( configuration, @@ -1843,7 +1863,8 @@ internal static class Av1FrameEncoder bitDepth, colorFormat, CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + CenteredChromaSamplePosition, + lumaBorder); this.reconstruction = new( configuration, @@ -1852,7 +1873,8 @@ internal static class Av1FrameEncoder bitDepth, colorFormat, CenteredChromaSamplePosition, - CenteredChromaSamplePosition); + CenteredChromaSamplePosition, + lumaBorder); } catch { @@ -1885,7 +1907,7 @@ internal static class Av1FrameEncoder Rectangle sourceRectangle = new(0, 0, image.Width, image.Height); this.SymbolEncoder.Reset(); - PrepareFrame( + bool isScreenContent = PrepareFrame( this.Configuration, image, sourceRectangle, @@ -1897,6 +1919,7 @@ internal static class Av1FrameEncoder this.ConversionWorkspace); this.PictureBuffer.Reset(frameHeader); + this.PictureBuffer.Picture.Parent.IsScreenContent = isScreenContent; Encode( this.ObuWriter, stream, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs index 05e148eb17..8846589978 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs @@ -180,6 +180,9 @@ internal static partial class Av1IntraSuperblockEncoder this.SelectedBlockStatistics = default; } + /// + public static bool UsesRetainedDecisions => false; + /// /// Gets the statistics of the most recently encoded block. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs index 7bf726c86a..8abfe95890 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.Operator.cs @@ -19,16 +19,13 @@ namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; /// internal static partial class Av1IntraSuperblockEncoder { - /// - /// The width and height of the fixed block currently used by inter motion search. - /// - private const int InterSearchBlockDimension = 8; - /// /// Defines type-specific block encoding without coupling traversal to sample storage width. /// /// The native unsigned sample storage type. - internal interface IBlockEncodingOperator : Av1IntraBlockCopySearchIndex.ISearchOperation + internal interface IBlockEncodingOperator : + Av1IntraBlockCopySearchIndex.ISearchOperation, + Av1MotionSearchBase.IMotionSearchOperator where TSample : unmanaged { /// @@ -330,22 +327,6 @@ internal static partial class Av1IntraSuperblockEncoder Av1TransformSize transformSize, Av1BitDepth bitDepth); - /// - /// Measures an 8x8 full-pixel reference candidate through the bordered plane storage. - /// - /// The coded source plane. - /// The source block origin in visible-plane coordinates. - /// The padded retained reference plane. - /// The candidate origin, which may lie inside the physical border. - /// The coded sample precision. - /// The squared error normalized to the eight-bit distortion domain. - public static abstract long GetInterPredictionError( - Buffer2DRegion source, - Point sourceOrigin, - Buffer2DRegion reference, - Point predictionOrigin, - Av1BitDepth bitDepth); - /// /// Encodes one prepared prediction with the selected transform into decision scratch. /// @@ -442,6 +423,80 @@ internal static partial class Av1IntraSuperblockEncoder /// internal readonly struct ByteOperator : IBlockEncodingOperator { + /// + public static void PreparePrediction( + ReadOnlySpan source, + int sourceStride, + ReadOnlySpan reference, + int referenceStride, + int referenceOrigin, + Span prediction, + Span residual, + Span scratch, + int width, + int height, + Av1InterpolationFilter horizontalFilter, + Av1InterpolationFilter verticalFilter, + int horizontalPhase, + int verticalPhase, + int bitDepth) + => Av1MotionSearchBase.ByteOperator.PreparePrediction( + source, + sourceStride, + reference, + referenceStride, + referenceOrigin, + prediction, + residual, + scratch, + width, + height, + horizontalFilter, + verticalFilter, + horizontalPhase, + verticalPhase, + bitDepth); + + /// + public static void Predict( + ReadOnlySpan reference, + int referenceStride, + int referenceOrigin, + Span buffer, + int width, + int height, + int horizontalPhase, + int verticalPhase, + int taps, + int bitDepth) + => Av1MotionSearchBase.ByteOperator.Predict( + reference, referenceStride, referenceOrigin, buffer, width, height, horizontalPhase, verticalPhase, taps, bitDepth); + + /// + public static int SumAbsoluteDifferences( + ReadOnlySpan source, + int sourceStride, + ReadOnlySpan prediction, + int predictionStride, + int width, + int height, + int rowStep) + => Av1MotionSearchBase.ByteOperator.SumAbsoluteDifferences( + source, sourceStride, prediction, predictionStride, width, height, rowStep); + + /// + public static void GetMoments( + ReadOnlySpan source, + int sourceStride, + ReadOnlySpan prediction, + int predictionStride, + int width, + int height, + out int sum, + out long squares) + => Av1MotionSearchBase.ByteOperator.GetMoments( + source, sourceStride, prediction, predictionStride, width, height, out sum, out squares); + /// public static Span GetLeftReference(Span residual, int length) => MemoryMarshal.AsBytes(residual)[..length]; @@ -470,36 +525,6 @@ internal static partial class Av1IntraSuperblockEncoder return true; } - /// - public static long GetInterPredictionError( - Buffer2DRegion source, - Point sourceOrigin, - Buffer2DRegion reference, - Point predictionOrigin, - Av1BitDepth bitDepth) - { - Rectangle sourceBounds = source.Bounds; - Rectangle referenceBounds = reference.Bounds; - int sourceIndex = - ((sourceBounds.Y + sourceOrigin.Y) * source.Stride) + - sourceBounds.X + - sourceOrigin.X; - - int referenceIndex = - ((referenceBounds.Y + predictionOrigin.Y) * reference.Stride) + - referenceBounds.X + - predictionOrigin.X; - - // The shared residual kernel selects the widest available vector width and handles the scalar tail. - return Av1ResidualBuilder.SumSquaredError( - source.Buffer.DangerousGetSingleSpan()[sourceIndex..], - source.Stride, - reference.Buffer.DangerousGetSingleSpan()[referenceIndex..], - reference.Stride, - InterSearchBlockDimension, - InterSearchBlockDimension); - } - /// public static int GetSumOfAbsoluteDifferences( Buffer2DRegion source, @@ -962,6 +987,80 @@ internal static partial class Av1IntraSuperblockEncoder /// internal readonly struct UInt16Operator : IBlockEncodingOperator { + /// + public static void PreparePrediction( + ReadOnlySpan source, + int sourceStride, + ReadOnlySpan reference, + int referenceStride, + int referenceOrigin, + Span prediction, + Span residual, + Span scratch, + int width, + int height, + Av1InterpolationFilter horizontalFilter, + Av1InterpolationFilter verticalFilter, + int horizontalPhase, + int verticalPhase, + int bitDepth) + => Av1MotionSearchBase.UInt16Operator.PreparePrediction( + source, + sourceStride, + reference, + referenceStride, + referenceOrigin, + prediction, + residual, + scratch, + width, + height, + horizontalFilter, + verticalFilter, + horizontalPhase, + verticalPhase, + bitDepth); + + /// + public static void Predict( + ReadOnlySpan reference, + int referenceStride, + int referenceOrigin, + Span buffer, + int width, + int height, + int horizontalPhase, + int verticalPhase, + int taps, + int bitDepth) + => Av1MotionSearchBase.UInt16Operator.Predict( + reference, referenceStride, referenceOrigin, buffer, width, height, horizontalPhase, verticalPhase, taps, bitDepth); + + /// + public static int SumAbsoluteDifferences( + ReadOnlySpan source, + int sourceStride, + ReadOnlySpan prediction, + int predictionStride, + int width, + int height, + int rowStep) + => Av1MotionSearchBase.UInt16Operator.SumAbsoluteDifferences( + source, sourceStride, prediction, predictionStride, width, height, rowStep); + + /// + public static void GetMoments( + ReadOnlySpan source, + int sourceStride, + ReadOnlySpan prediction, + int predictionStride, + int width, + int height, + out int sum, + out long squares) + => Av1MotionSearchBase.UInt16Operator.GetMoments( + source, sourceStride, prediction, predictionStride, width, height, out sum, out squares); + /// public static Span GetLeftReference(Span residual, int length) => MemoryMarshal.Cast(residual)[..length]; @@ -997,38 +1096,6 @@ internal static partial class Av1IntraSuperblockEncoder return true; } - /// - public static long GetInterPredictionError( - Buffer2DRegion source, - Point sourceOrigin, - Buffer2DRegion reference, - Point predictionOrigin, - Av1BitDepth bitDepth) - { - Rectangle sourceBounds = source.Bounds; - Rectangle referenceBounds = reference.Bounds; - int sourceIndex = - ((sourceBounds.Y + sourceOrigin.Y) * source.Stride) + - sourceBounds.X + - sourceOrigin.X; - - int referenceIndex = - ((referenceBounds.Y + predictionOrigin.Y) * reference.Stride) + - referenceBounds.X + - predictionOrigin.X; - - long error = Av1ResidualBuilder.SumSquaredError( - source.Buffer.DangerousGetSingleSpan()[sourceIndex..], - source.Stride, - reference.Buffer.DangerousGetSingleSpan()[referenceIndex..], - reference.Stride, - InterSearchBlockDimension, - InterSearchBlockDimension); - - int shift = (bitDepth.GetBitCount() - 8) * 2; - return shift == 0 ? error : (error + (1L << (shift - 1))) >> shift; - } - /// public static int GetSumOfAbsoluteDifferences( Buffer2DRegion source, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ReferenceModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ReferenceModeDecision.cs index 229b809730..be582a4a60 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ReferenceModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ReferenceModeDecision.cs @@ -18,16 +18,6 @@ namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; /// internal static partial class Av1IntraSuperblockEncoder { - /// - /// The first effort tier that searches a block-local motion vector. - /// - private const int MinimumInterMotionSearchEffort = 6; - - /// - /// The smallest full-pixel radius used by block-local inter search. - /// - private const int MinimumInterMotionSearchRadius = 4; - /// /// The first effort tier that refines full-pixel motion to quarter-pixel precision. /// @@ -38,16 +28,6 @@ internal static partial class Av1IntraSuperblockEncoder /// private const int MinimumHighPrecisionMotionSearchEffort = 8; - /// - /// The physical border reserved on each side for fractional eight-tap filtering. - /// - private const int FractionalInterpolationBorder = 4; - - /// - /// The number of cardinal and diagonal candidates examined at each search step. - /// - private const int InterMotionSearchDirectionCount = 8; - /// /// One nearest, three near, one global, and three new-motion candidates. /// @@ -236,22 +216,16 @@ internal static partial class Av1IntraSuperblockEncoder out Av1EncoderTransformBlockState lumaCandidateState, out int lumaRate, out long lumaDistortion, - out bool hasEmptyLuma, - out Av1EncoderTransformBlockState emptyLumaState, - out long emptyLumaDistortion); + out long lumaPredictionDistortion); int blueRate = 0; int redRate = 0; long blueDistortion = 0; long redDistortion = 0; - long emptyBlueDistortion = 0; - long emptyRedDistortion = 0; - bool hasEmptyBlue = true; - bool hasEmptyRed = true; + long bluePredictionDistortion = 0; + long redPredictionDistortion = 0; Av1EncoderTransformBlockState blueCandidateState = default; Av1EncoderTransformBlockState redCandidateState = default; - Av1EncoderTransformBlockState emptyBlueState = default; - Av1EncoderTransformBlockState emptyRedState = default; if (!this.source.IsMonochrome) { Av1TransformType chromaTransformType = lumaCandidateState.TransformType; @@ -292,9 +266,7 @@ internal static partial class Av1IntraSuperblockEncoder out blueCandidateState, out blueRate, out blueDistortion, - out hasEmptyBlue, - out emptyBlueState, - out emptyBlueDistortion); + out bluePredictionDistortion); this.EvaluateInterPlane( writer, @@ -321,40 +293,32 @@ internal static partial class Av1IntraSuperblockEncoder out redCandidateState, out redRate, out redDistortion, - out hasEmptyRed, - out emptyRedState, - out emptyRedDistortion); + out redPredictionDistortion); } - int displacementRate = writer.GetDisplacementVectorCost(candidate, reference); - int candidateRate = writer.GetUseIntraBlockCopyCost(true) + - displacementRate + - writer.GetSkipCost(false, skipContext) + + int predictionRate = writer.GetUseIntraBlockCopyCost(true) + + writer.GetDisplacementVectorCost(candidate, reference); + + int residualRate = writer.GetSkipCost(false, skipContext) + transformPartitionRate + lumaRate + blueRate + redRate; long candidateDistortion = lumaDistortion + blueDistortion + redDistortion; - Av1RateDistortionStatistics candidateStatistics = new(this.rateMultiplier, candidateRate, candidateDistortion); - bool candidateSkip = false; + int skipRate = writer.GetSkipCost(true, skipContext); + long skipDistortion = lumaPredictionDistortion + bluePredictionDistortion + redPredictionDistortion; - // The skip alternative is available only when every coded plane has an empty transform. Its - // distortion comes from prediction alone and its rate excludes the transform tree and coefficients. - if (hasEmptyLuma && hasEmptyBlue && hasEmptyRed) - { - int skipRate = writer.GetUseIntraBlockCopyCost(true) + - displacementRate + - writer.GetSkipCost(true, skipContext); + // Empty residuals omit the transform tree. Nonempty residuals may also be discarded when + // prediction alone costs no more; exclude shared prediction syntax before rounding either rate. + bool candidateSkip = (lumaCandidateState.EndOfBlock == 0 && + blueCandidateState.EndOfBlock == 0 && redCandidateState.EndOfBlock == 0) || + Av1RateDistortion.GetCost(this.rateMultiplier, skipRate, skipDistortion) <= + Av1RateDistortion.GetCost(this.rateMultiplier, residualRate, candidateDistortion); - long skipDistortion = emptyLumaDistortion + emptyBlueDistortion + emptyRedDistortion; - Av1RateDistortionStatistics skipStatistics = new(this.rateMultiplier, skipRate, skipDistortion); - if (skipStatistics.Cost < candidateStatistics.Cost) - { - candidateStatistics = skipStatistics; - candidateSkip = true; - } - } + Av1RateDistortionStatistics candidateStatistics = candidateSkip + ? new(this.rateMultiplier, predictionRate + skipRate, skipDistortion) + : new(this.rateMultiplier, predictionRate + residualRate, candidateDistortion); // Conventional intra and earlier IBC vectors retain strict search-order precedence on equal RD. if (candidateStatistics.Cost >= bestStatistics.Cost) @@ -370,7 +334,7 @@ internal static partial class Av1IntraSuperblockEncoder { workspace.LumaPrediction.CopyTo(workspace.SelectedLumaReconstruction); workspace.SelectedLumaCoefficients.Clear(); - selectedLumaState = emptyLumaState; + selectedLumaState = default; if (!this.source.IsMonochrome) { int chromaSampleCount = chromaTransformSize.GetSize2d(); @@ -378,8 +342,8 @@ internal static partial class Av1IntraSuperblockEncoder workspace.RedPrediction[..chromaSampleCount].CopyTo(workspace.SelectedRedReconstruction); workspace.SelectedBlueCoefficients[..chromaSampleCount].Clear(); workspace.SelectedRedCoefficients[..chromaSampleCount].Clear(); - selectedBlueState = emptyBlueState; - selectedRedState = emptyRedState; + selectedBlueState = default; + selectedRedState = default; } } else @@ -577,45 +541,31 @@ internal static partial class Av1IntraSuperblockEncoder Span candidateModes = stackalloc Av1PredictionMode[MaximumInterModeCandidateCount]; Span candidateReferenceIndices = stackalloc byte[MaximumInterModeCandidateCount]; int candidateCount = 0; - if (this.effort >= MinimumInterMotionSearchEffort) + + // Keep distinct syntax choices even when their prediction vectors are equal. + candidateVectors[candidateCount] = referenceMotionVectors.Nearest; + candidateModes[candidateCount] = Av1PredictionMode.NearestMotionVector; + candidateReferenceIndices[candidateCount++] = 0; + + int maximumNewIndex = Math.Min(2, Math.Max(0, referenceMotionVectors.Count - 1)); + for (int referenceIndex = 0; referenceIndex <= maximumNewIndex; referenceIndex++) { - // Predictor-stack modes precede global and new motion so strict ties retain the reference order. - candidateVectors[candidateCount] = referenceMotionVectors.Nearest; - candidateModes[candidateCount] = Av1PredictionMode.NearestMotionVector; - candidateReferenceIndices[candidateCount++] = 0; + candidateVectors[candidateCount] = referenceMotionVectors.GetNewReference(referenceIndex); + candidateModes[candidateCount] = Av1PredictionMode.NewMotionVector; + candidateReferenceIndices[candidateCount++] = (byte)referenceIndex; + } - int maximumNearIndex = Math.Min(2, Math.Max(0, referenceMotionVectors.Count - 2)); - for (int referenceIndex = 0; referenceIndex <= maximumNearIndex; referenceIndex++) - { - candidateVectors[candidateCount] = referenceMotionVectors.GetNearReference(referenceIndex); - candidateModes[candidateCount] = Av1PredictionMode.NearMotionVector; - candidateReferenceIndices[candidateCount++] = (byte)referenceIndex; - } + int maximumNearIndex = Math.Min(2, Math.Max(0, referenceMotionVectors.Count - 2)); + for (int referenceIndex = 0; referenceIndex <= maximumNearIndex; referenceIndex++) + { + candidateVectors[candidateCount] = referenceMotionVectors.GetNearReference(referenceIndex); + candidateModes[candidateCount] = Av1PredictionMode.NearMotionVector; + candidateReferenceIndices[candidateCount++] = (byte)referenceIndex; } candidateVectors[candidateCount] = globalMotion; candidateModes[candidateCount] = Av1PredictionMode.GlobalMotionVector; candidateReferenceIndices[candidateCount++] = 0; - if (this.effort >= MinimumInterMotionSearchEffort) - { - int maximumNewIndex = Math.Min(2, Math.Max(0, referenceMotionVectors.Count - 1)); - for (int referenceIndex = 0; referenceIndex <= maximumNewIndex; referenceIndex++) - { - Av1MotionVector newReference = referenceMotionVectors.GetNewReference(referenceIndex); - Av1MotionVector searched = this.FindInterMotionVector( - writer, - blockOrigin, - newReference, - referenceIndex); - - // Equal prediction vectors can carry different DRL and mode costs. Preserve each syntax choice - // as an independent candidate instead of deduplicating solely by reconstructed pixels. - candidateVectors[candidateCount] = searched; - candidateModes[candidateCount] = Av1PredictionMode.NewMotionVector; - candidateReferenceIndices[candidateCount++] = (byte)referenceIndex; - } - } - Span selectedLumaReconstruction = workspace.SelectedLumaReconstruction; Span candidateLumaReconstruction = workspace.LumaCandidateReconstruction; Span selectedBlueReconstruction = workspace.SelectedBlueReconstruction; @@ -698,10 +648,148 @@ internal static partial class Av1IntraSuperblockEncoder int horizontalFractionMask = (Av1MotionVector.SubpixelScale << (block.HasChroma && sequenceHeader.ColorConfig.SubSamplingX ? 1 : 0)) - 1; int verticalFractionMask = (Av1MotionVector.SubpixelScale << (block.HasChroma && sequenceHeader.ColorConfig.SubSamplingY ? 1 : 0)) - 1; + Buffer2DRegion sourcePlane = this.source.GetPlane(Av1Plane.Y); + Buffer2DRegion referencePlane = this.reference.GetPlane(Av1Plane.Y); + int sourceOrigin = ((sourcePlane.Bounds.Y + blockOrigin.Y) * sourcePlane.Stride) + sourcePlane.Bounds.X + blockOrigin.X; + int referenceOrigin = ((referencePlane.Bounds.Y + blockOrigin.Y) * referencePlane.Stride) + referencePlane.Bounds.X + blockOrigin.X; + Size frameSize = new( + this.picture.Parent.Common.ModeInfoColumnCount << Av1Constants.ModeInfoSizeLog2, + this.picture.Parent.Common.ModeInfoRowCount << Av1Constants.ModeInfoSizeLog2); + + Rectangle frameBounds = Av1MotionVector.GetFrameSearchBounds( + new Rectangle(blockOrigin, new Size(8)), + frameSize, + Math.Min(referencePlane.Bounds.X, referencePlane.Bounds.Y)); + + Av1NeighborArrayUnit coefficientContexts = this.picture.LuminanceDcSignLevelCoefficientNeighbors[tileIndex]; + ReadOnlySpan aboveContexts = coefficientContexts.Top[coefficientContexts.GetTopIndex(blockOrigin)..]; + ReadOnlySpan leftContexts = coefficientContexts.Left[coefficientContexts.GetLeftIndex(blockOrigin)..]; + + // Frame owners provide contiguous padded planes. Borrow those spans without copying source blocks + // or reconstructing border samples, and keep the search scratch disjoint from retained inter winners. + Av1MotionSearchBase.SingleReferenceSearch motionSearch = new( + sourcePlane.Buffer.DangerousGetSingleSpan()[sourceOrigin..], + sourcePlane.Stride, + referencePlane.Buffer.DangerousGetSingleSpan(), + referencePlane.Stride, + referenceOrigin, + BlockSize, + frameBounds, + this.blockWorkspace, + this.blockWorkspace.GetMotionSearchPrediction(), + this.blockWorkspace.Residual, + workspace.PredictionScratch, + workspace.TransformCoefficients, + writer, + aboveContexts, + leftContexts, + this.bitDepth, + this.quantization.QIndex[0], + this.quantization.DeltaQDc[0], + 0, + frameHeader.CodedLossless, + this.rateMultiplier, + transformPartitionRate, + writer.GetSkipCost(false, skipContext), + writer.GetSkipCost(true, skipContext), + defaultFilter, + defaultFilter, + this.blockWorkspace.GetMotionVectorCosts(frameHeader.MotionVectorPrecision)); + + // The first two reference predictors set the block's spatial range. Clamping at the last + // potentially visible interpolation tap bounds padded reads without changing their prediction. + int spatialMagnitude = 0; + for (int index = 0; index < 2; index++) + { + Av1MotionVector spatial = referenceMotionVectors.GetNewReference(index); + int column = Math.Clamp(spatial.Column, -(blockOrigin.X + 8 + 4) * 8, (frameSize.Width - blockOrigin.X + 4) * 8); + int row = Math.Clamp(spatial.Row, -(blockOrigin.Y + 8 + 4) * 8, (frameSize.Height - blockOrigin.Y + 4) * 8); + spatialMagnitude = Math.Max(spatialMagnitude, Math.Max(Math.Abs(row), Math.Abs(column)) >> 3); + } + + Av1MotionSearchSettings motionSettings = this.picture.Parent.MotionSearchSettings; + Av1MotionSearchBase.SingleReferenceState motionState = default; + Span motionStarts = stackalloc Av1MotionSearchBase.StartingCandidate[1]; + // Rank interpolation families with prediction-error modeling before running a full transform search. // The selected inter reconstruction remains untouched while two existing prediction views alternate. for (int candidateIndex = 0; candidateIndex < candidateCount; candidateIndex++) { + if (candidateModes[candidateIndex] == Av1PredictionMode.NewMotionVector) + { + int referenceIndex = candidateReferenceIndices[candidateIndex]; + Av1MotionVector referenceVector = candidateVectors[candidateIndex]; + int drlRate = 0; + for (int index = 0; index < 2 && referenceMotionVectors.Count > index + 1; index++) + { + bool advance = referenceIndex > index; + int context = Av1SymbolContextHelper.GetDrlContext(referenceMotionVectors.Weights, index); + drlRate += writer.GetDynamicReferenceListCost(advance, context); + if (!advance) + { + break; + } + } + + int searchRange = int.MaxValue; + if (motionSettings.ReduceSearchRange && referenceIndex > 0) + { + int minimumDifference = int.MaxValue; + int bestMatch = 0; + for (int index = 0; index < referenceIndex; index++) + { + Av1MotionVector previousReference = motionState.References[index].ReferenceVector; + int difference = Math.Max( + Math.Abs(referenceVector.Row - previousReference.Row), + Math.Abs(referenceVector.Column - previousReference.Column)); + + if (difference < minimumDifference) + { + minimumDifference = difference; + bestMatch = index; + } + } + + ref Av1MotionSearchBase.ReferenceSearchResult previous = ref motionState.References[bestMatch]; + if (minimumDifference < 16 * 8 && previous.IsValid) + { + int displacement = Math.Max( + Math.Abs(previous.Vector.Row - previous.ReferenceVector.Row), + Math.Abs(previous.Vector.Column - previous.ReferenceVector.Column)); + + searchRange = (minimumDifference + displacement + 4) >> 3; + } + } + + Point startVector = new( + (referenceVector.Column + 3 + (referenceVector.Column >= 0 ? 1 : 0)) >> 3, + (referenceVector.Row + 3 + (referenceVector.Row >= 0 ? 1 : 0)) >> 3); + + motionStarts[0] = new Av1MotionSearchBase.StartingCandidate(startVector, 0); + if (!motionSearch.Search( + motionSettings, + this.picture.Parent.MotionSearchStepParameter, + spatialMagnitude, + frameHeader.ShowFrame, + searchRange, + frameHeader.ForceIntegerMotionVector, + frameHeader.AllowHighPrecisionMotionVector, + fineMeshInterval: false, + referenceIndex, + referenceVector, + drlRate, + motionStarts, + totalWeight: 0, + ref motionState, + out Av1MotionSearchBase.FractionalResult searchResult) || + motionState.References[referenceIndex].Skip) + { + continue; + } + + candidateVectors[candidateIndex] = searchResult.Vector; + } + modeInfo.Block.Mode = candidateModes[candidateIndex]; bool writesFilters = Av1TileWriter.UsesSwitchableInterpolation(frameHeader, modeInfo.Block); Av1InterpolationFilter verticalFilter = defaultFilter; @@ -811,7 +899,7 @@ internal static partial class Av1IntraSuperblockEncoder out Av1EncoderTransformBlockState candidateBlueState, out Av1EncoderTransformBlockState candidateRedState); - // Strict replacement preserves predictor-stack, global, then new-motion order on equal RD cost. + // Strict replacement preserves nearest, new, near, then global mode order on equal RD cost. if (candidateStatistics.Cost >= selectedStatistics.Cost) { continue; @@ -988,9 +1076,8 @@ internal static partial class Av1IntraSuperblockEncoder int visibleHeight = Math.Min(height, ((this.source.Height + subsamplingY) >> subsamplingY) - planeOrigin.Y); long squaredError = 0; - // This view includes coded alignment samples, matching libaom when do_border_pad is false. - // Its conditional border-padding policy is not implemented here; these are not visible-frame bounds. - // Full blocks use one SIMD reduction; only a partial right edge needs row-sized reductions. + // The source view includes samples extended to the coded dimensions. Reduce complete rows together; + // a partial right edge needs separate row reductions to exclude samples beyond the source view. if (visibleWidth == width) { squaredError = Av1ResidualBuilder.SumSquares(residual[..(width * visibleHeight)]); @@ -1097,9 +1184,7 @@ internal static partial class Av1IntraSuperblockEncoder out lumaState, out int lumaRate, out long lumaDistortion, - out bool hasEmptyLuma, - out Av1EncoderTransformBlockState emptyLumaState, - out long emptyLumaDistortion); + out long lumaPredictionDistortion); // Empty luma transforms signal no transform type. Chroma inherits the decoder's inferred DCT // type, not the last searched luma type, so normalize before evaluating either chroma plane. @@ -1124,14 +1209,10 @@ internal static partial class Av1IntraSuperblockEncoder int redRate = 0; long blueDistortion = 0; long redDistortion = 0; - long emptyBlueDistortion = 0; - long emptyRedDistortion = 0; - bool hasEmptyBlue = true; - bool hasEmptyRed = true; + long bluePredictionDistortion = 0; + long redPredictionDistortion = 0; blueState = default; redState = default; - Av1EncoderTransformBlockState emptyBlueState = default; - Av1EncoderTransformBlockState emptyRedState = default; if (hasChroma) { Av1BlockSize chromaBlockSize = BlockSize.GetSubsampled( @@ -1188,9 +1269,7 @@ internal static partial class Av1IntraSuperblockEncoder out blueState, out blueRate, out blueDistortion, - out hasEmptyBlue, - out emptyBlueState, - out emptyBlueDistortion); + out bluePredictionDistortion); this.EvaluateInterPlane( writer, @@ -1217,9 +1296,7 @@ internal static partial class Av1IntraSuperblockEncoder out redState, out redRate, out redDistortion, - out hasEmptyRed, - out emptyRedState, - out emptyRedDistortion); + out redPredictionDistortion); } int predictionRate = commonPredictionRate + @@ -1239,258 +1316,34 @@ internal static partial class Av1IntraSuperblockEncoder long codedDistortion = lumaDistortion + blueDistortion + redDistortion; Av1RateDistortionStatistics selectedStatistics = new(this.rateMultiplier, codedRate, codedDistortion); - skip = false; - if (hasEmptyLuma && hasEmptyBlue && hasEmptyRed) - { - int skipRate = predictionRate + writer.GetSkipCost(true, skipContext); - long skipDistortion = emptyLumaDistortion + emptyBlueDistortion + emptyRedDistortion; - Av1RateDistortionStatistics skipStatistics = new(this.rateMultiplier, skipRate, skipDistortion); - if (skipStatistics.Cost < selectedStatistics.Cost) - { - selectedStatistics = skipStatistics; - skip = true; - workspace.LumaPrediction[..LumaTransformSize.GetSize2d()].CopyTo(lumaReconstruction); - lumaCoefficients[..LumaTransformSize.GetSize2d()].Clear(); - lumaState = emptyLumaState; - if (hasChroma) - { - int chromaSampleCount = chromaTransformSize.GetSize2d(); - workspace.BluePrediction[..chromaSampleCount].CopyTo(blueReconstruction); - workspace.RedPrediction[..chromaSampleCount].CopyTo(redReconstruction); - blueCoefficients[..chromaSampleCount].Clear(); - redCoefficients[..chromaSampleCount].Clear(); - blueState = emptyBlueState; - redState = emptyRedState; - } - } - } - - return selectedStatistics; - } - - /// - /// Searches a bounded full-pixel neighborhood around the spatial reference vector. - /// - /// The live tile entropy model used to measure vector syntax. - /// The current 8x8 luma origin. - /// The differential reference from the spatial candidate stack. - /// The selected dynamic-reference-list entry. - /// The lowest-cost full-pixel vector found by the effort-scaled search. - private Av1MotionVector FindInterMotionVector( - Av1SymbolEncoder writer, - Point blockOrigin, - Av1MotionVector referenceVector, - int referenceMotionVectorIndex) - { - int effortShift = this.effort - MinimumInterMotionSearchEffort; - int searchRadius = Math.Min( - MinimumInterMotionSearchRadius << effortShift, - Av1EncoderFrame.LumaBorder); - - int referenceColumn = referenceVector.Column >> Av1MotionVector.SubpixelBits; - int referenceRow = referenceVector.Row >> Av1MotionVector.SubpixelBits; - int minimumColumn = Math.Max(-Av1EncoderFrame.LumaBorder, referenceColumn - searchRadius); - int maximumColumn = Math.Min(Av1EncoderFrame.LumaBorder, referenceColumn + searchRadius); - int minimumRow = Math.Max(-Av1EncoderFrame.LumaBorder, referenceRow - searchRadius); - int maximumRow = Math.Min(Av1EncoderFrame.LumaBorder, referenceRow + searchRadius); - Point best = new( - Av1Math.Clamp(referenceColumn, minimumColumn, maximumColumn), - Av1Math.Clamp(referenceRow, minimumRow, maximumRow)); + int skipRate = writer.GetSkipCost(true, skipContext); + long skipDistortion = lumaPredictionDistortion + bluePredictionDistortion + redPredictionDistortion; - ref Av1ReferenceMotionVectors referenceMotionVectors = ref this.blockWorkspace.ReferenceMotionVectors; - Av1MotionVector bestVector = new( - best.Y * Av1MotionVector.SubpixelScale, - best.X * Av1MotionVector.SubpixelScale); - - long bestCost = this.GetInterMotionCandidateCost( - writer, - blockOrigin, - bestVector, - Av1PredictionMode.NewMotionVector, - referenceMotionVectorIndex, - in referenceMotionVectors); + // All-empty residuals omit the transform tree. Nonempty residuals can also be discarded when + // prediction alone costs no more; shared prediction syntax must not affect the rounded comparison. + skip = (lumaState.EndOfBlock == 0 && blueState.EndOfBlock == 0 && redState.EndOfBlock == 0) || + Av1RateDistortion.GetCost(this.rateMultiplier, skipRate, skipDistortion) <= + Av1RateDistortion.GetCost(this.rateMultiplier, codedRate - predictionRate, codedDistortion); - for (int step = searchRadius; step > 0; step >>= 1) + if (skip) { - Point stageBest = best; - long stageBestCost = bestCost; - for (int directionIndex = 0; directionIndex < InterMotionSearchDirectionCount; directionIndex++) + selectedStatistics = new(this.rateMultiplier, predictionRate + skipRate, skipDistortion); + workspace.LumaPrediction[..LumaTransformSize.GetSize2d()].CopyTo(lumaReconstruction); + lumaCoefficients[..LumaTransformSize.GetSize2d()].Clear(); + lumaState = default; + if (hasChroma) { - Point direction = GetInterMotionSearchDirection(directionIndex); - Point candidate = new( - best.X + (direction.X * step), - best.Y + (direction.Y * step)); - - if (candidate.X < minimumColumn || candidate.X > maximumColumn || - candidate.Y < minimumRow || candidate.Y > maximumRow) - { - continue; - } - - Av1MotionVector candidateVector = new( - candidate.Y * Av1MotionVector.SubpixelScale, - candidate.X * Av1MotionVector.SubpixelScale); - - long candidateCost = this.GetInterMotionCandidateCost( - writer, - blockOrigin, - candidateVector, - Av1PredictionMode.NewMotionVector, - referenceMotionVectorIndex, - in referenceMotionVectors); - - // Strict replacement preserves the earlier reference-centered search position on ties. - if (candidateCost < stageBestCost) - { - stageBestCost = candidateCost; - stageBest = candidate; - } + int chromaSampleCount = chromaTransformSize.GetSize2d(); + workspace.BluePrediction[..chromaSampleCount].CopyTo(blueReconstruction); + workspace.RedPrediction[..chromaSampleCount].CopyTo(redReconstruction); + blueCoefficients[..chromaSampleCount].Clear(); + redCoefficients[..chromaSampleCount].Clear(); + blueState = default; + redState = default; } - - best = stageBest; - bestCost = stageBestCost; - } - - bestVector = new( - best.Y * Av1MotionVector.SubpixelScale, - best.X * Av1MotionVector.SubpixelScale); - - if (this.effort < MinimumSubpixelMotionSearchEffort) - { - return bestVector; } - int minimumSubpixel = (-Av1EncoderFrame.LumaBorder + FractionalInterpolationBorder) * - Av1MotionVector.SubpixelScale; - - int maximumSubpixel = (Av1EncoderFrame.LumaBorder - FractionalInterpolationBorder) * - Av1MotionVector.SubpixelScale; - - if (bestVector.Column < minimumSubpixel || bestVector.Column > maximumSubpixel || - bestVector.Row < minimumSubpixel || bestVector.Row > maximumSubpixel) - { - return bestVector; - } - - int finalStep = this.effort >= MinimumHighPrecisionMotionSearchEffort ? 1 : 2; - for (int step = Av1MotionVector.SubpixelScale >> 1; step >= finalStep; step >>= 1) - { - Av1MotionVector stageBest = bestVector; - long stageBestCost = bestCost; - for (int directionIndex = 0; directionIndex < InterMotionSearchDirectionCount; directionIndex++) - { - Point direction = GetInterMotionSearchDirection(directionIndex); - Av1MotionVector candidate = new( - bestVector.Row + (direction.Y * step), - bestVector.Column + (direction.X * step)); - - if (candidate.Column < minimumSubpixel || candidate.Column > maximumSubpixel || - candidate.Row < minimumSubpixel || candidate.Row > maximumSubpixel) - { - continue; - } - - long candidateCost = this.GetInterMotionCandidateCost( - writer, - blockOrigin, - candidate, - Av1PredictionMode.NewMotionVector, - referenceMotionVectorIndex, - in referenceMotionVectors); - - // Each precision stage remains centered on its incoming winner; strict replacement keeps - // the integer or coarser fractional vector when an interpolated candidate only ties it. - if (candidateCost < stageBestCost) - { - stageBestCost = candidateCost; - stageBest = candidate; - } - } - - bestVector = stageBest; - bestCost = stageBestCost; - } - - return bestVector; - } - - /// - /// Combines normalized prediction error with the exact mode and vector syntax rate. - /// - /// The live tile entropy model. - /// The current 8x8 luma origin. - /// The candidate motion vector. - /// The candidate single-reference inter mode. - /// The selected dynamic-reference-list entry. - /// The current spatial candidate stack. - /// The rate-distortion search cost. - private long GetInterMotionCandidateCost( - Av1SymbolEncoder writer, - Point blockOrigin, - Av1MotionVector vector, - Av1PredictionMode mode, - int referenceMotionVectorIndex, - in Av1ReferenceMotionVectors referenceMotionVectors) - { - long predictionError; - if (((vector.Row | vector.Column) & (Av1MotionVector.SubpixelScale - 1)) == 0) - { - Point predictionOrigin = new( - blockOrigin.X + (vector.Column >> Av1MotionVector.SubpixelBits), - blockOrigin.Y + (vector.Row >> Av1MotionVector.SubpixelBits)); - - predictionError = TOperator.GetInterPredictionError( - this.source.GetPlane(Av1Plane.Y), - blockOrigin, - this.reference.GetPlane(Av1Plane.Y), - predictionOrigin, - this.bitDepth); - } - else - { - const Av1TransformSize SearchTransformSize = Av1TransformSize.Size8x8; - Av1EncoderInterPredictionWorkspace workspace = - this.blockWorkspace.GetInterPredictionWorkspace(); - - int sourceColumnQ4 = (blockOrigin.X << 4) + (vector.Column << 1); - int sourceRowQ4 = (blockOrigin.Y << 4) + (vector.Row << 1); - Point predictionOrigin = new(sourceColumnQ4 >> 4, sourceRowQ4 >> 4); - ObuFrameHeader frameHeader = this.picture.Parent.FrameHeader; - - // Fractional candidates must pass through the same interpolation and residual kernels used by - // final reconstruction; comparing only their integer origins would choose the wrong phase. - TOperator.PrepareTranslationalInterPrediction( - this.source.GetPlane(Av1Plane.Y), - blockOrigin, - this.reference.GetPlane(Av1Plane.Y), - predictionOrigin, - frameHeader.InterpolationFilter == Av1InterpolationFilter.Switchable ? Av1InterpolationFilter.Regular : frameHeader.InterpolationFilter, - frameHeader.InterpolationFilter == Av1InterpolationFilter.Switchable ? Av1InterpolationFilter.Regular : frameHeader.InterpolationFilter, - sourceColumnQ4 & 15, - sourceRowQ4 & 15, - workspace.LumaPrediction, - workspace.Residual, - workspace.PredictionScratch, - SearchTransformSize, - this.bitDepth); - - predictionError = Av1ResidualBuilder.SumSquares(workspace.Residual); - int normalizationShift = (this.bitDepth.GetBitCount() - 8) * 2; - if (normalizationShift != 0) - { - predictionError = (predictionError + (1L << (normalizationShift - 1))) >> - normalizationShift; - } - } - - int rate = this.GetInterModeRate( - writer, - mode, - vector, - referenceMotionVectorIndex, - in referenceMotionVectors); - - return Av1RateDistortion.GetCost(this.rateMultiplier, rate, predictionError); + return selectedStatistics; } /// @@ -1543,29 +1396,12 @@ internal static partial class Av1IntraSuperblockEncoder } Av1MotionVector reference = referenceMotionVectors.GetNewReference(referenceMotionVectorIndex); - return rate + writer.GetMotionVectorCost( - vector, - reference, - this.picture.Parent.FrameHeader.MotionVectorPrecision); - } + Av1MotionVectorCosts costs = this.blockWorkspace.GetMotionVectorCosts(this.picture.Parent.FrameHeader.MotionVectorPrecision); - /// - /// Gets one cardinal or diagonal search direction in stable reference order. - /// - /// The zero-based direction index. - /// The unit full-pixel direction. - private static Point GetInterMotionSearchDirection(int index) - => index switch - { - 0 => new Point(0, -1), - 1 => new Point(0, 1), - 2 => new Point(-1, 0), - 3 => new Point(1, 0), - 4 => new Point(-1, -1), - 5 => new Point(1, 1), - 6 => new Point(1, -1), - _ => new Point(-1, 1) - }; + // Mode selection discounts motion syntax to 108/128 of its estimated rate. Apply the rounded + // weight to the vector alone; mode and dynamic-reference-list symbols retain their full rate. + return rate + (((costs.GetCost(vector, reference) * 108) + 64) >> 7); + } /// /// Builds one plane prediction and selects its transform without repeating interpolation for each transform type. @@ -1595,9 +1431,7 @@ internal static partial class Av1IntraSuperblockEncoder out Av1EncoderTransformBlockState selectedState, out int selectedRate, out long selectedDistortion, - out bool hasEmptyTransform, - out Av1EncoderTransformBlockState emptyState, - out long emptyDistortion) + out long predictionDistortion) { Point planeOrigin = new(lumaOrigin.X >> subsamplingX, lumaOrigin.Y >> subsamplingY); int sourceColumnQ4 = (planeOrigin.X << 4) + (vector.Column << (1 - subsamplingX)); @@ -1647,8 +1481,15 @@ internal static partial class Av1IntraSuperblockEncoder transformSize); } - // Motion compensation and subtraction do not depend on transform type. Keep them outside the - // transform loop so exhaustive luma search traverses the source and reference blocks only once. + // Prediction-only error remains available even when every transform quantizes to nonzero coefficients. + // Normalize squared sample precision with rounding before adding four fractional distortion bits. + long predictionSquaredError = Av1ResidualBuilder.SumSquares(residual[..sampleCount]); + int normalizationShift = (this.bitDepth.GetBitCount() - 8) * 2; + predictionDistortion = normalizationShift == 0 + ? predictionSquaredError << 4 + : ((predictionSquaredError + (1L << (normalizationShift - 1))) >> normalizationShift) << 4; + + // Motion compensation and subtraction are shared by all transform types for this prediction. Av1TransformSetType transformSetType = Av1SymbolContextHelper.GetExtendedTransformSetType( transformSize, isInter: true, @@ -1665,13 +1506,9 @@ internal static partial class Av1IntraSuperblockEncoder selectedState = default; selectedRate = 0; selectedDistortion = 0; - hasEmptyTransform = false; - emptyState = default; - emptyDistortion = 0; - // The candidate and best spans alternate ownership whenever a transform improves the result. - // This mirrors the reference's buffer-pointer swap and replaces a copy on every improvement - // with at most one normalization copy after the transform search. + // Alternate candidate and best spans on improvement. The winning storage stays intact during + // later trials, with at most one normalization copy into the caller's destination after the search. Span candidateReconstruction = transformReconstruction[..sampleCount]; Span candidateCoefficients = transformCoefficients[..sampleCount]; Span bestReconstruction = selectedReconstruction[..sampleCount]; @@ -1737,14 +1574,6 @@ internal static partial class Av1IntraSuperblockEncoder selectedRate = candidateRate; selectedDistortion = candidateDistortion; } - - if (candidateState.EndOfBlock == 0 && - (!hasEmptyTransform || candidateDistortion < emptyDistortion)) - { - hasEmptyTransform = true; - emptyState = candidateState; - emptyDistortion = candidateDistortion; - } } // Callers retain the designated selected spans after this scratch workspace is reused by the diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1ScreenContentDetector.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1ScreenContentDetector.cs index 52683b8dca..cc163297dc 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1ScreenContentDetector.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1ScreenContentDetector.cs @@ -55,7 +55,8 @@ internal static class Av1ScreenContentDetector /// The converted source frame. /// Receives whether palette syntax should be enabled. /// Receives whether intra-block copy should be enabled. - public static void Detect( + /// Whether the frame is classified as screen content for encoder decisions. + public static bool Detect( Av1EncoderFrame source, out bool allowScreenContentTools, out bool allowIntraBlockCopy) @@ -67,22 +68,23 @@ internal static class Av1ScreenContentDetector /// The converted source frame. /// Receives whether palette syntax should be enabled. /// Receives whether intra-block copy should be enabled. - public static void Detect( + /// Whether the frame is classified as screen content for encoder decisions. + public static bool Detect( Av1EncoderFrame source, out bool allowScreenContentTools, out bool allowIntraBlockCopy) => Detect(source, out allowScreenContentTools, out allowIntraBlockCopy); - private static void Detect( + private static bool Detect( Av1EncoderFrame source, out bool allowScreenContentTools, out bool allowIntraBlockCopy) where TSample : unmanaged where TOperator : struct, ISampleOperator { - Av1EncoderFrame.PlanarView view = source.View; - int width = source.Width; - int height = source.Height; + Av1EncoderFrame.PlanarView view = source.CodedView; + int width = (source.Width + 7) & ~7; + int height = (source.Height + 7) & ~7; long frameArea = (long)width * height; int bitDepthShift = source.LumaBitDepth - 8; int paletteBlockCount = 0; @@ -91,7 +93,8 @@ internal static class Av1ScreenContentDetector allowScreenContentTools = false; allowIntraBlockCopy = false; - // Complete 16x16 blocks and the strict frame-area threshold preserve the reference detector's decision. + // Analyze complete 16x16 blocks in the source's eight-sample-aligned extent. Padding participates in + // both color counting and the area thresholds, while any final partial block is omitted. for (int blockRow = 0; blockRow + DetectionBlockLength <= height; blockRow += DetectionBlockLength) { for (int blockColumn = 0; blockColumn + DetectionBlockLength <= width; blockColumn += DetectionBlockLength) @@ -152,11 +155,14 @@ internal static class Av1ScreenContentDetector if (allowIntraBlockCopy) { - return; + return true; } } } } + + return (long)paletteBlockCount * DetectionBlockArea * 10 > frameArea * 4 && + (long)intraBlockCopyBlockCount * DetectionBlockArea * 30 > frameArea; } private static long RoundPowerOfTwo(long value, int shift) diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TileEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TileEncoder.cs index e2cda37de0..38588b0d47 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TileEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1TileEncoder.cs @@ -4,6 +4,7 @@ using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; using SixLabors.ImageSharp.Formats.Heif.Av1.Motion; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; +using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.LoopFilter; using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; @@ -38,7 +39,8 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter int effort) { this.picture = picture; - this.tileData = Encode( + this.tileData = Encode( writer, source, reconstruction, @@ -74,7 +76,8 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter int effort) { this.picture = picture; - this.tileData = Encode( + this.tileData = Encode( writer, source, reference, @@ -110,7 +113,8 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter int effort) { this.picture = picture; - this.tileData = Encode( + this.tileData = Encode( writer, source, reference, @@ -144,7 +148,8 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter int effort) { this.picture = picture; - this.tileData = Encode( + this.tileData = Encode( writer, source, reconstruction, @@ -180,7 +185,8 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter int effort) { this.picture = picture; - this.tileData = Encode( + this.tileData = Encode( writer, source, reference, @@ -216,7 +222,8 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter int effort) { this.picture = picture; - this.tileData = Encode( + this.tileData = Encode( writer, source, reference, @@ -236,7 +243,7 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter return this.tileData.Span.Slice(offset, length); } - private static ReadOnlyMemory Encode( + private static ReadOnlyMemory Encode( Av1SymbolEncoder writer, Av1EncoderFrame source, Av1EncoderFrame reference, @@ -248,6 +255,66 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter int effort) where TSample : unmanaged where TOperator : struct, Av1IntraSuperblockEncoder.IBlockEncodingOperator + where TVerticalOperator : struct, Av1DeblockingFilter.IEdgeOperator + where THorizontalOperator : struct, Av1DeblockingFilter.IEdgeOperator + { + Av1PictureParentControlSet parent = picture.Parent; + ObuFrameHeader frameHeader = parent.FrameHeader; + Av1MotionSearchSettings motionSettings = new( + parent.EncodingSpeed, + picture.Sequence.SequenceHeader.IsStillPicture, + new Size(source.Width, source.Height), + frameHeader.QuantizationParameters.BaseQIndex, + frameHeader.IsIntra, + parent.IsScreenContent); + + parent.MotionSearchSettings = motionSettings; + int maximumDimension = Math.Max(source.Width, source.Height); + int stepParameter = Av1MotionSearchBase.GetInitialStepParameter(maximumDimension); + if (frameHeader.IsIntra) + { + // A key frame seeds the following inter frame with the complete frame range. + parent.MaximumMotionVectorMagnitude = maximumDimension; + } + else if (motionSettings.AutomaticStepSizeLevel != 0) + { + if (frameHeader.ShowFrame && motionSettings.AutomaticStepSizeLevel >= 2 && parent.MaximumMotionVectorMagnitude != -1) + { + int range = Math.Min(maximumDimension, 2 * parent.MaximumMotionVectorMagnitude); + stepParameter = Av1MotionSearchBase.GetInitialStepParameter(range); + } + + // The packing pass accumulates actual NEWMV magnitudes. Trial candidates and inherited vectors + // do not contribute; a frame with no written NEWMV leaves a zero maximum for the next frame. + parent.MaximumMotionVectorMagnitude = 0; + } + + parent.MotionSearchStepParameter = stepParameter; + _ = ProcessTiles( + writer, source, reference, reconstruction, picture, coefficientBuffer, tileWorkspace, blockWorkspace, effort); + + Av1LoopFilterEncoder.ApplyFrame(picture, reconstruction); + + // Analysis retains the selected modes, coefficients, palette tokens, and motion contexts. Packing starts + // from the same entropy edges and probabilities while the completed frame decisions remain available. + picture.ResetEntropyContexts(); + return ProcessTiles( + writer, source, reference, reconstruction, picture, coefficientBuffer, tileWorkspace, blockWorkspace, effort); + } + + private static ReadOnlyMemory ProcessTiles( + Av1SymbolEncoder writer, + Av1EncoderFrame source, + Av1EncoderFrame reference, + Av1EncoderFrame reconstruction, + Av1PictureControlSet picture, + Av1EncoderCoefficientBuffer coefficientBuffer, + Av1EncoderTileWorkspace tileWorkspace, + Av1EncoderBlockWorkspace blockWorkspace, + int effort) + where TSample : unmanaged + where TOperator : struct, Av1IntraSuperblockEncoder.IBlockEncodingOperator + where TSymbolOperation : struct, Av1SymbolEncoder.ISymbolOperation { ObuFrameHeader frameHeader = picture.Parent.FrameHeader; ObuSequenceHeader sequenceHeader = picture.Sequence.SequenceHeader; @@ -260,7 +327,7 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter ObuTileGroupHeader tileLayout = frameHeader.TilesInfo; Span tileDataOffsets = picture.TileDataOffsets.Span; Span tileDataLengths = picture.TileDataLengths.Span; - if (frameHeader.AllowIntraBlockCopy) + if (!TSymbolOperation.WritesOutput && frameHeader.AllowIntraBlockCopy) { // Hash the visible source once before reconstruction begins so candidate discovery never depends // on coding order and the workspace can be reused as compact bucket links afterward. @@ -276,12 +343,10 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter for (int tileColumn = 0; tileColumn < tileLayout.TileColumnCount; tileColumn++) { tile.SetTileColumn(tileLayout, frameHeader.ModeInfoColumnCount, tileColumn); - if (tileIndex > 0) - { - // Every tile begins from the same frame probabilities, while its bytes follow the preceding - // tile in the retained output allocation. - writer.Reset(tileDataEnd); - } + + // Each pass begins every tile from the same frame probabilities. Only the packing pass + // advances the output offset; the analysis operation does not touch range-coder state. + writer.Reset(tileDataEnd); Point firstModeInfoPosition = new(tile.ModeInfoColumnStart, tile.ModeInfoRowStart); entropyContext.MacroBlockModeInfo = picture.GetMacroBlockModeInfo(firstModeInfoPosition); @@ -300,36 +365,64 @@ internal readonly struct Av1TileEncoder : IAv1TileWriter modeInfoColumn << Av1Constants.ModeInfoSizeLog2, modeInfoRow << Av1Constants.ModeInfoSizeLog2); - Av1IntraSuperblockEncoder.Prepare( - picture, - superblock, - entropyContext.SuperblockOrigin); - - Av1IntraSuperblockEncoder.ModeDecision blockEncoder = new( - source, - reference, - reconstruction, - picture, - superblock, - coefficientBuffer, - blockWorkspace, - effort); - - Av1TileWriter.WriteSuperblock( - picture, - entropyContext, - writer, - superblock, - coefficientBuffer, - (ushort)tileIndex, - ref blockEncoder); + if (TSymbolOperation.WritesOutput) + { + Av1TileWriter.RetainedBlockEncodingHandler blockEncoder = new(picture); + Av1TileWriter.WriteSuperblock( + picture, + entropyContext, + writer, + superblock, + coefficientBuffer, + (ushort)tileIndex, + ref blockEncoder); + } + else + { + if (!frameHeader.IsIntra) + { + // Candidates within a superblock share one entropy snapshot. Updating while + // trying partitions would make the search depend on discarded alternatives. + writer.FillMotionVectorCosts(blockWorkspace.GetMotionVectorCosts(frameHeader.MotionVectorPrecision)); + } + + Av1IntraSuperblockEncoder.Prepare( + picture, + superblock, + entropyContext.SuperblockOrigin); + + Av1IntraSuperblockEncoder.ModeDecision blockEncoder = new( + source, + reference, + reconstruction, + picture, + superblock, + coefficientBuffer, + blockWorkspace, + effort); + + Av1TileWriter.WriteSuperblock< + TSymbolOperation, + Av1IntraSuperblockEncoder.ModeDecision>( + picture, + entropyContext, + writer, + superblock, + coefficientBuffer, + (ushort)tileIndex, + ref blockEncoder); + } } } - _ = writer.Exit(out int tileDataLength); - tileDataOffsets[tileIndex] = tileDataEnd; - tileDataLengths[tileIndex] = tileDataLength; - tileDataEnd += tileDataLength; + if (TSymbolOperation.WritesOutput) + { + _ = writer.Exit(out int tileDataLength); + tileDataOffsets[tileIndex] = tileDataEnd; + tileDataLengths[tileIndex] = tileDataLength; + tileDataEnd += tileDataLength; + } + tileIndex++; } } diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.HorizontalByteEdgeOperator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.HorizontalByteEdgeOperator.cs index 8871d36a45..cce6aeff23 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.HorizontalByteEdgeOperator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.HorizontalByteEdgeOperator.cs @@ -11,7 +11,7 @@ internal static partial class Av1DeblockingFilter /// /// Accesses four columns across a horizontal edge in eight-bit storage. /// - private readonly struct HorizontalByteEdgeOperator : IEdgeOperator + public readonly struct HorizontalByteEdgeOperator : IEdgeOperator { /// [MethodImpl(MethodImplOptions.AggressiveInlining)] diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.HorizontalUInt16EdgeOperator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.HorizontalUInt16EdgeOperator.cs index 115b3bae55..ffa2a36771 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.HorizontalUInt16EdgeOperator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.HorizontalUInt16EdgeOperator.cs @@ -11,7 +11,7 @@ internal static partial class Av1DeblockingFilter /// /// Accesses four columns across a horizontal edge in 16-bit storage. /// - private readonly struct HorizontalUInt16EdgeOperator : IEdgeOperator + public readonly struct HorizontalUInt16EdgeOperator : IEdgeOperator { /// [MethodImpl(MethodImplOptions.AggressiveInlining)] diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.Operator.cs index 18f949b049..aa1c871360 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.Operator.cs @@ -11,7 +11,7 @@ internal static partial class Av1DeblockingFilter /// Defines orientation- and storage-specific access to the four samples running along one edge segment. /// /// The reconstructed sample storage type. - private interface IEdgeOperator + public interface IEdgeOperator where TSample : unmanaged { /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.VerticalByteEdgeOperator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.VerticalByteEdgeOperator.cs index cd491a755e..4690c460d4 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.VerticalByteEdgeOperator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.VerticalByteEdgeOperator.cs @@ -11,7 +11,7 @@ internal static partial class Av1DeblockingFilter /// /// Accesses four rows across a vertical edge in eight-bit storage. /// - private readonly struct VerticalByteEdgeOperator : IEdgeOperator + public readonly struct VerticalByteEdgeOperator : IEdgeOperator { /// [MethodImpl(MethodImplOptions.AggressiveInlining)] diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.VerticalUInt16EdgeOperator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.VerticalUInt16EdgeOperator.cs index 707c4226c5..c002b20a33 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.VerticalUInt16EdgeOperator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.VerticalUInt16EdgeOperator.cs @@ -11,7 +11,7 @@ internal static partial class Av1DeblockingFilter /// /// Accesses four rows across a vertical edge in 16-bit storage. /// - private readonly struct VerticalUInt16EdgeOperator : IEdgeOperator + public readonly struct VerticalUInt16EdgeOperator : IEdgeOperator { /// [MethodImpl(MethodImplOptions.AggressiveInlining)] diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.cs index e9d36757b1..f673541c99 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1DeblockingFilter.cs @@ -115,7 +115,7 @@ internal static partial class Av1DeblockingFilter /// The eight-bit-domain edge-discontinuity threshold. /// The eight-bit-domain high-edge-variance threshold. /// The sample bit depth. - private static void Filter( + public static void Filter( Span samples, int q0Offset, int stride, diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1LoopFilterBase.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1LoopFilterBase.cs new file mode 100644 index 0000000000..0c4b220b47 --- /dev/null +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1LoopFilterBase.cs @@ -0,0 +1,276 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; +using SixLabors.ImageSharp.Formats.Heif.Av1.Prediction; +using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; +using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; + +namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.LoopFilter; + +/// +/// Traverses reconstructed transform boundaries for in-place deblocking. +/// +internal static class Av1LoopFilterBase +{ + /// + /// Supplies the selected block and transform state at a frame position. + /// + /// The owning encoder or decoder state. + /// The retained block-mode type. + public interface IFrameOperator + where TMode : struct + { + /// + /// Resolves the block and filter parameters covering one plane position. + /// + /// The owning frame state. + /// The position in luma 4x4 units, adjusted for chroma ownership. + /// The component plane. + /// Zero for vertical boundaries; one for horizontal boundaries. + /// The horizontal subsampling shift. + /// The vertical subsampling shift. + /// The storage index identifying the owning prediction block. + /// Whether an inter block omits its residual. + /// The transform covering the plane position. + /// The mode state retained for subsequent level derivation. + public static abstract void GetParameters( + TState state, + Point position, + Av1Plane plane, + int pass, + int subX, + int subY, + out int blockIndex, + out bool skippedTransform, + out Av1TransformSize transformSize, + out TMode mode); + + /// + /// Derives the level for a block whose boundary requires filtering. + /// + /// The owning frame state. + /// The resolved block mode, read without another grid lookup or structure copy. + /// The block position in luma 4x4 units. + /// The component plane. + /// Zero for vertical boundaries; one for horizontal boundaries. + /// The adjusted level in the zero-to-63 domain. + public static abstract int GetFilterLevel(TState state, ref TMode mode, Point position, Av1Plane plane, int pass); + } + + /// + /// Filters one plane band in vertical-then-horizontal boundary order. + /// + /// The reconstructed sample storage type. + /// The encoder or decoder state type. + /// The retained block-mode type. + /// The selected block-state accessor. + /// The vertical sample accessor. + /// The horizontal sample accessor. + /// The owning frame state. + /// The decoded or selected frame header. + /// The component plane. + /// The inclusive band origin in luma 4x4 units. + /// The exclusive band limit in luma 4x4 units. + /// The horizontal subsampling shift. + /// The vertical subsampling shift. + /// The plane storage including the required edge neighborhoods. + /// The offset of the top-left coded sample within the storage. + /// The plane stride in samples. + /// The component precision. + public static void FilterBand( + TState state, + ObuFrameHeader header, + Av1Plane plane, + int rowStart, + int rowEnd, + int subX, + int subY, + Span samples, + int origin, + int stride, + int bitDepth) + where TSample : unmanaged + where TMode : struct + where TFrameOperator : struct, IFrameOperator + where TVerticalOperator : struct, Av1DeblockingFilter.IEdgeOperator + where THorizontalOperator : struct, Av1DeblockingFilter.IEdgeOperator + { + int rowStep = 1 << subY; + int columnStep = 1 << subX; + + // Chroma remains in luma-grid coordinates. Each step still covers four samples in its own plane. + // Finish vertical edges across each row before starting the horizontal traversal of this band. + for (int row = rowStart; row < rowEnd; row += rowStep) + { + for (int column = 0; column < header.ModeInfoColumnCount; column += columnStep) + { + FilterEdge( + state, header, plane, 0, row, column, subX, subY, samples, origin, stride, bitDepth); + } + } + + // Visit horizontal edges down each column: adjacent filters can read samples modified by earlier edges. + for (int column = 0; column < header.ModeInfoColumnCount; column += columnStep) + { + for (int row = rowStart; row < rowEnd; row += rowStep) + { + FilterEdge( + state, header, plane, 1, row, column, subX, subY, samples, origin, stride, bitDepth); + } + } + } + + /// + /// Combines the frame, superblock, segment, reference, and mode adjustments for one boundary. + /// + /// The frame's filter and segmentation state. + /// The luma-direction or chroma-component index. + /// The selected frame level for that component. + /// The superblock adjustment for that component. + /// The owning block's segment. + /// The primary prediction reference. + /// The luma prediction mode. + /// The adjusted level in the zero-to-63 domain. + public static int GetFilterLevel( + ObuFrameHeader header, + int filterIndex, + int baseLevel, + int delta, + int segmentId, + Av1ReferenceFrameType referenceFrame, + Av1PredictionMode mode) + { + ObuLoopFilterParameters parameters = header.LoopFilterParameters; + int level = Av1Math.Clip3(0, Av1Constants.MaxLoopFilter, baseLevel + delta); + ObuSegmentationLevelFeature feature = (ObuSegmentationLevelFeature)((int)ObuSegmentationLevelFeature.AlternativeLoopFilterYVertical + filterIndex); + ObuSegmentationParameters segmentation = header.SegmentationParameters; + + if (segmentation.IsFeatureActive(segmentId, feature)) + { + level = Av1Math.Clip3( + 0, + Av1Constants.MaxLoopFilter, + level + segmentation.GetFeatureData(segmentId, (int)feature)); + } + + if (parameters.ReferenceDeltaModeEnabled) + { + int referenceScale = 1 << (level >> 5); + level += parameters.ReferenceDeltas[(int)referenceFrame] * referenceScale; + + if (referenceFrame > Av1ReferenceFrameType.Intra) + { + // Every inter mode except the two global-motion modes belongs to the second mode-delta class. + int modeDeltaIndex = mode is Av1PredictionMode.GlobalMotionVector or + Av1PredictionMode.GlobalGlobalMotionVector ? 0 : 1; + + level += parameters.ModeDeltas[modeDeltaIndex] * referenceScale; + } + + // Reference and mode adjustments use the same scale and may cancel beyond either limit. + // Clipping the intermediate reference sum would discard part of that cancellation. + level = Av1Math.Clip3(0, Av1Constants.MaxLoopFilter, level); + } + + return level; + } + + /// + /// Derives and applies the kernel at one transform boundary. + /// + /// The reconstructed sample storage type. + /// The encoder or decoder state type. + /// The retained block-mode type. + /// The selected block-state accessor. + /// The oriented sample accessor. + /// The owning frame state. + /// The decoded or selected frame header. + /// The component plane. + /// Zero for vertical boundaries; one for horizontal boundaries. + /// The boundary row in luma 4x4 units. + /// The boundary column in luma 4x4 units. + /// The horizontal subsampling shift. + /// The vertical subsampling shift. + /// The plane storage including the required edge neighborhoods. + /// The offset of the top-left coded sample. + /// The plane stride in samples. + /// The component precision. + private static void FilterEdge( + TState state, + ObuFrameHeader header, + Av1Plane plane, + int pass, + int row, + int column, + int subX, + int subY, + Span samples, + int origin, + int stride, + int bitDepth) + where TSample : unmanaged + where TMode : struct + where TFrameOperator : struct, IFrameOperator + where TEdgeOperator : struct, Av1DeblockingFilter.IEdgeOperator + { + int x = column << Av1Constants.ModeInfoSizeLog2; + int y = row << Av1Constants.ModeInfoSizeLog2; + bool verticalBoundary = pass == 0; + if (x >= header.FrameSize.FrameWidth || y >= header.FrameSize.FrameHeight || (verticalBoundary ? x == 0 : y == 0)) + { + return; + } + + // Subsampled chroma belongs to the bottom/right luma unit within its 8-sample footprint. + Point position = new(column | subX, row | subY); + Point previous = new(position.X - (verticalBoundary ? 1 << subX : 0), position.Y - (verticalBoundary ? 0 : 1 << subY)); + TFrameOperator.GetParameters( + state, position, plane, pass, subX, subY, out int blockIndex, out bool skipped, out Av1TransformSize transform, out TMode mode); + + TFrameOperator.GetParameters( + state, + previous, + plane, + pass, + subX, + subY, + out int previousIndex, + out bool previousSkipped, + out Av1TransformSize previousTransform, + out TMode previousMode); + + int planeX = x >> subX; + int planeY = y >> subY; + bool isTransformEdge = verticalBoundary ? planeX % transform.GetWidth() == 0 : planeY % transform.GetHeight() == 0; + + // Residual-free intra predictions still need deblocking. Only skipped inter transforms suppress + // internal boundaries; two different prediction blocks retain their common boundary. + if (!isTransformEdge || (blockIndex == previousIndex && skipped && previousSkipped)) + { + return; + } + + int level = TFrameOperator.GetFilterLevel(state, ref mode, position, plane, pass); + int filterLevel = level != 0 ? level : TFrameOperator.GetFilterLevel(state, ref previousMode, previous, plane, pass); + if (filterLevel == 0) + { + return; + } + + int size = verticalBoundary + ? Math.Min(transform.GetWidth(), previousTransform.GetWidth()) + : Math.Min(transform.GetHeight(), previousTransform.GetHeight()); + + int kernelLength = plane == Av1Plane.Y ? size == 4 ? 4 : size == 8 ? 8 : 14 : size == 4 ? 4 : 6; + int sharpness = header.LoopFilterParameters.SharpnessLevel; + int shift = sharpness > 4 ? 2 : sharpness > 0 ? 1 : 0; + int limit = sharpness > 0 ? Av1Math.Clip3(1, 9 - sharpness, filterLevel >> shift) : Math.Max(1, filterLevel); + int boundaryLimit = (2 * (filterLevel + 2)) + limit; + + // Thresholds stay in the eight-bit domain; the sample operator and kernel handle storage and precision. + // The explicit origin accommodates bordered encoder planes and the decoder's preceding-row view. + Av1DeblockingFilter.Filter( + samples, origin + (planeY * stride) + planeX, stride, kernelLength, limit, boundaryLimit, filterLevel >> 4, bitDepth); + } +} diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1LoopFilterEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1LoopFilterEncoder.cs new file mode 100644 index 0000000000..14737e2671 --- /dev/null +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopFilter/Av1LoopFilterEncoder.cs @@ -0,0 +1,141 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; +using SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; +using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; +using SixLabors.ImageSharp.Memory; + +namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.LoopFilter; + +/// +/// Applies selected deblocking parameters to the encoder reconstruction. +/// +internal static class Av1LoopFilterEncoder +{ + /// + /// Filters the completed reconstruction before it becomes a prediction reference. + /// + /// The reconstructed sample storage type. + /// The vertical sample accessor. + /// The horizontal sample accessor. + /// The completed frame decisions. + /// The writable reconstructed component planes. + public static void ApplyFrame( + Av1PictureControlSet picture, + Av1EncoderFrame reconstruction) + where TSample : unmanaged + where TVerticalOperator : struct, Av1DeblockingFilter.IEdgeOperator + where THorizontalOperator : struct, Av1DeblockingFilter.IEdgeOperator + { + ObuFrameHeader header = picture.Parent.FrameHeader; + ObuLoopFilterParameters parameters = header.LoopFilterParameters; + if (header.CodedLossless || header.AllowIntraBlockCopy || (parameters.FilterLevel[0] == 0 && parameters.FilterLevel[1] == 0)) + { + return; + } + + int planeCount = picture.Sequence.SequenceHeader.ColorConfig.PlaneCount; + for (int planeIndex = 0; planeIndex < planeCount; planeIndex++) + { + ApplyPlane(picture, reconstruction, (Av1Plane)planeIndex); + } + } + + /// + /// Filters one component using the current frame levels. + /// + /// The reconstructed sample storage type. + /// The vertical sample accessor. + /// The horizontal sample accessor. + /// The completed frame decisions. + /// The writable reconstructed component planes. + /// The component to filter. + public static void ApplyPlane( + Av1PictureControlSet picture, + Av1EncoderFrame reconstruction, + Av1Plane plane) + where TSample : unmanaged + where TVerticalOperator : struct, Av1DeblockingFilter.IEdgeOperator + where THorizontalOperator : struct, Av1DeblockingFilter.IEdgeOperator + { + ObuFrameHeader header = picture.Parent.FrameHeader; + ObuLoopFilterParameters parameters = header.LoopFilterParameters; + int level = plane switch + { + Av1Plane.U => parameters.FilterLevelU, + Av1Plane.V => parameters.FilterLevelV, + _ => Math.Max(parameters.FilterLevel[0], parameters.FilterLevel[1]) + }; + + if (level == 0) + { + return; + } + + Buffer2DRegion samples = reconstruction.CodedView.GetPlane(plane); + int origin = (samples.Bounds.Y * samples.Stride) + samples.Bounds.X; + int subX = plane == Av1Plane.Y ? 0 : reconstruction.ChromaSubsamplingX; + int subY = plane == Av1Plane.Y ? 0 : reconstruction.ChromaSubsamplingY; + int rowsPerBand = 1 << (Av1Constants.MaxSuperBlockSizeLog2 - Av1Constants.ModeInfoSizeLog2); + + // The frame owner supplies one contiguous allocation. Retain its full bordered view so kernels may + // access their edge neighborhoods without copying the plane or materializing decoder frame state. + Span storage = samples.Buffer.DangerousGetSingleSpan(); + for (int rowStart = 0; rowStart < header.ModeInfoRowCount; rowStart += rowsPerBand) + { + int rowEnd = Math.Min(rowStart + rowsPerBand, header.ModeInfoRowCount); + Av1LoopFilterBase.FilterBand( + picture, header, plane, rowStart, rowEnd, subX, subY, storage, origin, samples.Stride, reconstruction.LumaBitDepth); + } + } + + /// + /// Reads deblocking parameters from the retained encoder mode grid. + /// + private readonly struct FrameOperator : Av1LoopFilterBase.IFrameOperator + { + /// + public static void GetParameters( + Av1PictureControlSet state, + Point position, + Av1Plane plane, + int pass, + int subX, + int subY, + out int blockIndex, + out bool skippedTransform, + out Av1TransformSize transformSize, + out Av1EncoderBlockModeInfo mode) + { + blockIndex = state.ModeInfoGrid.Span[(position.Y * state.ModeInfoStride) + position.X]; + mode = state.ModeInfoAllocation.Span[blockIndex].Block; + skippedTransform = mode.Skip && mode.ReferenceFrame > Av1ReferenceFrameType.Intra; + + // The selected encoder transform size is uniform within each luma block. Chroma owns its + // maximum plane transform independently of luma splits, while lossless segments always use 4x4. + transformSize = state.Parent.FrameHeader.LosslessArray[mode.SegmentId] + ? Av1TransformSize.Size4x4 + : plane == Av1Plane.Y ? mode.TransformSize : mode.BlockSize.GetMaxUvTransformSize(subX != 0, subY != 0); + } + + /// + public static int GetFilterLevel(Av1PictureControlSet state, ref Av1EncoderBlockModeInfo mode, Point position, Av1Plane plane, int pass) + { + ObuFrameHeader header = state.Parent.FrameHeader; + ObuLoopFilterParameters parameters = header.LoopFilterParameters; + int index = plane == Av1Plane.Y ? pass : (int)plane + 1; + int level = index switch + { + 0 => parameters.FilterLevel[0], + 1 => parameters.FilterLevel[1], + 2 => parameters.FilterLevelU, + _ => parameters.FilterLevelV + }; + + // The encoder does not emit superblock filter deltas. Frame, segment, reference, and mode + // adjustments still follow the same clipping and scaling as the decoder. + return Av1LoopFilterBase.GetFilterLevel(header, index, level, 0, mode.SegmentId, mode.ReferenceFrame, mode.Mode); + } + } +} diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderDisplacementVector.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderDisplacementVector.cs index 2ab0d32f54..52a3f7526c 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderDisplacementVector.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderDisplacementVector.cs @@ -18,3 +18,29 @@ internal struct Av1EncoderDisplacementVector /// public short Column; } + +/// +/// Stores the reference-vector contexts selected before later blocks populate the frame grid. +/// +internal struct Av1EncoderReferenceContext +{ + /// + /// Stores the differential reference vectors for the four usable stack entries. + /// + public InlineArray4 References; + + /// + /// Stores the candidate weights used by dynamic-reference-list syntax. + /// + public InlineArray4 Weights; + + /// + /// Stores the packed inter-mode context. + /// + public ushort ModeContext; + + /// + /// Stores the number of discovered candidates. + /// + public byte Count; +} diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs index cb70eb6094..514207620e 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderPictureBuffer.cs @@ -132,18 +132,40 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable : 0; int paletteStorageEnd = checked(paletteStorageOffset + paletteStorageLength); + int blockPaletteStorageLength = allocateScreenContentState + ? checked(this.modeInfo.Allocation.Length * Unsafe.SizeOf()) + : 0; + + int paletteTokenStorageOffset = checked(paletteStorageEnd + blockPaletteStorageLength); + + // Each of the two palette planes needs at most one packed token per sample. Maximum-superblock + // rounding keeps this picture-owned capacity valid for either supported superblock geometry. + int paletteTokenStorageLength = allocateScreenContentState + ? checked( + Av1Math.AlignPowerOf2(width, Av1Constants.MaxSuperBlockSizeLog2) * + Av1Math.AlignPowerOf2(height, Av1Constants.MaxSuperBlockSizeLog2) * + Math.Min(2, colorConfig.PlaneCount)) + : 0; + + int paletteTokenStorageEnd = checked(paletteTokenStorageOffset + paletteTokenStorageLength); + int blockEncodingStorageLength = checked(this.modeInfo.Allocation.Length * Av1EncoderBlockStruct.StorageSize); + int blockEncodingStorageEnd = checked(paletteTokenStorageEnd + blockEncodingStorageLength); int displacementVectorLength = allocateMotionVectorState ? this.modeInfo.Allocation.Length : 0; int displacementVectorStorageOffset = allocateMotionVectorState - ? Av1Math.AlignPowerOf2(paletteStorageEnd, 1) - : paletteStorageEnd; + ? Av1Math.AlignPowerOf2(blockEncodingStorageEnd, 1) + : blockEncodingStorageEnd; int displacementVectorStorageLength = checked( displacementVectorLength * Unsafe.SizeOf()); int displacementVectorStorageEnd = checked(displacementVectorStorageOffset + displacementVectorStorageLength); + int referenceContextStorageLength = checked( + displacementVectorLength * Unsafe.SizeOf()); + + int referenceContextStorageEnd = checked(displacementVectorStorageEnd + referenceContextStorageLength); int intraBlockCopySearchStorageOffset = allocateIntraBlockCopySearch - ? Av1Math.AlignPowerOf2(displacementVectorStorageEnd, 2) - : displacementVectorStorageEnd; + ? Av1Math.AlignPowerOf2(referenceContextStorageEnd, 2) + : referenceContextStorageEnd; int intraBlockCopySearchStorageLength = allocateIntraBlockCopySearch ? Av1IntraBlockCopySearchIndex.GetStorageLength(width, height) @@ -178,6 +200,7 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable this.redCoefficientContexts = new Av1NeighborArrayUnit[tileCount]; this.transformContexts = new Av1NeighborArrayUnit[tileCount]; Memory paletteStorage = Memory.Empty; + Memory blockPalettes = Memory.Empty; if (allocateScreenContentState) { // Palette entries contain 16-bit colors, so their packed typed region begins at an even byte offset. @@ -185,6 +208,10 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable stateStorage.Slice(paletteStorageOffset, paletteStorageLength)); paletteStorage = paletteMemory.Memory; + ByteMemoryManager blockPaletteMemory = new( + stateStorage.Slice(paletteStorageEnd, blockPaletteStorageLength)); + + blockPalettes = blockPaletteMemory.Memory; this.paletteContexts = new Av1NeighborArrayUnit[tileCount]; } else @@ -193,6 +220,13 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable } Memory displacementVectors = Memory.Empty; + + // Final syntax parameters use their own eight-byte entries so frequent neighbor lookups retain + // the compact mode-info layout. This typed view borrows the same picture-state owner. + ByteMemoryManager blockEncodingMemory = new( + stateStorage.Slice(paletteTokenStorageEnd, blockEncodingStorageLength)); + + Memory referenceContexts = Memory.Empty; if (allocateMotionVectorState) { // Each component lies strictly inside plus or minus 16384. Two signed 16-bit fields preserve both @@ -201,6 +235,10 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable stateStorage.Slice(displacementVectorStorageOffset, displacementVectorStorageLength)); displacementVectors = displacementVectorMemory.Memory; + ByteMemoryManager referenceContextMemory = new( + stateStorage.Slice(displacementVectorStorageEnd, referenceContextStorageLength)); + + referenceContexts = referenceContextMemory.Memory; } Av1IntraBlockCopySearchIndex intraBlockCopySearch = default; @@ -314,6 +352,10 @@ internal sealed class Av1EncoderPictureBuffer : IDisposable ModeInfoGrid = this.modeInfo.Grid, ModeInfoAllocation = this.modeInfo.Allocation, DisplacementVectors = displacementVectors, + ReferenceContexts = referenceContexts, + BlockEncodings = blockEncodingMemory.Memory, + BlockPalettes = blockPalettes, + PaletteTokens = stateStorage.Slice(paletteTokenStorageOffset, paletteTokenStorageLength), IntraBlockCopySearch = intraBlockCopySearch, ModeInfoStride = this.modeInfo.ModeInfoStride, Disallow4x4AllFrames = this.modeInfo.Disallow4x4AllFrames, diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderTransformBlockState.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderTransformBlockState.cs index bab0da76be..6b88620c0d 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderTransformBlockState.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EncoderTransformBlockState.cs @@ -22,6 +22,11 @@ internal struct Av1EncoderTransformBlockState /// private Av1TransformType transformType; + /// + /// Stores the skip context in bits 0 through 3 and DC-sign context in bits 4 and 5. + /// + public byte EntropyContext; + /// /// Gets or sets the position after the final nonzero coefficient. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EntropyCodingContext.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EntropyCodingContext.cs index d61947a7c5..5dcb300010 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EntropyCodingContext.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1EntropyCodingContext.cs @@ -37,5 +37,10 @@ internal partial class Av1TileWriter /// Gets or sets the number of chroma coefficient positions consumed in the current superblock. /// public int CodedAreaSuperblockUv { get; set; } + + /// + /// Gets or sets the next palette token position in the picture token region. + /// + public int PaletteTokenOffset { get; set; } } } diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureControlSet.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureControlSet.cs index aa96f70b98..a36bdc1da5 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureControlSet.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureControlSet.cs @@ -70,6 +70,26 @@ internal class Av1PictureControlSet /// public Memory DisplacementVectors { get; set; } + /// + /// Gets or sets the reference contexts retained at each allocated block origin. + /// + public Memory ReferenceContexts { get; set; } + + /// + /// Gets or sets the prediction parameters retained at each allocated block origin. + /// + public Memory BlockEncodings { get; set; } + + /// + /// Gets or sets the palette colors retained at each allocated block origin. + /// + public Memory BlockPalettes { get; set; } + + /// + /// Gets or sets the packed color-map tokens retained for final syntax packing. + /// + public Memory PaletteTokens { get; set; } + /// /// Gets or sets the non-owning visible-frame hash index used by intra-block-copy motion search. /// @@ -101,6 +121,38 @@ internal class Av1PictureControlSet /// public required Memory TileDataLengths { get; set; } + /// + /// Restores initial tile entropy edges while preserving selected block decisions and reconstruction. + /// + public void ResetEntropyContexts() + { + this.SegmentationNeighborMap.Span.Clear(); + for (int tileIndex = 0; tileIndex < this.PartitionContexts.Length; tileIndex++) + { + this.PartitionContexts[tileIndex].Left.Clear(); + this.PartitionContexts[tileIndex].Top.Clear(); + this.LuminanceDcSignLevelCoefficientNeighbors[tileIndex].Left.Clear(); + this.LuminanceDcSignLevelCoefficientNeighbors[tileIndex].Top.Clear(); + this.CbDcSignLevelCoefficientNeighbors[tileIndex].Left.Clear(); + this.CbDcSignLevelCoefficientNeighbors[tileIndex].Top.Clear(); + this.CrDcSignLevelCoefficientNeighbors[tileIndex].Left.Clear(); + this.CrDcSignLevelCoefficientNeighbors[tileIndex].Top.Clear(); + + // Transform contexts use the maximum-size sentinel until a preceding block supplies a size. + this.TransformFunctionContexts[tileIndex].Left.Fill((byte)Av1Constants.MaxTransformSize); + this.TransformFunctionContexts[tileIndex].Top.Fill((byte)Av1Constants.MaxTransformSize); + } + + foreach (Av1NeighborArrayUnit context in this.PaletteContexts) + { + context.Left.Clear(); + context.Top.Clear(); + } + + this.CdefPreset.Span.Fill(-1); + this.Parent.PreviousQIndex.Span.Fill(this.Parent.FrameHeader.QuantizationParameters.BaseQIndex); + } + /// /// Gets the mode-information entry mapped to a frame position. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureParentControlSet.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureParentControlSet.cs index 108102fb2d..c16d8b0c6c 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureParentControlSet.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1PictureParentControlSet.cs @@ -1,6 +1,7 @@ // Copyright (c) Six Labors. // Licensed under the Six Labors Split License. +using SixLabors.ImageSharp.Formats.Heif.Av1.Motion; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; namespace SixLabors.ImageSharp.Formats.Heif.Av1.Tiling; @@ -29,4 +30,29 @@ internal class Av1PictureParentControlSet /// Gets or sets the encoder palette-search level. /// public int PaletteLevel { get; set; } + + /// + /// Gets or sets the native-valued encoding speed. + /// + public HeifEncodingSpeed EncodingSpeed { get; set; } + + /// + /// Gets or sets a value indicating whether source analysis classifies this frame as screen content. + /// + public bool IsScreenContent { get; set; } + + /// + /// Gets or sets the resolved motion-search policy for the current frame. + /// + public Av1MotionSearchSettings MotionSearchSettings { get; set; } + + /// + /// Gets or sets the initial full-pixel search step derived before the frame's first block. + /// + public int MotionSearchStepParameter { get; set; } + + /// + /// Gets or sets the largest whole-sample magnitude written by a new-motion mode in the preceding frame. + /// + public int MaximumMotionVectorMagnitude { get; set; } = -1; } diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.BlockEncoding.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.BlockEncoding.cs index fd44f70736..b526d8dc59 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.BlockEncoding.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.BlockEncoding.cs @@ -15,6 +15,11 @@ internal partial class Av1TileWriter /// internal interface IBlockEncodingHandler { + /// + /// Gets a value indicating whether decisions come from completed frame analysis. + /// + static abstract bool UsesRetainedDecisions { get; } + /// /// Selects the partition used for the current tree node. /// @@ -53,8 +58,104 @@ internal partial class Av1TileWriter ref Av1EncoderPaletteInfo paletteInfo); } + internal readonly struct RetainedBlockEncodingHandler : IBlockEncodingHandler + { + private readonly Av1PictureControlSet picture; + + public RetainedBlockEncodingHandler(Av1PictureControlSet picture) + => this.picture = picture; + + public static bool UsesRetainedDecisions => true; + + public Av1PartitionType SelectPartition( + Av1SymbolEncoder writer, + Av1MacroBlockD macroBlock, + Point blockOrigin, + ushort tileIndex, + Av1BlockSize blockSize, + Av1PartitionType preparedPartition) + { + Point position = new(blockOrigin.X >> Av1Constants.ModeInfoSizeLog2, blockOrigin.Y >> Av1Constants.ModeInfoSizeLog2); + Av1BlockSize selectedSize = this.picture.GetFromModeInfoGrid(position).Block.BlockSize; + if (selectedSize == blockSize) + { + return Av1PartitionType.None; + } + + int width = blockSize.Get4x4WideCount(); + int height = blockSize.Get4x4HighCount(); + int selectedWidth = selectedSize.Get4x4WideCount(); + int selectedHeight = selectedSize.Get4x4HighCount(); + if (blockSize > Av1BlockSize.Block8x8 && + position.Y + (height / 2) < this.picture.Parent.Common.ModeInfoRowCount && + position.X + (width / 2) < this.picture.Parent.Common.ModeInfoColumnCount) + { + // A half-sized top-left block alone cannot distinguish an asymmetric partition from a split. + // The mapped blocks at the two half boundaries identify which half remains unsplit. + Av1BlockSize below = this.picture.GetFromModeInfoGrid(position + new Size(0, height / 2)).Block.BlockSize; + Av1BlockSize right = this.picture.GetFromModeInfoGrid(position + new Size(width / 2, 0)).Block.BlockSize; + if (selectedWidth == width) + { + return selectedHeight * 4 == height + ? Av1PartitionType.Horizontal4 + : below == selectedSize ? Av1PartitionType.Horizontal : Av1PartitionType.HorizontalB; + } + + if (selectedHeight == height) + { + return selectedWidth * 4 == width + ? Av1PartitionType.Vertical4 + : right == selectedSize ? Av1PartitionType.Vertical : Av1PartitionType.VerticalB; + } + + if (selectedWidth * 2 == width && selectedHeight * 2 == height) + { + if (below.Get4x4WideCount() == width) + { + return Av1PartitionType.HorizontalA; + } + + if (right.Get4x4HighCount() == height) + { + return Av1PartitionType.VerticalA; + } + } + + return Av1PartitionType.Split; + } + + // At a frame edge only the basic partitions are available. Each smaller dimension contributes + // one split axis; recursive descent then reaches the retained leaf geometry. + return selectedWidth == width + ? Av1PartitionType.Horizontal + : selectedHeight == height ? Av1PartitionType.Vertical : Av1PartitionType.Split; + } + + public void EncodeBlock( + Av1SymbolEncoder writer, + Av1MacroBlockD macroBlock, + Point blockOrigin, + ushort tileIndex, + ref Av1MacroBlockModeInfo modeInfo, + ref Av1EncoderBlockStruct block, + ref Av1EncoderPaletteInfo paletteInfo) + { + int row = blockOrigin.Y >> Av1Constants.ModeInfoSizeLog2; + int column = blockOrigin.X >> Av1Constants.ModeInfoSizeLog2; + int allocationOffset = this.picture.ModeInfoGrid.Span[(row * this.picture.ModeInfoStride) + column]; + block = this.picture.BlockEncodings.Span[allocationOffset]; + if (this.picture.Parent.FrameHeader.AllowScreenContentTools) + { + paletteInfo = this.picture.BlockPalettes.Span[allocationOffset]; + } + } + } + private readonly struct PrecomputedBlockEncodingHandler : IBlockEncodingHandler { + /// + public static bool UsesRetainedDecisions => false; + /// public Av1PartitionType SelectPartition( Av1SymbolEncoder writer, diff --git a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs index 235d0f27d9..e1cebda557 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Tiling/Av1TileWriter.cs @@ -102,16 +102,42 @@ internal partial class Av1TileWriter ushort tileIndex, ref TBlockEncoder blockEncoder) where TBlockEncoder : struct, IBlockEncodingHandler + => WriteSuperblock( + pcs, + ec_ctx, + writer, + superblock, + coefficientBuffer, + tileIndex, + ref blockEncoder); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + public static void WriteSuperblock( + Av1PictureControlSet pcs, + Av1EntropyCodingContext ec_ctx, + Av1SymbolEncoder writer, + Av1Superblock superblock, + Av1EncoderCoefficientBuffer coefficientBuffer, + ushort tileIndex, + ref TBlockEncoder blockEncoder) + where TBlockEncoder : struct, IBlockEncodingHandler + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { ec_ctx.CodedAreaSuperblock = 0; ec_ctx.CodedAreaSuperblockUv = 0; + ec_ctx.PaletteTokenOffset = superblock.Index * + (1 << (2 * pcs.Sequence.SequenceHeader.SuperblockSizeLog2)) * + Math.Min(2, pcs.Sequence.SequenceHeader.ColorConfig.PlaneCount); + ec_ctx.MacroBlock.Tile = superblock.TileInfo; int partitionIndex = 0; int finalBlockIndex = 0; - // Current libaom writes the selected partition tree recursively from the superblock origin. Keeping the - // decisions in preorder removes the global geometry catalog and keeps traversal state on this stack. - WritePartitionTree( + // Partition decisions are stored in preorder, so recursive traversal keeps the current geometry + // on the stack and visits each selected child after its parent. + WritePartitionTree( pcs, ec_ctx, writer, @@ -128,7 +154,7 @@ internal partial class Av1TileWriter /// /// Writes one selected partition node and recursively visits its split children. /// - private static void WritePartitionTree( + private static void WritePartitionTree( Av1PictureControlSet pcs, Av1EntropyCodingContext entropyCodingContext, Av1SymbolEncoder writer, @@ -141,6 +167,7 @@ internal partial class Av1TileWriter ref int finalBlockIndex, ref TBlockEncoder blockEncoder) where TBlockEncoder : struct, IBlockEncodingHandler + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { Av1EncoderCommon common = pcs.Parent.Common; int modeInfoRow = blockOrigin.Y >> Av1Constants.ModeInfoSizeLog2; @@ -167,7 +194,7 @@ internal partial class Av1TileWriter int halfBlockSize = blockSize.GetWidth() >> 1; int quarterBlockSize = blockSize.GetWidth() >> 2; - EncodePartition( + EncodePartition( pcs, writer, blockSize, @@ -178,7 +205,7 @@ internal partial class Av1TileWriter switch (partition) { case Av1PartitionType.None: - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -191,7 +218,7 @@ internal partial class Av1TileWriter break; case Av1PartitionType.Horizontal: - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -204,7 +231,7 @@ internal partial class Av1TileWriter if (modeInfoRow + (blockSize.Get4x4HighCount() >> 1) < common.ModeInfoRowCount) { - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -218,7 +245,7 @@ internal partial class Av1TileWriter break; case Av1PartitionType.Vertical: - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -231,7 +258,7 @@ internal partial class Av1TileWriter if (modeInfoColumn + (blockSize.Get4x4WideCount() >> 1) < common.ModeInfoColumnCount) { - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -262,7 +289,7 @@ internal partial class Av1TileWriter continue; } - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -276,7 +303,7 @@ internal partial class Av1TileWriter } else { - WritePartitionTree( + WritePartitionTree( pcs, entropyCodingContext, writer, @@ -289,7 +316,7 @@ internal partial class Av1TileWriter ref finalBlockIndex, ref blockEncoder); - WritePartitionTree( + WritePartitionTree( pcs, entropyCodingContext, writer, @@ -302,7 +329,7 @@ internal partial class Av1TileWriter ref finalBlockIndex, ref blockEncoder); - WritePartitionTree( + WritePartitionTree( pcs, entropyCodingContext, writer, @@ -315,7 +342,7 @@ internal partial class Av1TileWriter ref finalBlockIndex, ref blockEncoder); - WritePartitionTree( + WritePartitionTree( pcs, entropyCodingContext, writer, @@ -331,7 +358,7 @@ internal partial class Av1TileWriter break; case Av1PartitionType.HorizontalA: - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -341,7 +368,7 @@ internal partial class Av1TileWriter blockOrigin, ref finalBlockIndex, ref blockEncoder); - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -351,7 +378,7 @@ internal partial class Av1TileWriter blockOrigin + new Size(halfBlockSize, 0), ref finalBlockIndex, ref blockEncoder); - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -364,7 +391,7 @@ internal partial class Av1TileWriter break; case Av1PartitionType.HorizontalB: - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -374,7 +401,7 @@ internal partial class Av1TileWriter blockOrigin, ref finalBlockIndex, ref blockEncoder); - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -384,7 +411,7 @@ internal partial class Av1TileWriter blockOrigin + new Size(0, halfBlockSize), ref finalBlockIndex, ref blockEncoder); - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -397,7 +424,7 @@ internal partial class Av1TileWriter break; case Av1PartitionType.VerticalA: - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -407,7 +434,7 @@ internal partial class Av1TileWriter blockOrigin, ref finalBlockIndex, ref blockEncoder); - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -417,7 +444,7 @@ internal partial class Av1TileWriter blockOrigin + new Size(0, halfBlockSize), ref finalBlockIndex, ref blockEncoder); - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -430,7 +457,7 @@ internal partial class Av1TileWriter break; case Av1PartitionType.VerticalB: - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -440,7 +467,7 @@ internal partial class Av1TileWriter blockOrigin, ref finalBlockIndex, ref blockEncoder); - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -450,7 +477,7 @@ internal partial class Av1TileWriter blockOrigin + new Size(halfBlockSize, 0), ref finalBlockIndex, ref blockEncoder); - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -472,7 +499,7 @@ internal partial class Av1TileWriter break; } - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -495,7 +522,7 @@ internal partial class Av1TileWriter break; } - WriteFinalBlock( + WriteFinalBlock( pcs, entropyCodingContext, writer, @@ -521,7 +548,7 @@ internal partial class Av1TileWriter /// /// Writes the next final block selected by partition traversal. /// - private static void WriteFinalBlock( + private static void WriteFinalBlock( Av1PictureControlSet pcs, Av1EntropyCodingContext entropyCodingContext, Av1SymbolEncoder writer, @@ -532,9 +559,10 @@ internal partial class Av1TileWriter ref int finalBlockIndex, ref TBlockEncoder blockEncoder) where TBlockEncoder : struct, IBlockEncodingHandler + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { ref Av1EncoderBlockStruct block = ref superblock.FinalBlocks[finalBlockIndex++]; - WriteModesBlock( + WriteModesBlock( pcs, entropyCodingContext, writer, @@ -697,6 +725,25 @@ internal partial class Av1TileWriter Av1PartitionType partitionType, Point blockOrigin, Av1NeighborArrayUnit partition_context_na) + => EncodePartition( + pcs, + writer, + blockSize, + partitionType, + blockOrigin, + partition_context_na); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + public static void EncodePartition( + Av1PictureControlSet pcs, + Av1SymbolEncoder writer, + Av1BlockSize blockSize, + Av1PartitionType partitionType, + Point blockOrigin, + Av1NeighborArrayUnit partition_context_na) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { bool is_partition_point = blockSize >= Av1BlockSize.Block8x8; @@ -721,15 +768,15 @@ internal partial class Av1TileWriter if (has_rows && has_cols) { - writer.WritePartitionType(partitionType, context_index); + writer.WritePartitionType(partitionType, context_index); } else if (!has_rows && has_cols) { - writer.WriteSplitOrHorizontal(partitionType, blockSize, context_index); + writer.WriteSplitOrHorizontal(partitionType, blockSize, context_index); } else { - writer.WriteSplitOrVertical(partitionType, blockSize, context_index); + writer.WriteSplitOrVertical(partitionType, blockSize, context_index); } return; @@ -780,7 +827,7 @@ internal partial class Av1TileWriter /// The absolute luma-sample origin of the block. /// The transformed coefficients retained by raster-ordered superblock. /// The final-block decision producer. - private static void WriteModesBlock( + private static void WriteModesBlock( Av1PictureControlSet pcs, Av1EntropyCodingContext entropyCodingContext, Av1SymbolEncoder writer, @@ -791,6 +838,7 @@ internal partial class Av1TileWriter Av1EncoderCoefficientBuffer coefficientBuffer, ref TBlockEncoder blockEncoder) where TBlockEncoder : struct, IBlockEncodingHandler + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { Av1SequenceControlSet scs = pcs.Sequence; ObuFrameHeader frm_hdr = pcs.Parent.FrameHeader; @@ -831,6 +879,16 @@ internal partial class Av1TileWriter ref blk_ptr, ref paletteInfo); + int allocationOffset = pcs.ModeInfoGrid.Span[(mi_row * mi_stride) + mi_col]; + if (!TOperation.WritesOutput) + { + pcs.BlockEncodings.Span[allocationOffset] = blk_ptr; + if (frm_hdr.AllowScreenContentTools) + { + pcs.BlockPalettes.Span[allocationOffset] = paletteInfo; + } + } + bool skipWritingCoefficients = macroBlockModeInfo.Block.Skip; // Segmentation, skip, filter, and quantizer syntax precede the prediction-domain branch in both @@ -838,7 +896,7 @@ internal partial class Av1TileWriter { if (pcs.Parent.FrameHeader.SegmentationParameters.Enabled && pcs.Parent.FrameHeader.SegmentationParameters.SegmentIdPrecedesSkip) { - WriteSegmentId( + WriteSegmentId( pcs, writer, blockSize, @@ -849,11 +907,11 @@ internal partial class Av1TileWriter beforeSkip: true); } - EncodeSkipCoefficients(writer, macroBlock, skipWritingCoefficients); + EncodeSkipCoefficients(writer, macroBlock, skipWritingCoefficients); if (pcs.Parent.FrameHeader.SegmentationParameters.Enabled && !pcs.Parent.FrameHeader.SegmentationParameters.SegmentIdPrecedesSkip) { - WriteSegmentId( + WriteSegmentId( pcs, writer, blockSize, @@ -864,7 +922,7 @@ internal partial class Av1TileWriter beforeSkip: false); } - WriteCdef( + WriteCdef( scs, pcs, writer, @@ -883,7 +941,7 @@ internal partial class Av1TileWriter int reduced_delta_qindex = (current_q_index - pcs.Parent.PreviousQIndex.Span[tile_idx]) / frm_hdr.DeltaQParameters.Resolution; - writer.WriteDeltaQuantizerIndex(reduced_delta_qindex); + writer.WriteDeltaQuantizerIndex(reduced_delta_qindex); pcs.Parent.PreviousQIndex.Span[tile_idx] = current_q_index; } } @@ -906,7 +964,7 @@ internal partial class Av1TileWriter if (!isReferenceForced && !isGlobalMotionForced) { int intraInterContext = GetIntraInterContext(macroBlock); - writer.WriteIsInter(isInterBlock, intraInterContext); + writer.WriteIsInter(isInterBlock, intraInterContext); } } @@ -918,35 +976,68 @@ internal partial class Av1TileWriter { Span referenceCounts = stackalloc byte[Av1Constants.ReferenceFrameCount]; CollectNeighborReferenceCounts(macroBlock, referenceCounts); - writer.WriteSingleReference( + writer.WriteSingleReference( macroBlockModeInfo.Block.ReferenceFrame, referenceCounts); } if (!isGlobalMotionForced) { - ref Av1ReferenceMotionVectors referenceMotionVectors = ref tb_ptr.Workspace.ReferenceMotionVectors; - referenceMotionVectors.Build( - pcs, - macroBlock, - modeInfoPosition, - blockSize, - macroBlockModeInfo.Block.PartitionType, - scs.SequenceHeader, - frm_hdr, - macroBlockModeInfo.Block.ReferenceFrame); + Av1EncoderReferenceContext referenceContext; + if (TBlockEncoder.UsesRetainedDecisions) + { + referenceContext = pcs.ReferenceContexts.Span[allocationOffset]; + } + else + { + ref Av1ReferenceMotionVectors referenceMotionVectors = ref tb_ptr.Workspace.ReferenceMotionVectors; + referenceMotionVectors.Build( + pcs, + macroBlock, + modeInfoPosition, + blockSize, + macroBlockModeInfo.Block.PartitionType, + scs.SequenceHeader, + frm_hdr, + macroBlockModeInfo.Block.ReferenceFrame); + + referenceContext = default; + referenceContext.Count = (byte)referenceMotionVectors.Count; + referenceContext.ModeContext = (ushort)referenceMotionVectors.ModeContext; + int candidateCount = Math.Min(4, referenceMotionVectors.Count); + referenceMotionVectors.Weights[..candidateCount].CopyTo(referenceContext.Weights); + + // A stack with no candidates still has a differential fallback vector. Candidate + // weights exist only for discovered entries, while reference zero always remains usable. + int referenceCount = Math.Max(1, candidateCount); + for (int index = 0; index < referenceCount; index++) + { + Av1MotionVector candidate = referenceMotionVectors.GetNewReference(index); + referenceContext.References[index] = new Av1EncoderDisplacementVector + { + Row = (short)candidate.Row, + Column = (short)candidate.Column + }; + } + + // Save the contexts before later blocks become visible through the completed frame grid. + if (!TOperation.WritesOutput) + { + pcs.ReferenceContexts.Span[allocationOffset] = referenceContext; + } + } - writer.WriteInterMode(lumaMode, referenceMotionVectors.ModeContext); + writer.WriteInterMode(lumaMode, referenceContext.ModeContext); int referenceMotionVectorIndex = blk_ptr.ReferenceMotionVectorIndex; if (lumaMode == Av1PredictionMode.NearMotionVector) { // NEARMV reserves stack entry zero for NEARESTMV, so its DRL decisions advance from // near entry zero to one and then from one to two. - for (int index = 1; index < 3 && referenceMotionVectors.Count > index + 1; index++) + for (int index = 1; index < 3 && referenceContext.Count > index + 1; index++) { bool advance = referenceMotionVectorIndex >= index; - int context = Av1SymbolContextHelper.GetDrlContext(referenceMotionVectors.Weights, index); - writer.WriteDynamicReferenceList(advance, context); + int context = Av1SymbolContextHelper.GetDrlContext(referenceContext.Weights, index); + writer.WriteDynamicReferenceList(advance, context); if (!advance) { break; @@ -956,11 +1047,11 @@ internal partial class Av1TileWriter else if (lumaMode == Av1PredictionMode.NewMotionVector) { // NEWMV begins at stack entry zero and can advance through entries one and two. - for (int index = 0; index < 2 && referenceMotionVectors.Count > index + 1; index++) + for (int index = 0; index < 2 && referenceContext.Count > index + 1; index++) { bool advance = referenceMotionVectorIndex > index; - int context = Av1SymbolContextHelper.GetDrlContext(referenceMotionVectors.Weights, index); - writer.WriteDynamicReferenceList(advance, context); + int context = Av1SymbolContextHelper.GetDrlContext(referenceContext.Weights, index); + writer.WriteDynamicReferenceList(advance, context); if (!advance) { break; @@ -968,10 +1059,20 @@ internal partial class Av1TileWriter } Av1MotionVector vector = pcs.GetDisplacementVector(modeInfoPosition); - writer.WriteMotionVector( + writer.WriteMotionVector( vector, - referenceMotionVectors.GetNewReference(referenceMotionVectorIndex), + new Av1MotionVector( + referenceContext.References[referenceMotionVectorIndex].Row, + referenceContext.References[referenceMotionVectorIndex].Column), frm_hdr.MotionVectorPrecision); + + if (TOperation.WritesOutput && pcs.Parent.MotionSearchSettings.AutomaticStepSizeLevel != 0) + { + // Retain the absolute displacement, not the coded difference from the reference. + // Only packed NEWMV syntax contributes to the following frame's search range. + int magnitude = Math.Max(Math.Abs(vector.Row), Math.Abs(vector.Column)) >> Av1MotionVector.SubpixelBits; + pcs.Parent.MaximumMotionVectorMagnitude = Math.Max(pcs.Parent.MaximumMotionVectorMagnitude, magnitude); + } } } @@ -983,7 +1084,7 @@ internal partial class Av1TileWriter macroBlock, direction: 0); - writer.WriteSwitchableInterpolationFilter(macroBlockModeInfo.Block.VerticalInterpolationFilter, verticalContext); + writer.WriteSwitchableInterpolationFilter(macroBlockModeInfo.Block.VerticalInterpolationFilter, verticalContext); if (scs.SequenceHeader.EnableDualFilter) { int horizontalContext = Av1SymbolContextHelper.GetSwitchableInterpolationContext( @@ -991,23 +1092,24 @@ internal partial class Av1TileWriter macroBlock, direction: 1); - writer.WriteSwitchableInterpolationFilter(macroBlockModeInfo.Block.HorizontalInterpolationFilter, horizontalContext); + writer.WriteSwitchableInterpolationFilter(macroBlockModeInfo.Block.HorizontalInterpolationFilter, horizontalContext); } } } else if (IsIntraBlockCopyAllowed(pcs.Parent.FrameHeader/*, pcs.Parent.SliceType*/)) { - WriteIntraBlockCopyInfo( + WriteIntraBlockCopyInfo( pcs, writer, macroBlock, modeInfoPosition, - macroBlockModeInfo); + macroBlockModeInfo, + TBlockEncoder.UsesRetainedDecisions); } if (!isInterBlock && !macroBlockModeInfo.Block.UseIntraBlockCopy) { - EncodeIntraLumaMode( + EncodeIntraLumaMode( writer, frm_hdr, macroBlockModeInfo, @@ -1021,7 +1123,7 @@ internal partial class Av1TileWriter { if (blk_ptr.HasChroma) { - EncodeIntraChromaMode( + EncodeIntraChromaMode( writer, frm_hdr, scs.SequenceHeader.ColorConfig, @@ -1039,7 +1141,7 @@ internal partial class Av1TileWriter if (paletteAllowed) { - WritePaletteModeInfo( + WritePaletteModeInfo( scs, pcs, writer, @@ -1060,7 +1162,7 @@ internal partial class Av1TileWriter paletteInfo.PaletteSizes[0], lumaMode)) { - writer.WriteFilterIntraMode(blk_ptr.FilterIntraMode, blockSize); + writer.WriteFilterIntraMode(blk_ptr.FilterIntraMode, blockSize); } if (paletteAllowed) @@ -1087,20 +1189,41 @@ internal partial class Av1TileWriter // that the decoder consumes before transform syntax and corrupt the remainder of the tile. int columns = (blockWidth + (Math.Min(0, macroBlock.ToRightEdge) >> 3)) >> subX; int rows = (blockHeight + (Math.Min(0, macroBlock.ToBottomEdge) >> 3)) >> subY; - Buffer2DRegion colorIndexMap = tb_ptr.Workspace - .GetPaletteMaps() - .GetMap(planeType, planeWidth, planeHeight); - - writer.WritePaletteColorMap( - paletteSize, - planeType, - rows, - columns, - colorIndexMap); + int tokenCount = rows * columns; + if (TBlockEncoder.UsesRetainedDecisions) + { + writer.WritePaletteTokens( + paletteSize, + planeType, + pcs.PaletteTokens.Span.Slice(entropyCodingContext.PaletteTokenOffset, tokenCount)); + } + else + { + Buffer2DRegion colorIndexMap = tb_ptr.Workspace + .GetPaletteMaps() + .GetMap(planeType, planeWidth, planeHeight); + + if (TOperation.WritesOutput) + { + writer.WritePaletteColorMap(paletteSize, planeType, rows, columns, colorIndexMap); + } + else + { + writer.TokenizePaletteColorMap( + paletteSize, + planeType, + rows, + columns, + colorIndexMap, + pcs.PaletteTokens.Span.Slice(entropyCodingContext.PaletteTokenOffset, tokenCount)); + } + } + + entropyCodingContext.PaletteTokenOffset += tokenCount; } } - WriteTransformSize( + WriteTransformSize( pcs, writer, ref macroBlockModeInfo, @@ -1112,7 +1235,7 @@ internal partial class Av1TileWriter entropyCodingContext.MacroBlockModeInfo = macroBlockModeInfo; if (!skipWritingCoefficients) { - EncodeCoefficients1d( + EncodeCoefficients1d( pcs, entropyCodingContext, writer, @@ -1124,7 +1247,8 @@ internal partial class Av1TileWriter tb_ptr.Index, luma_dc_sign_level_coeff_na, cr_dc_sign_level_coeff_na, - cb_dc_sign_level_coeff_na); + cb_dc_sign_level_coeff_na, + TBlockEncoder.UsesRetainedDecisions); } } @@ -1210,6 +1334,27 @@ internal partial class Av1TileWriter Av1BlockSize blockSize, Point blockOrigin, int tileIndex) + => WriteTransformSize( + pcs, + writer, + ref macroBlockModeInfo, + macroBlock, + blockSize, + blockOrigin, + tileIndex); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + internal static void WriteTransformSize( + Av1PictureControlSet pcs, + Av1SymbolEncoder writer, + ref Av1MacroBlockModeInfo macroBlockModeInfo, + Av1MacroBlockD macroBlock, + Av1BlockSize blockSize, + Point blockOrigin, + int tileIndex) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { ObuFrameHeader frameHeader = pcs.Parent.FrameHeader; bool isLossless = frameHeader.LosslessArray[macroBlockModeInfo.Block.SegmentId]; @@ -1236,7 +1381,7 @@ internal partial class Av1TileWriter if (writesUniformTransformSize) { int context = GetTransformSizeContext(transformContexts, macroBlock, blockOrigin, blockSize); - writer.WriteTransformSize(blockSize, transformSize, context); + writer.WriteTransformSize(blockSize, transformSize, context); } else if (writesVariableTransformSize) { @@ -1249,7 +1394,7 @@ internal partial class Av1TileWriter blockSize.GetMaximumTransformSize()); // Current inter decisions retain the maximum transform, so their variable-transform tree has one unsplit root. - writer.WriteTransformPartition(false, context); + writer.WriteTransformPartition(false, context); } Size blockDimensions = new(blockSize.GetWidth(), blockSize.GetHeight()); @@ -1325,6 +1470,29 @@ internal partial class Av1TileWriter Av1BlockSize blockSize, Av1PredictionMode lumaMode, Av1ChromaPredictionMode chromaMode) + => EncodeIntraChromaMode( + writer, + frameHeader, + colorConfig, + macroBlockModeInfo, + ref blk_ptr, + blockSize, + lumaMode, + chromaMode); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + public static void EncodeIntraChromaMode( + Av1SymbolEncoder writer, + ObuFrameHeader frameHeader, + ObuColorConfig colorConfig, + Av1MacroBlockModeInfo macroBlockModeInfo, + ref Av1EncoderBlockStruct blk_ptr, + Av1BlockSize blockSize, + Av1PredictionMode lumaMode, + Av1ChromaPredictionMode chromaMode) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { bool isChromaFromLumaAllowed = IsChromaFromLumaAllowed( frameHeader, @@ -1332,18 +1500,18 @@ internal partial class Av1TileWriter macroBlockModeInfo, blockSize); - writer.WriteChromaMode(chromaMode, isChromaFromLumaAllowed, lumaMode); + writer.WriteChromaMode(chromaMode, isChromaFromLumaAllowed, lumaMode); if (chromaMode == Av1ChromaPredictionMode.ChromaFromLuma) { - writer.WriteChromaFromLumaAlphas( + writer.WriteChromaFromLumaAlphas( blk_ptr.PredictionUnit.ChromaFromLumaIndex, blk_ptr.PredictionUnit.ChromaFromLumaSigns); } if (blockSize >= Av1BlockSize.Block8x8 && macroBlockModeInfo.Block.UvMode.IsDirectional()) { - writer.WriteAngleDelta( + writer.WriteAngleDelta( blk_ptr.PredictionUnit.AngleDelta[(int)Av1PlaneType.Uv] + Av1Constants.MaxAngleDelta, chromaMode.ToLumaMode()); } @@ -1431,7 +1599,7 @@ internal partial class Av1TileWriter /// The encoder prediction-unit state. /// The block size. /// The selected luma prediction mode. - private static void EncodeIntraLumaMode( + private static void EncodeIntraLumaMode( Av1SymbolEncoder writer, ObuFrameHeader frameHeader, Av1MacroBlockModeInfo macroBlockModeInfo, @@ -1439,20 +1607,21 @@ internal partial class Av1TileWriter ref Av1EncoderBlockStruct blk_ptr, Av1BlockSize blockSize, Av1PredictionMode lumaMode) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { if (frameHeader.IsIntra) { GetYModeContext(macroBlock, out byte topContext, out byte leftContext); - writer.WriteLumaMode(lumaMode, topContext, leftContext); + writer.WriteLumaMode(lumaMode, topContext, leftContext); } else { - writer.WriteInterFrameLumaMode(lumaMode, blockSize); + writer.WriteInterFrameLumaMode(lumaMode, blockSize); } if (blockSize >= Av1BlockSize.Block8x8 && macroBlockModeInfo.Block.Mode.IsDirectional()) { - writer.WriteAngleDelta(blk_ptr.PredictionUnit.AngleDelta[(int)Av1PlaneType.Y] + Av1Constants.MaxAngleDelta, lumaMode); + writer.WriteAngleDelta(blk_ptr.PredictionUnit.AngleDelta[(int)Av1PlaneType.Y] + Av1Constants.MaxAngleDelta, lumaMode); } } @@ -1565,6 +1734,33 @@ internal partial class Av1TileWriter Point blockOrigin, int tileIndex, bool hasChroma) + => WritePaletteModeInfo( + scs, + pcs, + writer, + macroBlock, + macroBlockModeInfo, + ref paletteInfo, + blockSize, + blockOrigin, + tileIndex, + hasChroma); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + internal static void WritePaletteModeInfo( + Av1SequenceControlSet scs, + Av1PictureControlSet pcs, + Av1SymbolEncoder writer, + Av1MacroBlockD macroBlock, + Av1MacroBlockModeInfo macroBlockModeInfo, + ref Av1EncoderPaletteInfo paletteInfo, + Av1BlockSize blockSize, + Point blockOrigin, + int tileIndex, + bool hasChroma) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { int blockSizeContext = GetPaletteBlockSizeContext(blockSize); Av1NeighborArrayUnit paletteContexts = pcs.PaletteContexts[tileIndex]; @@ -1572,10 +1768,10 @@ internal partial class Av1TileWriter if (macroBlockModeInfo.Block.Mode == Av1PredictionMode.DC) { int neighborContext = GetPaletteYModeContext(paletteContexts, macroBlock, blockOrigin); - writer.WritePaletteYMode(yPaletteSize != 0, blockSizeContext, neighborContext); + writer.WritePaletteYMode(yPaletteSize != 0, blockSizeContext, neighborContext); if (yPaletteSize != 0) { - writer.WritePaletteSize(yPaletteSize, blockSizeContext, Av1PlaneType.Y); + writer.WritePaletteSize(yPaletteSize, blockSizeContext, Av1PlaneType.Y); Span colorCache = stackalloc ushort[2 * Av1Constants.PaletteMaxSize]; int cacheSize = GetPaletteCache( paletteContexts, @@ -1584,7 +1780,7 @@ internal partial class Av1TileWriter Av1Plane.Y, colorCache); - writer.WritePaletteYColors( + writer.WritePaletteYColors( colorCache[..cacheSize], paletteInfo.GetColors(Av1Plane.Y), scs.SequenceHeader.ColorConfig.BitDepth.GetBitCount()); @@ -1596,10 +1792,10 @@ internal partial class Av1TileWriter macroBlockModeInfo.Block.UvMode == Av1ChromaPredictionMode.DC && hasChroma) { - writer.WritePaletteUvMode(uvPaletteSize != 0, yPaletteSize != 0); + writer.WritePaletteUvMode(uvPaletteSize != 0, yPaletteSize != 0); if (uvPaletteSize != 0) { - writer.WritePaletteSize(uvPaletteSize, blockSizeContext, Av1PlaneType.Uv); + writer.WritePaletteSize(uvPaletteSize, blockSizeContext, Av1PlaneType.Uv); Span colorCache = stackalloc ushort[2 * Av1Constants.PaletteMaxSize]; int cacheSize = GetPaletteCache( paletteContexts, @@ -1608,7 +1804,7 @@ internal partial class Av1TileWriter Av1Plane.U, colorCache); - writer.WritePaletteUvColors( + writer.WritePaletteUvColors( colorCache[..cacheSize], paletteInfo.GetColors(Av1Plane.U), paletteInfo.GetColors(Av1Plane.V), @@ -1719,23 +1915,62 @@ internal partial class Av1TileWriter Av1MacroBlockD macroBlock, Point modeInfoPosition, Av1MacroBlockModeInfo macroBlockModeInfo) + => WriteIntraBlockCopyInfo( + picture, + writer, + macroBlock, + modeInfoPosition, + macroBlockModeInfo, + false); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + public static void WriteIntraBlockCopyInfo( + Av1PictureControlSet picture, + Av1SymbolEncoder writer, + Av1MacroBlockD macroBlock, + Point modeInfoPosition, + Av1MacroBlockModeInfo macroBlockModeInfo, + bool useRetainedContext) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { bool useIntraBlockCopy = macroBlockModeInfo.Block.UseIntraBlockCopy; - writer.WriteUseIntraBlockCopy(useIntraBlockCopy); + writer.WriteUseIntraBlockCopy(useIntraBlockCopy); if (useIntraBlockCopy) { - Span candidates = stackalloc Av1MotionVector[8]; - Span weights = stackalloc int[8]; - Av1MotionVector reference = Av1IntraBlockCopy.FindReference( - picture, - macroBlock, - modeInfoPosition, - macroBlockModeInfo.Block.BlockSize, - macroBlockModeInfo.Block.PartitionType, - candidates, - weights); + int gridOffset = (modeInfoPosition.Y * picture.ModeInfoStride) + modeInfoPosition.X; + int allocationOffset = picture.ModeInfoGrid.Span[gridOffset]; + Av1MotionVector reference; + if (useRetainedContext) + { + Av1EncoderDisplacementVector retained = picture.ReferenceContexts.Span[allocationOffset].References[0]; + reference = new Av1MotionVector(retained.Row, retained.Column); + } + else + { + Span candidates = stackalloc Av1MotionVector[8]; + Span weights = stackalloc int[8]; + reference = Av1IntraBlockCopy.FindReference( + picture, + macroBlock, + modeInfoPosition, + macroBlockModeInfo.Block.BlockSize, + macroBlockModeInfo.Block.PartitionType, + candidates, + weights); + + if (!TOperation.WritesOutput) + { + picture.ReferenceContexts.Span[allocationOffset].References[0] = new Av1EncoderDisplacementVector + { + Row = (short)reference.Row, + Column = (short)reference.Column + }; + } + } - writer.WriteDisplacementVector( + writer.WriteDisplacementVector( picture.GetDisplacementVector(modeInfoPosition), reference); } @@ -1858,6 +2093,25 @@ internal partial class Av1TileWriter int tileIndex, bool skip, Point modeInfoPosition) + => WriteCdef( + scs, + pcs, + writer, + tileIndex, + skip, + modeInfoPosition); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + internal static void WriteCdef( + Av1SequenceControlSet scs, + Av1PictureControlSet pcs, + Av1SymbolEncoder writer, + int tileIndex, + bool skip, + Point modeInfoPosition) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { ObuFrameHeader frameHeader = pcs.Parent.FrameHeader; @@ -1893,7 +2147,7 @@ internal partial class Av1TileWriter // CDEF strength belongs to the first mode-info block in the 64x64 filter unit even when skipped // blocks delay transmission until a later coding block. - writer.WriteCdefStrength(firstBlock.CdefStrength, frameHeader.CdefParameters.BitCount); + writer.WriteCdefStrength(firstBlock.CdefStrength, frameHeader.CdefParameters.BitCount); cdefPreset[index] = firstBlock.CdefStrength; } } @@ -1950,7 +2204,8 @@ internal partial class Av1TileWriter /// The luma coefficient neighbor contexts. /// The red-difference chroma coefficient neighbor contexts. /// The blue-difference chroma coefficient neighbor contexts. - private static void EncodeCoefficients1d( + /// Whether coefficient contexts come from completed block analysis. + private static void EncodeCoefficients1d( Av1PictureControlSet pcs, Av1EntropyCodingContext ec_ctx, Av1SymbolEncoder writer, @@ -1962,9 +2217,11 @@ internal partial class Av1TileWriter int superblockIndex, Av1NeighborArrayUnit luma_dc_sign_level_coeff_na, Av1NeighborArrayUnit cr_dc_sign_level_coeff_na, - Av1NeighborArrayUnit cb_dc_sign_level_coeff_na) + Av1NeighborArrayUnit cb_dc_sign_level_coeff_na, + bool useRetainedContexts) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { - EncodeTransformCoefficientRegions( + EncodeTransformCoefficientRegions( pcs, ec_ctx, writer, @@ -1976,7 +2233,8 @@ internal partial class Av1TileWriter superblockIndex, luma_dc_sign_level_coeff_na, cr_dc_sign_level_coeff_na, - cb_dc_sign_level_coeff_na); + cb_dc_sign_level_coeff_na, + useRetainedContexts); } /// @@ -2003,6 +2261,35 @@ internal partial class Av1TileWriter Av1EncoderCoefficientBuffer coefficientBuffer, int superblockIndex, Av1NeighborArrayUnit luma_dc_sign_level_coeff_na) + => EncodeTransformCoefficientsY( + pcs, + entropyCodingContext, + writer, + ref blk_ptr, + blockOrigin, + intraLumaDir, + plane_bsize, + coefficientBuffer, + superblockIndex, + luma_dc_sign_level_coeff_na, + false); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + public static void EncodeTransformCoefficientsY( + Av1PictureControlSet pcs, + Av1EntropyCodingContext entropyCodingContext, + Av1SymbolEncoder writer, + ref Av1EncoderBlockStruct blk_ptr, + Point blockOrigin, + Av1PredictionMode intraLumaDir, + Av1BlockSize plane_bsize, + Av1EncoderCoefficientBuffer coefficientBuffer, + int superblockIndex, + Av1NeighborArrayUnit luma_dc_sign_level_coeff_na, + bool useRetainedContexts) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { Av1MacroBlockD macroBlock = entropyCodingContext.MacroBlock; int maximumBlocksWide = plane_bsize.GetWidth(); @@ -2033,7 +2320,7 @@ internal partial class Av1TileWriter for (int regionColumn = 0; regionColumn < maximumBlocksWide; regionColumn += maximumUnitBlocksWide) { int unitRight = Math.Min(regionColumn + maximumUnitBlocksWide, maximumBlocksWide); - EncodeTransformCoefficientRegion( + EncodeTransformCoefficientRegion( pcs, entropyCodingContext, writer, @@ -2048,7 +2335,8 @@ internal partial class Av1TileWriter regionRow, regionColumn, unitBottom, - unitRight); + unitRight, + useRetainedContexts); } } } @@ -2067,7 +2355,8 @@ internal partial class Av1TileWriter /// The raster-ordered index of the containing superblock. /// The red-difference chroma coefficient neighbor contexts. /// The blue-difference chroma coefficient neighbor contexts. - private static void EncodeTransformCoefficientsUv( + /// Whether coefficient contexts come from completed block analysis. + private static void EncodeTransformCoefficientsUv( Av1PictureControlSet pcs, Av1EntropyCodingContext entropyCodingContext, Av1SymbolEncoder writer, @@ -2078,7 +2367,9 @@ internal partial class Av1TileWriter Av1EncoderCoefficientBuffer coefficientBuffer, int superblockIndex, Av1NeighborArrayUnit cr_dc_sign_level_coeff_na, - Av1NeighborArrayUnit cb_dc_sign_level_coeff_na) + Av1NeighborArrayUnit cb_dc_sign_level_coeff_na, + bool useRetainedContexts) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { ObuColorConfig colorConfig = pcs.Sequence.SequenceHeader.ColorConfig; if (!blk_ptr.HasChroma || colorConfig.IsMonochrome) @@ -2117,7 +2408,7 @@ internal partial class Av1TileWriter for (int regionColumn = 0; regionColumn < maximumBlocksWide; regionColumn += maximumUnitBlocksWide) { int unitRight = Math.Min(regionColumn + maximumUnitBlocksWide, maximumBlocksWide); - EncodeTransformCoefficientRegion( + EncodeTransformCoefficientRegion( pcs, entropyCodingContext, writer, @@ -2132,9 +2423,10 @@ internal partial class Av1TileWriter regionRow, regionColumn, unitBottom, - unitRight); + unitRight, + useRetainedContexts); - EncodeTransformCoefficientRegion( + EncodeTransformCoefficientRegion( pcs, entropyCodingContext, writer, @@ -2149,12 +2441,13 @@ internal partial class Av1TileWriter regionRow, regionColumn, unitBottom, - unitRight); + unitRight, + useRetainedContexts); } } } - private static void EncodeTransformCoefficientRegions( + private static void EncodeTransformCoefficientRegions( Av1PictureControlSet pcs, Av1EntropyCodingContext entropyCodingContext, Av1SymbolEncoder writer, @@ -2166,7 +2459,9 @@ internal partial class Av1TileWriter int superblockIndex, Av1NeighborArrayUnit lumaCoefficientNeighbors, Av1NeighborArrayUnit redCoefficientNeighbors, - Av1NeighborArrayUnit blueCoefficientNeighbors) + Av1NeighborArrayUnit blueCoefficientNeighbors, + bool useRetainedContexts) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { Av1MacroBlockD macroBlock = entropyCodingContext.MacroBlock; int maximumBlocksWide = blockSize.GetWidth(); @@ -2205,7 +2500,7 @@ internal partial class Av1TileWriter for (int regionColumn = 0; regionColumn < maximumBlocksWide; regionColumn += maximumUnitBlocksWide) { int unitRight = Math.Min(regionColumn + maximumUnitBlocksWide, maximumBlocksWide); - EncodeTransformCoefficientRegion( + EncodeTransformCoefficientRegion( pcs, entropyCodingContext, writer, @@ -2220,7 +2515,8 @@ internal partial class Av1TileWriter regionRow, regionColumn, unitBottom, - unitRight); + unitRight, + useRetainedContexts); if (hasChroma) { @@ -2231,7 +2527,7 @@ internal partial class Av1TileWriter // 4x4, 4x8, or 8x4 luma block still emits its shared 4x4 chroma transform. int chromaUnitBottom = Av1Math.RoundPowerOf2(unitBottom, subsamplingY); int chromaUnitRight = Av1Math.RoundPowerOf2(unitRight, subsamplingX); - EncodeTransformCoefficientRegion( + EncodeTransformCoefficientRegion( pcs, entropyCodingContext, writer, @@ -2246,9 +2542,10 @@ internal partial class Av1TileWriter chromaRegionRow, chromaRegionColumn, chromaUnitBottom, - chromaUnitRight); + chromaUnitRight, + useRetainedContexts); - EncodeTransformCoefficientRegion( + EncodeTransformCoefficientRegion( pcs, entropyCodingContext, writer, @@ -2263,13 +2560,14 @@ internal partial class Av1TileWriter chromaRegionRow, chromaRegionColumn, chromaUnitBottom, - chromaUnitRight); + chromaUnitRight, + useRetainedContexts); } } } } - private static void EncodeTransformCoefficientRegion( + private static void EncodeTransformCoefficientRegion( Av1PictureControlSet pcs, Av1EntropyCodingContext entropyCodingContext, Av1SymbolEncoder writer, @@ -2284,7 +2582,9 @@ internal partial class Av1TileWriter int regionRow, int regionColumn, int unitBottom, - int unitRight) + int unitRight, + bool useRetainedContexts) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { ObuFrameHeader frameHeader = pcs.Parent.FrameHeader; ObuColorConfig colorConfig = pcs.Sequence.SequenceHeader.ColorConfig; @@ -2333,12 +2633,29 @@ internal partial class Av1TileWriter blockRow << Av1Constants.ModeInfoSizeLog2); Span coefficients = planeCoefficients[codedArea..]; - Av1TransformBlockContext blockContext = GetTransformBlockContexts( - componentType, - coefficientNeighbors, - transformOrigin, - planeBlockSize, - transformSize); + Av1TransformBlockContext blockContext; + if (useRetainedContexts) + { + byte packedContext = transformBlock.EntropyContext; + blockContext = new Av1TransformBlockContext + { + SkipContext = packedContext & 15, + DcSignContext = packedContext >> 4 + }; + } + else + { + blockContext = GetTransformBlockContexts( + componentType, + coefficientNeighbors, + transformOrigin, + planeBlockSize, + transformSize); + + // Neighbor probabilities must describe the selected transform at analysis time, before + // final packing revisits the frame. Both context alphabets fit in the existing spare byte. + transformBlock.EntropyContext = (byte)(blockContext.SkipContext | (blockContext.DcSignContext << 4)); + } Av1TransformType transformType = transformBlock.TransformType; if (isLuma && transformBlock.EndOfBlock == 0) @@ -2347,7 +2664,7 @@ internal partial class Av1TileWriter transformType = transformBlock.TransformType = Av1TransformType.DctDct; } - int culLevel = writer.WriteCoefficients( + int culLevel = writer.WriteCoefficients( transformSize, transformType, intraLumaMode, @@ -2510,7 +2827,7 @@ internal partial class Av1TileWriter /// The encoder block state. /// A value indicating whether residual coefficients are omitted. /// Whether the segment identifier is written before the skip flag. - private static void WriteSegmentId( + private static void WriteSegmentId( Av1PictureControlSet pcs, Av1SymbolEncoder writer, Av1BlockSize blockSize, @@ -2519,6 +2836,7 @@ internal partial class Av1TileWriter ref Av1EncoderBlockStruct block, bool skip, bool beforeSkip) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation { ObuSegmentationParameters segmentation_params = pcs.Parent.FrameHeader.SegmentationParameters; if (!segmentation_params.Enabled) @@ -2536,7 +2854,7 @@ internal partial class Av1TileWriter } int coded_id = Av1SymbolContextHelper.NegativeDeinterleave(block.SegmentId, spatial_pred, segmentation_params.LastActiveSegmentId + 1); - writer.WriteSegmentId(coded_id, cdf_num); + writer.WriteSegmentId(coded_id, cdf_num); pcs.UpdateSegmentation(blockSize, blockOrigin, block.SegmentId); } @@ -2657,5 +2975,15 @@ internal partial class Av1TileWriter /// The reusable macroblock edge and neighbor state. /// The skip value to write. public static void EncodeSkipCoefficients(Av1SymbolEncoder writer, Av1MacroBlockD macroBlock, bool skip) - => writer.WriteSkip(skip, GetSkipContext(macroBlock)); + => EncodeSkipCoefficients( + writer, + macroBlock, + skip); + + /// + /// Processes the selected syntax and its adaptive probability state. + /// + public static void EncodeSkipCoefficients(Av1SymbolEncoder writer, Av1MacroBlockD macroBlock, bool skip) + where TOperation : struct, Av1SymbolEncoder.ISymbolOperation + => writer.WriteSkip(skip, GetSkipContext(macroBlock)); } diff --git a/src/ImageSharp/Formats/Heif/HeifEncoder.cs b/src/ImageSharp/Formats/Heif/HeifEncoder.cs index 1b6472ee5a..540f902c44 100644 --- a/src/ImageSharp/Formats/Heif/HeifEncoder.cs +++ b/src/ImageSharp/Formats/Heif/HeifEncoder.cs @@ -23,6 +23,11 @@ public sealed class HeifEncoder : AnimatedImageEncoder /// private int effort = 5; + /// + /// The AV1 encoding speed. + /// + private HeifEncodingSpeed speed; + /// /// Gets the compression method used for the primary image item. /// The default is . @@ -96,6 +101,25 @@ public sealed class HeifEncoder : AnimatedImageEncoder /// public bool Lossless { get; init; } + /// + /// Gets the AV1 encoding speed. Higher levels prioritize speed over compression efficiency. + /// The default is . + /// + /// The speed is outside the range 0 to 9. + public HeifEncodingSpeed Speed + { + get => this.speed; + init + { + if (value is < HeifEncodingSpeed.Level0 or > HeifEncodingSpeed.Level9) + { + throw new ArgumentException("Speed must be in the range [0..9]."); + } + + this.speed = value; + } + } + /// /// Gets the encoded precision of each image component, or to use the HEIF metadata bit /// depth. Metadata that does not specify a bit depth defaults to . Legacy JPEG diff --git a/src/ImageSharp/Formats/Heif/HeifEncoderCore.Sequence.cs b/src/ImageSharp/Formats/Heif/HeifEncoderCore.Sequence.cs index 22d3ff710b..746a92900c 100644 --- a/src/ImageSharp/Formats/Heif/HeifEncoderCore.Sequence.cs +++ b/src/ImageSharp/Formats/Heif/HeifEncoderCore.Sequence.cs @@ -235,7 +235,8 @@ internal sealed partial class HeifEncoderCore image.Height, settings.ColorConfig, settings.ColorQIndex, - this.encoder.Effort)) + this.encoder.Effort, + speed: this.encoder.Speed)) { cancellationToken.ThrowIfCancellationRequested(); long colorOffset = stream.Length; @@ -292,7 +293,8 @@ internal sealed partial class HeifEncoderCore image.Height, settings.AlphaConfig, settings.AlphaQIndex, - this.encoder.Effort)) + this.encoder.Effort, + speed: this.encoder.Speed)) { cancellationToken.ThrowIfCancellationRequested(); long alphaOffset = stream.Length; diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs b/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs index 619e3b30cd..0ef5d2264f 100644 --- a/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs +++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs @@ -3,6 +3,7 @@ using System.Numerics; using BenchmarkDotNet.Attributes; +using SixLabors.ImageSharp.Formats.Heif; using SixLabors.ImageSharp.Formats.Heif.Av1; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; @@ -106,7 +107,7 @@ public class Av1SequenceEncoderBenchmarks // Export the production-converted source planes only for checking reconstructed output quality. // Neither timed encoder reads this file: both convert the original RGB frames during each operation. - using Av1EncoderFrameBuffer planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0); + using Av1EncoderFrameBuffer planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0, lumaBorder: 64); using FileStream raw = File.Create(Path.Combine(this.outputDirectory, $"bike-{this.Dimension}-3frames.source.yuv")); foreach (ImageFrame frame in this.sequence.Frames) { @@ -131,7 +132,13 @@ public class Av1SequenceEncoderBenchmarks { this.output.SetLength(0); using Av1FrameEncoder.SequenceEncoder encoder = Av1FrameEncoder.CreateColorSequenceEncoder( - this.configuration, this.Dimension, this.Dimension, this.colorConfig, QIndex, this.Effort); + this.configuration, + this.Dimension, + this.Dimension, + this.colorConfig, + QIndex, + this.Effort, + speed: HeifEncodingSpeed.Level0); // One operation owns the real sequence lifetime: allocation, conversion, key/inter coding, and disposal. // The caller's destination is reused, excluding filesystem and MemoryStream growth from steady-state timing. @@ -155,7 +162,7 @@ public class Av1SequenceEncoderBenchmarks // good-quality speed six while comparing the three managed interpolation-search boundaries. this.output.SetLength(0); using LibaomBenchmarkEncoder encoder = LibaomBenchmarkEncoder.Open(this.Dimension, this.Dimension, NativeQuality, NativeCpuUsed); - using Av1EncoderFrameBuffer planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0); + using Av1EncoderFrameBuffer planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0, lumaBorder: 64); using Av1FrameEncoder.Av1EncoderConversionWorkspace conversion = new(this.configuration, this.Dimension, this.colorConfig, false, false); Rectangle bounds = new(0, 0, this.Dimension, this.Dimension); for (int frameIndex = 0; frameIndex < FrameCount; frameIndex++) diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs index 6cec574eaf..64bf6bd712 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs @@ -3,6 +3,7 @@ using System.Runtime.InteropServices; using SixLabors.ImageSharp.Formats; +using SixLabors.ImageSharp.Formats.Heif; using SixLabors.ImageSharp.Formats.Heif.Av1; using SixLabors.ImageSharp.Formats.Heif.Av1.Motion; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; @@ -316,14 +317,16 @@ public class Av1EncoderFrameTests Height, colorConfig, qIndex: 37, - effort: 5) + effort: 5, + speed: HeifEncodingSpeed.Level0) : Av1FrameEncoder.CreateColorSequenceEncoder( Configuration.Default, Width, Height, colorConfig, qIndex: 37, - effort: 5); + effort: 5, + speed: HeifEncodingSpeed.Level0); encoder.EncodeKeyFrame(source.Frames.RootFrame, stream); ObuSequenceHeader encodedHeader = encoder.SequenceHeader; @@ -345,19 +348,50 @@ public class Av1EncoderFrameTests /// Verifies dependent color samples with odd visible dimensions and motion across subsampled chroma phases. /// [Theory] - [InlineData(EightBit, Yuv420, 8)] - [InlineData(TenBit, Yuv420, 8)] - [InlineData(TwelveBit, Yuv420, 8)] - [InlineData(EightBit, Yuv420, 9)] - [InlineData(TenBit, Yuv420, 9)] - [InlineData(TwelveBit, Yuv420, 9)] - [InlineData(EightBit, Yuv422, 9)] - [InlineData(TenBit, Yuv422, 9)] - [InlineData(TwelveBit, Yuv422, 9)] - [InlineData(EightBit, Yuv444, 9)] - [InlineData(TenBit, Yuv444, 9)] - [InlineData(TwelveBit, Yuv444, 9)] - public void SequenceEncoderPreservesNativeColorPlanesWithSubpixelMotion(int bitDepthValue, int colorFormatValue, int effort) + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level0)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level0)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level0)] + [InlineData(EightBit, Yuv420, 9, HeifEncodingSpeed.Level0)] + [InlineData(TenBit, Yuv420, 9, HeifEncodingSpeed.Level0)] + [InlineData(TwelveBit, Yuv420, 9, HeifEncodingSpeed.Level0)] + [InlineData(EightBit, Yuv422, 9, HeifEncodingSpeed.Level0)] + [InlineData(TenBit, Yuv422, 9, HeifEncodingSpeed.Level0)] + [InlineData(TwelveBit, Yuv422, 9, HeifEncodingSpeed.Level0)] + [InlineData(EightBit, Yuv444, 9, HeifEncodingSpeed.Level0)] + [InlineData(TenBit, Yuv444, 9, HeifEncodingSpeed.Level0)] + [InlineData(TwelveBit, Yuv444, 9, HeifEncodingSpeed.Level0)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level1)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level2)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level3)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level4)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level5)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level6)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level7)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level8)] + [InlineData(EightBit, Yuv420, 8, HeifEncodingSpeed.Level9)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level1)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level2)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level3)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level4)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level5)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level6)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level7)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level8)] + [InlineData(TenBit, Yuv420, 8, HeifEncodingSpeed.Level9)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level1)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level2)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level3)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level4)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level5)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level6)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level7)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level8)] + [InlineData(TwelveBit, Yuv420, 8, HeifEncodingSpeed.Level9)] + public void SequenceEncoderPreservesNativeColorPlanesWithSubpixelMotion( + int bitDepthValue, + int colorFormatValue, + int effort, + HeifEncodingSpeed speed) { const int Width = 23; const int Height = 19; @@ -369,15 +403,21 @@ public class Av1EncoderFrameTests ReadOnlySpan period = [0, 28, 40, 28, 0, -28, -40, -12]; using Image source = new(Width, Height); using Av1FrameEncoder.SequenceEncoder encoder = Av1FrameEncoder.CreateColorSequenceEncoder( - Configuration.Default, Width, Height, colorConfig, QIndex, effort); + Configuration.Default, + Width, + Height, + colorConfig, + QIndex, + effort, + speed); string outputDirectory = TestEnvironment.CreateOutputDirectory("Heif", "Av1", nameof(this.SequenceEncoderPreservesNativeColorPlanesWithSubpixelMotion)); - string outputName = $"{bitDepth.GetBitCount()}-{colorFormat}-effort{effort}"; + string outputName = $"{bitDepth.GetBitCount()}-{colorFormat}-effort{effort}-speed{(int)speed}"; using FileStream output = File.Create(Path.Combine(outputDirectory, outputName + ".obu")); using BinaryWriter rawOutput = new(File.Create(Path.Combine(outputDirectory, outputName + ".managed.yuv"))); using Av1Decoder decoder = new(Configuration.Default); using MemoryStream sample = new(); - for (int frameIndex = 0; frameIndex < 2; frameIndex++) + for (int frameIndex = 0; frameIndex < 3; frameIndex++) { // The second source translates all three channels by one luma sample on each axis. Chroma is // converted independently by the production converter, so 4:2:0 and 4:2:2 cannot hide behind @@ -504,7 +544,8 @@ public class Av1EncoderFrameTests Height, colorConfig, qIndex: 37, - effort); + effort, + speed: HeifEncodingSpeed.Level0); encoder.EncodeKeyFrame(source.Frames.RootFrame, firstSample); encoder.EncodeInterFrame(source.Frames.RootFrame, secondSample); @@ -601,7 +642,8 @@ public class Av1EncoderFrameTests Height, colorConfig, qIndex: 4, - effort: 6); + effort: 6, + speed: HeifEncodingSpeed.Level0); encoder.EncodeKeyFrame(first.Frames.RootFrame, firstSample); encoder.EncodeInterFrame(second.Frames.RootFrame, secondSample); @@ -649,7 +691,13 @@ public class Av1EncoderFrameTests Assert.Throws(() => { using Av1FrameEncoder.SequenceEncoder encoder = Av1FrameEncoder.CreateColorSequenceEncoder( - configuration, 32, 32, colorConfig, 17, 9); + configuration, + 32, + 32, + colorConfig, + 17, + 9, + speed: HeifEncodingSpeed.Level0); }); Assert.Empty(allocator.AllocationLog); @@ -674,8 +722,8 @@ public class Av1EncoderFrameTests successfulAllocator.EnableNonThreadSafeLogging(); configuration.MemoryAllocator = successfulAllocator; using (Av1FrameEncoder.SequenceEncoder encoder = encodeAlpha - ? Av1FrameEncoder.CreateAlphaSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9) - : Av1FrameEncoder.CreateColorSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9)) + ? Av1FrameEncoder.CreateAlphaSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9, speed: HeifEncodingSpeed.Level0) + : Av1FrameEncoder.CreateColorSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9, speed: HeifEncodingSpeed.Level0)) { Assert.NotEmpty(successfulAllocator.AllocationLog); } @@ -691,8 +739,8 @@ public class Av1EncoderFrameTests InvalidMemoryOperationException exception = Assert.Throws(() => { using Av1FrameEncoder.SequenceEncoder encoder = encodeAlpha - ? Av1FrameEncoder.CreateAlphaSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9) - : Av1FrameEncoder.CreateColorSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9); + ? Av1FrameEncoder.CreateAlphaSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9, speed: HeifEncodingSpeed.Level0) + : Av1FrameEncoder.CreateColorSequenceEncoder(configuration, 32, 32, colorConfig, 17, 9, speed: HeifEncodingSpeed.Level0); }); Assert.Equal("Sequence allocation failure.", exception.Message); @@ -739,14 +787,16 @@ public class Av1EncoderFrameTests Height, colorConfig, qIndex: 37, - effort: 6) + effort: 6, + speed: HeifEncodingSpeed.Level0) : Av1FrameEncoder.CreateColorSequenceEncoder( configuration, Width, Height, colorConfig, qIndex: 37, - effort: 6)) + effort: 6, + speed: HeifEncodingSpeed.Level0)) { rowStorage = Assert.Single( allocator.AllocationLog, @@ -808,7 +858,8 @@ public class Av1EncoderFrameTests bitDepth.GetBitCount(), Av1ColorFormat.Yuv444, 1, - 1); + 1, + lumaBorder: 64); Av1FrameEncoder.PrepareSource( Configuration.Default, @@ -1268,7 +1319,8 @@ public class Av1EncoderFrameTests 12, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); TestMemoryAllocator allocator = new(); allocator.EnableNonThreadSafeLogging(); @@ -1339,7 +1391,8 @@ public class Av1EncoderFrameTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); for (int row = 0; row < height; row++) { @@ -1384,7 +1437,8 @@ public class Av1EncoderFrameTests 10, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); for (int row = 0; row < 16; row++) { @@ -1448,7 +1502,8 @@ public class Av1EncoderFrameTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion luma = frame.Frame.View.GetPlane(Av1Plane.Y); for (int row = 0; row < Height; row++) @@ -1762,7 +1817,8 @@ public class Av1EncoderFrameTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); ObuColorConfig colorConfig = CreateColorConfig(Av1BitDepth.EightBit); @@ -1792,7 +1848,8 @@ public class Av1EncoderFrameTests 10, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); ObuColorConfig colorConfig = CreateColorConfig(Av1BitDepth.TenBit); @@ -1802,13 +1859,15 @@ public class Av1EncoderFrameTests AssertReplicatedSingleRow(frameBuffer.Luma, border, expected); } - [Fact] - public void ExtendBordersReplicatesEveryPhysicalPlaneEdge() + [Theory] + [InlineData(64)] + [InlineData(96)] + [InlineData(160)] + public void ExtendBordersReplicatesEveryPhysicalPlaneEdge(int lumaBorder) { const int visibleWidth = 5; const int visibleHeight = 3; - const int lumaBorder = Av1EncoderFrame.LumaBorder; - const int chromaBorder = lumaBorder / 2; + int chromaBorder = lumaBorder / 2; using Av1EncoderFrameBuffer frameBuffer = new( Configuration.Default, @@ -1817,7 +1876,8 @@ public class Av1EncoderFrameTests 8, Av1ColorFormat.Yuv420, 1, - 1); + 1, + lumaBorder); Buffer2D luma = frameBuffer.Luma; Buffer2D chromaBlue = Assert.IsType>(frameBuffer.ChromaBlue); @@ -1835,12 +1895,19 @@ public class Av1EncoderFrameTests } [Theory] - [InlineData(5, 3, 0, 0, 160, 136)] - [InlineData(5, 3, 1, 0, 80, 136)] - [InlineData(5, 3, 1, 1, 80, 68)] - [InlineData(1921, 1081, 0, 0, 2080, 1216)] - [InlineData(1921, 1081, 1, 1, 1040, 608)] + [InlineData(64, 5, 3, 0, 0, 160, 136)] + [InlineData(64, 5, 3, 1, 0, 80, 136)] + [InlineData(64, 5, 3, 1, 1, 80, 68)] + [InlineData(64, 1921, 1081, 0, 0, 2080, 1216)] + [InlineData(64, 1921, 1081, 1, 1, 1040, 608)] + [InlineData(96, 5, 3, 0, 0, 224, 200)] + [InlineData(96, 5, 3, 1, 0, 112, 200)] + [InlineData(96, 5, 3, 1, 1, 112, 100)] + [InlineData(160, 5, 3, 0, 0, 352, 328)] + [InlineData(160, 5, 3, 1, 0, 176, 328)] + [InlineData(160, 5, 3, 1, 1, 176, 164)] public void GetPlaneBufferSizeMatchesLibaomLayout( + int lumaBorder, int width, int height, int subsamplingX, @@ -1848,13 +1915,16 @@ public class Av1EncoderFrameTests int expectedWidth, int expectedHeight) { - Size actual = Av1EncoderFrame.GetPlaneBufferSize(width, height, subsamplingX, subsamplingY); + Size actual = Av1EncoderFrame.GetPlaneBufferSize(width, height, subsamplingX, subsamplingY, lumaBorder); Assert.Equal(new Size(expectedWidth, expectedHeight), actual); } - [Fact] - public void FrameBufferUsesOneExactSizeOwnerForAllPlanes() + [Theory] + [InlineData(64, 55_296)] + [InlineData(96, 98_304)] + [InlineData(160, 221_184)] + public void FrameBufferUsesOneExactSizeOwnerForAllPlanes(int lumaBorder, int expectedLength) { TestMemoryAllocator allocator = new(); allocator.EnableNonThreadSafeLogging(); @@ -1869,12 +1939,13 @@ public class Av1EncoderFrameTests 8, Av1ColorFormat.Yuv420, 1, - 1)) + 1, + lumaBorder)) { allocation = Assert.Single(allocator.AllocationLog); Assert.Empty(allocator.ReturnLog); Assert.Equal(typeof(byte), allocation.ElementType); - Assert.Equal(55_296, allocation.Length); + Assert.Equal(expectedLength, allocation.Length); Assert.Single(frameBuffer.Luma.MemoryGroup); Assert.Single(Assert.IsType>(frameBuffer.ChromaBlue).MemoryGroup); Assert.Single(Assert.IsType>(frameBuffer.ChromaRed).MemoryGroup); diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderModeInfoBufferTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderModeInfoBufferTests.cs index 726a7c3b80..b10d3a153b 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderModeInfoBufferTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderModeInfoBufferTests.cs @@ -1,6 +1,7 @@ // Copyright (c) Six Labors. // Licensed under the Six Labors Split License. +using System.Runtime.InteropServices; using SixLabors.ImageSharp.Formats.Heif.Av1; using SixLabors.ImageSharp.Formats.Heif.Av1.Motion; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; @@ -61,6 +62,21 @@ public class Av1EncoderModeInfoBufferTests // Four CDEF presets, the preceding quantizer, and two payload bounds follow the context regions. const int TileStateStorageLength = 7 * sizeof(int); + const int AllocatedBlockCount = 256; + const int BlockEncodingStorageLength = AllocatedBlockCount * 8; + const int BlockPaletteStorageLength = AllocatedBlockCount * 50; + const int PaletteTokenStorageLength = 2 * 128 * 128; + + // Four vectors (16 bytes), four weights (8), mode context (2), count (1), and one alignment byte. + const int ReferenceContextStorageLength = AllocatedBlockCount * 28; + + // Retained syntax uses fixed-width entries at the existing block origins. Palette tokens reserve + // two complete maximum-superblock planes, including coded padding beyond this small visible frame. + int retainedStorageLength = BlockEncodingStorageLength + + (allowScreenContentTools ? BlockPaletteStorageLength + PaletteTokenStorageLength : 0) + + (allowIntraBlockCopy ? ReferenceContextStorageLength : 0); + + int expectedTileStateOffset = expectedContextStorageLength + retainedStorageLength; TestMemoryAllocator allocator = new(); allocator.EnableNonThreadSafeLogging(); Configuration configuration = Configuration.Default.Clone(); @@ -111,11 +127,15 @@ public class Av1EncoderModeInfoBufferTests Assert.Equal(6_144, allocations[0].Length); Assert.Equal(AllocationOptions.Clean, allocations[0].AllocationOptions); Assert.Equal(typeof(byte), allocations[1].ElementType); - Assert.Equal(expectedContextStorageLength + TileStateStorageLength, allocations[1].Length); + Assert.Equal(expectedTileStateOffset + TileStateStorageLength, allocations[1].Length); Assert.Equal(AllocationOptions.Clean, allocations[1].AllocationOptions); Assert.Empty(allocator.ReturnLog); Av1PictureControlSet picture = buffer.Picture; + Assert.Equal(AllocatedBlockCount, picture.BlockEncodings.Length); + Assert.Equal(8, sizeof(Av1EncoderBlockStruct)); + Assert.Equal(28, sizeof(Av1EncoderReferenceContext)); + Assert.Equal(-1, MemoryMarshal.AsBytes(picture.BlockEncodings.Span).IndexOfAnyExcept((byte)0)); Assert.Equal(16, picture.SegmentationNeighborMap.Length); Assert.Equal(32, picture.PartitionContexts[0].Left.Length); Assert.Equal(32, picture.PartitionContexts[0].Top.Length); @@ -140,7 +160,7 @@ public class Av1EncoderModeInfoBufferTests lengths = picture.TileDataLengths.Span) { Assert.Equal((nuint)0, (nuint)cdef % (nuint)sizeof(int)); - Assert.Equal(expectedContextStorageLength, (byte*)cdef - state); + Assert.Equal(expectedTileStateOffset, (byte*)cdef - state); Assert.Equal(4, quantizer - cdef); Assert.Equal(1, offsets - quantizer); Assert.Equal(1, lengths - offsets); @@ -150,6 +170,22 @@ public class Av1EncoderModeInfoBufferTests if (allowScreenContentTools) { + Assert.Equal(AllocatedBlockCount, picture.BlockPalettes.Length); + Assert.Equal(PaletteTokenStorageLength, picture.PaletteTokens.Length); + Assert.Equal(-1, MemoryMarshal.AsBytes(picture.BlockPalettes.Span).IndexOfAnyExcept((byte)0)); + Assert.Equal(0, picture.PaletteTokens.Span[^1]); + fixed (byte* tokens = picture.PaletteTokens.Span) + { + fixed (Av1EncoderPaletteInfo* palettes = picture.BlockPalettes.Span) + { + fixed (Av1EncoderBlockStruct* encodings = picture.BlockEncodings.Span) + { + Assert.Equal(BlockPaletteStorageLength, tokens - (byte*)palettes); + Assert.Equal(PaletteTokenStorageLength, (byte*)encodings - tokens); + } + } + } + Av1NeighborArrayUnit paletteContext = Assert.Single(picture.PaletteContexts); Assert.Equal(32, paletteContext.Left.Length); Assert.Equal(32, paletteContext.Top.Length); @@ -169,10 +205,26 @@ public class Av1EncoderModeInfoBufferTests else { Assert.Empty(picture.PaletteContexts); + Assert.True(picture.BlockPalettes.IsEmpty); + Assert.True(picture.PaletteTokens.IsEmpty); } if (allowIntraBlockCopy) { + Assert.Equal(AllocatedBlockCount, picture.ReferenceContexts.Length); + Assert.Equal(-1, MemoryMarshal.AsBytes(picture.ReferenceContexts.Span).IndexOfAnyExcept((byte)0)); + fixed (Av1EncoderDisplacementVector* vectors = picture.DisplacementVectors.Span) + { + fixed (Av1EncoderReferenceContext* references = picture.ReferenceContexts.Span) + { + fixed (Av1EncoderBlockStruct* encodings = picture.BlockEncodings.Span) + { + Assert.Equal(BlockEncodingStorageLength, (byte*)vectors - (byte*)encodings); + Assert.Equal(AllocatedBlockCount * 4, (byte*)references - (byte*)vectors); + } + } + } + Assert.Equal(256, picture.DisplacementVectors.Length); Assert.Equal(4, sizeof(Av1EncoderDisplacementVector)); Assert.Equal(9, picture.IntraBlockCopySearch.OriginWidth); @@ -193,6 +245,7 @@ public class Av1EncoderModeInfoBufferTests else { Assert.Equal(0, picture.DisplacementVectors.Length); + Assert.True(picture.ReferenceContexts.IsEmpty); } } @@ -313,6 +366,10 @@ public class Av1EncoderModeInfoBufferTests picture.TransformFunctionContexts[0].Top[0] = 8; picture.PaletteContexts[0].Left[0].PaletteSizes[0] = 2; picture.DisplacementVectors.Span[0] = new Av1EncoderDisplacementVector { Row = -8, Column = 16 }; + picture.BlockEncodings.Span[0].QuantizationIndex = 53; + picture.BlockPalettes.Span[0].PaletteSizes[0] = 3; + picture.PaletteTokens.Span[0] = 0x42; + picture.ReferenceContexts.Span[0].ModeContext = 37; picture.CdefPreset.Span[0] = 2; picture.Parent.PreviousQIndex.Span[0] = InitialQIndex + 1; picture.TileDataOffsets.Span[0] = 11; @@ -345,12 +402,35 @@ public class Av1EncoderModeInfoBufferTests Assert.Equal(Av1Constants.MaxTransformSize, picture.TransformFunctionContexts[0].Top[0]); Assert.Equal(0, picture.PaletteContexts[0].Left[0].PaletteSizes[0]); Assert.Equal(default, picture.DisplacementVectors.Span[0]); + Assert.Equal(-1, MemoryMarshal.AsBytes(picture.BlockEncodings.Span).IndexOfAnyExcept((byte)0)); + Assert.Equal(-1, MemoryMarshal.AsBytes(picture.BlockPalettes.Span).IndexOfAnyExcept((byte)0)); + Assert.Equal(0, picture.PaletteTokens.Span[0]); + Assert.Equal(-1, MemoryMarshal.AsBytes(picture.ReferenceContexts.Span).IndexOfAnyExcept((byte)0)); Assert.Equal(-1, picture.CdefPreset.Span[0]); Assert.Equal(NextQIndex, picture.Parent.PreviousQIndex.Span[0]); Assert.Equal(0, picture.TileDataOffsets.Span[0]); Assert.Equal(0, picture.TileDataLengths.Span[0]); Assert.Same(nextFrameHeader, picture.Parent.FrameHeader); Assert.Same(nextTiles, picture.Parent.Common.TilesInfo); + + picture.ModeInfoAllocation.Span[0].Block.Mode = Av1PredictionMode.Paeth; + picture.BlockEncodings.Span[0].QuantizationIndex = 53; + picture.BlockPalettes.Span[0].PaletteSizes[0] = 3; + picture.PaletteTokens.Span[0] = 0x42; + picture.ReferenceContexts.Span[0].ModeContext = 37; + picture.LuminanceDcSignLevelCoefficientNeighbors[0].Top[0] = 0x41; + picture.TransformFunctionContexts[0].Left[0] = 4; + picture.ResetEntropyContexts(); + + Assert.Equal(allocationCount, allocator.AllocationLog.Count); + Assert.Empty(allocator.ReturnLog); + Assert.Equal(Av1PredictionMode.Paeth, picture.ModeInfoAllocation.Span[0].Block.Mode); + Assert.Equal(53, picture.BlockEncodings.Span[0].QuantizationIndex); + Assert.Equal(3, picture.BlockPalettes.Span[0].PaletteSizes[0]); + Assert.Equal(0x42, picture.PaletteTokens.Span[0]); + Assert.Equal(37, picture.ReferenceContexts.Span[0].ModeContext); + Assert.Equal(0, picture.LuminanceDcSignLevelCoefficientNeighbors[0].Top[0]); + Assert.Equal(Av1Constants.MaxTransformSize, picture.TransformFunctionContexts[0].Left[0]); } [Theory] diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs index 0fcaba5cdb..00e7a3162f 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraBlockCopyTests.cs @@ -266,7 +266,8 @@ public class Av1IntraBlockCopyTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -275,7 +276,8 @@ public class Av1IntraBlockCopyTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourceLuma = source.Frame.View.GetPlane(Av1Plane.Y); Buffer2DRegion reconstructionLuma = reconstruction.Frame.View.GetPlane(Av1Plane.Y); @@ -352,7 +354,8 @@ public class Av1IntraBlockCopyTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -361,7 +364,8 @@ public class Av1IntraBlockCopyTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourceLuma = source.Frame.View.GetPlane(Av1Plane.Y); Buffer2DRegion reconstructionLuma = reconstruction.Frame.View.GetPlane(Av1Plane.Y); @@ -448,7 +452,8 @@ public class Av1IntraBlockCopyTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -457,7 +462,8 @@ public class Av1IntraBlockCopyTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourceLuma = source.Frame.View.GetPlane(Av1Plane.Y); Buffer2DRegion reconstructionLuma = reconstruction.Frame.View.GetPlane(Av1Plane.Y); @@ -501,6 +507,90 @@ public class Av1IntraBlockCopyTests Assert.Equal(2048, candidates[0].Column); } + /// + /// Checks that high-bit-depth pixel search trades prediction error against motion rate in one common scale. + /// + /// The coded sample precision. + [Theory] + [InlineData(10)] + [InlineData(12)] + public void PixelSearchNormalizesSadBeforeComparingMotionRate(int bits) + { + const int Width = 640; + const int Height = 256; + const int QIndex = 90; + int scale = 1 << (bits - 8); + Point blockOrigin = new(0, 128); + Point predictionOrigin = new(15, 120); + Av1MotionVector reference = new(-64, 120); + ObuSequenceHeader sequenceHeader = CreateSequenceHeader(); + sequenceHeader.ColorConfig.BitDepth = bits == 10 ? Av1BitDepth.TenBit : Av1BitDepth.TwelveBit; + ObuFrameHeader frameHeader = CreateFrameHeader(); + frameHeader.AllowScreenContentTools = true; + frameHeader.AllowIntraBlockCopy = true; + using Av1EncoderPictureBuffer pictureBuffer = new( + Configuration.Default, + sequenceHeader, + frameHeader, + Width, + Height, + disallow4x4AllFrames: true); + + using Av1EncoderFrameBuffer source = new( + Configuration.Default, Width, Height, bits, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); + + using Av1EncoderFrameBuffer reconstruction = new( + Configuration.Default, Width, Height, bits, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); + + Buffer2DRegion sourceLuma = source.Frame.CodedView.GetPlane(Av1Plane.Y); + Buffer2DRegion reconstructedLuma = reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y); + for (int row = 0; row < Height; row++) + { + sourceLuma.DangerousGetRowSpan(row).Clear(); + reconstructedLuma.DangerousGetRowSpan(row).Clear(); + } + + // The reference candidate differs by one eight-bit unit in its first column. Moving right one pixel + // removes that error but adds motion syntax. Raw high-bit-depth SAD would overvalue that small gain. + for (int row = 0; row < 8; row++) + { + sourceLuma.DangerousGetRowSpan(blockOrigin.Y + row).Slice(blockOrigin.X, 8).Fill((ushort)(100 * scale)); + reconstructedLuma.DangerousGetRowSpan(predictionOrigin.Y + row).Slice(predictionOrigin.X, 9).Fill((ushort)(100 * scale)); + reconstructedLuma.DangerousGetRowSpan(predictionOrigin.Y + row)[predictionOrigin.X] = (ushort)(101 * scale); + } + + using Av1SymbolEncoder writer = new(Configuration.Default, 64, QIndex, updateCdf: true); + Span candidates = stackalloc Av1MotionVector[2]; + for (int i = 0; i < 64; i++) + { + // Repeated use of the spatial reference makes a new differential vector appreciably more costly. + writer.WriteDisplacementVector(reference, reference); + } + + int sadPerBit = Av1RateDistortion.GetMotionSearchSadPerBit(QIndex, sequenceHeader.ColorConfig.BitDepth); + int referenceRate = writer.GetDisplacementVectorSearchCost(reference, reference); + int adjacentRate = writer.GetDisplacementVectorSearchCost(new Av1MotionVector(-64, 128), reference); + int referenceMotionCost = ((referenceRate * sadPerBit) + 256) >> 9; + int adjacentMotionCost = ((adjacentRate * sadPerBit) + 256) >> 9; + Assert.True(8 + referenceMotionCost < adjacentMotionCost); + Assert.True((8 * scale) + referenceMotionCost > adjacentMotionCost); + + int count = pictureBuffer.Picture.IntraBlockCopySearch.FindPixelCandidates( + sourceLuma, + reconstructedLuma, + blockOrigin, + new Av1TileInfo(0, 0, frameHeader), + sequenceHeader, + writer, + reference, + QIndex, + Av1RateDistortion.GetKeyFrameRateMultiplier(QIndex, sequenceHeader.ColorConfig.BitDepth), + candidates); + + Assert.Equal(1, count); + Assert.Equal(reference, candidates[0]); + } + /// /// Verifies high-bit-depth SIMD variance normalization against the eight-bit search domain. /// @@ -515,7 +605,8 @@ public class Av1IntraBlockCopyTests 12, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -524,7 +615,8 @@ public class Av1IntraBlockCopyTests 12, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourceLuma = source.Frame.View.GetPlane(Av1Plane.Y); Buffer2DRegion reconstructionLuma = reconstruction.Frame.View.GetPlane(Av1Plane.Y); diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs index 36f25efbb1..b6ab67d5ee 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs @@ -4,8 +4,10 @@ using System.Buffers; using System.Numerics; using System.Runtime.InteropServices; +using SixLabors.ImageSharp.Formats.Heif; using SixLabors.ImageSharp.Formats.Heif.Av1; using SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; +using SixLabors.ImageSharp.Formats.Heif.Av1.Motion; using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; @@ -69,9 +71,9 @@ public class Av1IntraSuperblockEncoderTests }; using Image referenceImage = new(Width, Height); - using Av1EncoderFrameBuffer reference = new(configuration, Width, Height, 8, Av1ColorFormat.Yuv400, 0, 0); - using Av1EncoderFrameBuffer source = new(configuration, Width, Height, 8, Av1ColorFormat.Yuv400, 0, 0); - using Av1EncoderFrameBuffer reconstruction = new(configuration, Width, Height, 8, Av1ColorFormat.Yuv400, 0, 0); + using Av1EncoderFrameBuffer reference = new(configuration, Width, Height, 8, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); + using Av1EncoderFrameBuffer source = new(configuration, Width, Height, 8, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); + using Av1EncoderFrameBuffer reconstruction = new(configuration, Width, Height, 8, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); for (int y = 0; y < Height; y++) { Span pixels = referenceImage.Frames.RootFrame.PixelBuffer.DangerousGetRowSpan(y); @@ -98,7 +100,8 @@ public class Av1IntraSuperblockEncoderTests Height, colorConfig, qIndex: 0, - effort); + effort, + speed: HeifEncodingSpeed.Level0); keyEncoder.EncodeKeyFrame(referenceImage.Frames.RootFrame, firstSample); ObuSequenceHeader sequenceHeader = keyEncoder.SequenceHeader; @@ -125,7 +128,7 @@ public class Av1IntraSuperblockEncoderTests using Av1EncoderPictureBuffer picture = new(configuration, sequenceHeader, frameHeader, Width, Height, disallow4x4AllFrames: true); using Av1EncoderCoefficientBuffer coefficients = new(configuration, sequenceHeader, Width, Height); using Av1EncoderSuperblockWorkspace superblockWorkspace = new(configuration); - using Av1EncoderBlockWorkspace blockWorkspace = new(configuration); + using Av1EncoderBlockWorkspace blockWorkspace = new(configuration, allocateInterMotionCosts: true); using Av1SymbolEncoder symbolEncoder = new(configuration, TileBufferLength, QIndex, updateCdf: true); Av1EncoderTileWorkspace tileWorkspace = new(frameHeader, superblockWorkspace); int allocationCount = allocator.AllocationLog.Count; @@ -240,9 +243,9 @@ public class Av1IntraSuperblockEncoderTests }; using Image referenceImage = new(Width, Height); - using Av1EncoderFrameBuffer reference = new(configuration, Width, Height, bitDepth, Av1ColorFormat.Yuv400, 0, 0); - using Av1EncoderFrameBuffer source = new(configuration, Width, Height, bitDepth, Av1ColorFormat.Yuv400, 0, 0); - using Av1EncoderFrameBuffer reconstruction = new(configuration, Width, Height, bitDepth, Av1ColorFormat.Yuv400, 0, 0); + using Av1EncoderFrameBuffer reference = new(configuration, Width, Height, bitDepth, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); + using Av1EncoderFrameBuffer source = new(configuration, Width, Height, bitDepth, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); + using Av1EncoderFrameBuffer reconstruction = new(configuration, Width, Height, bitDepth, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); for (int y = 0; y < Height; y++) { Span pixels = referenceImage.Frames.RootFrame.PixelBuffer.DangerousGetRowSpan(y); @@ -266,7 +269,13 @@ public class Av1IntraSuperblockEncoderTests ClearPlane(reconstruction.Luma); using MemoryStream firstSample = new(); using Av1FrameEncoder.SequenceEncoder keyEncoder = Av1FrameEncoder.CreateColorSequenceEncoder( - configuration, Width, Height, colorConfig, qIndex: 0, Effort); + configuration, + Width, + Height, + colorConfig, + qIndex: 0, + Effort, + speed: HeifEncodingSpeed.Level0); keyEncoder.EncodeKeyFrame(referenceImage.Frames.RootFrame, firstSample); ObuSequenceHeader sequenceHeader = keyEncoder.SequenceHeader; @@ -293,7 +302,7 @@ public class Av1IntraSuperblockEncoderTests using Av1EncoderPictureBuffer picture = new(configuration, sequenceHeader, frameHeader, Width, Height, disallow4x4AllFrames: true); using Av1EncoderCoefficientBuffer coefficients = new(configuration, sequenceHeader, Width, Height); using Av1EncoderSuperblockWorkspace superblockWorkspace = new(configuration); - using Av1EncoderBlockWorkspace blockWorkspace = new(configuration); + using Av1EncoderBlockWorkspace blockWorkspace = new(configuration, allocateInterMotionCosts: true); using Av1SymbolEncoder symbolEncoder = new(configuration, TileBufferLength, QIndex, updateCdf: true); Av1EncoderTileWorkspace tileWorkspace = new(frameHeader, superblockWorkspace); int allocationCount = allocator.AllocationLog.Count; @@ -374,7 +383,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv420, 1, - 1); + 1, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -383,7 +393,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv420, 1, - 1); + 1, + lumaBorder: 64); FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.Y), (byte)128); FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.U), (byte)128); @@ -560,7 +571,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv420, 1, - 1); + 1, + lumaBorder: 64); ClearPlane(tileReconstruction.Luma); ClearPlane(Assert.IsType>(tileReconstruction.ChromaBlue)); @@ -599,6 +611,369 @@ public class Av1IntraSuperblockEncoderTests Assert.True(encoded.GetSpan().SequenceEqual(tileWriter.GetTileData(0))); } + [Theory] + [InlineData(8)] + [InlineData(10)] + [InlineData(12)] + public void InterBlockCanSkipNonzeroQuantizedResiduals(int bitDepthValue) + { + if (bitDepthValue == 8) + { + VerifyReferenceBlockCanSkipNonzeroQuantizedResiduals( + Av1BitDepth.EightBit, + bitDepthValue, + isIntraBlockCopy: false, + static value => (byte)value); + } + else + { + VerifyReferenceBlockCanSkipNonzeroQuantizedResiduals( + bitDepthValue == 10 ? Av1BitDepth.TenBit : Av1BitDepth.TwelveBit, + bitDepthValue, + isIntraBlockCopy: false, + static value => (ushort)value); + } + } + + [Theory] + [InlineData(8)] + [InlineData(10)] + [InlineData(12)] + public void IntraBlockCopyCanSkipNonzeroQuantizedResiduals(int bitDepthValue) + { + if (bitDepthValue == 8) + { + VerifyReferenceBlockCanSkipNonzeroQuantizedResiduals( + Av1BitDepth.EightBit, + bitDepthValue, + isIntraBlockCopy: true, + static value => (byte)value); + } + else + { + VerifyReferenceBlockCanSkipNonzeroQuantizedResiduals( + bitDepthValue == 10 ? Av1BitDepth.TenBit : Av1BitDepth.TwelveBit, + bitDepthValue, + isIntraBlockCopy: true, + static value => (ushort)value); + } + } + + private static void VerifyReferenceBlockCanSkipNonzeroQuantizedResiduals( + Av1BitDepth bitDepth, + int bitDepthValue, + bool isIntraBlockCopy, + SampleFactory createSample) + where TSample : unmanaged + where TOperator : struct, Av1IntraSuperblockEncoder.IBlockEncodingOperator + { + const int Width = 8; + const int Height = 8; + Point blockOrigin = new(isIntraBlockCopy ? 320 : 0, 0); + Point modeInfoPosition = blockOrigin >> Av1Constants.ModeInfoSizeLog2; + int frameWidth = blockOrigin.X + Width; + int superblockIndex = blockOrigin.X / 64; + ObuColorConfig colorConfig = new() + { + IsMonochrome = true, + SubSamplingX = true, + SubSamplingY = true, + BitDepth = bitDepth + }; + + using Av1EncoderFrameBuffer source = new( + Configuration.Default, + frameWidth, + Height, + bitDepthValue, + Av1ColorFormat.Yuv400, + 0, + 0, + lumaBorder: 64); + + using Av1EncoderFrameBuffer reference = new( + Configuration.Default, + frameWidth, + Height, + bitDepthValue, + Av1ColorFormat.Yuv400, + 0, + 0, + lumaBorder: 64); + + using Av1EncoderFrameBuffer reconstruction = new( + Configuration.Default, + frameWidth, + Height, + bitDepthValue, + Av1ColorFormat.Yuv400, + 0, + 0, + lumaBorder: 64); + + TSample[] prediction = new TSample[Width * Height]; + short[] residual = new short[Width * Height]; + TSample[] trialReconstruction = new TSample[Width * Height]; + int[] trialCoefficients = new int[Width * Height]; + int verifiedCases = 0; + for (int qIndex = 64; qIndex <= 192; qIndex += 32) + { + for (int amplitude = 2; amplitude <= 12; amplitude += 2) + { + // Independent texture and small, signed perturbations distinguish temporal prediction from spatial DC. + // Samples remain inside the coded range. Unaligned high-depth perturbations exercise SSE rounding. + long squaredError = 0; + for (int y = 0; y < Height; y++) + { + Span sourceRow = source.Frame.CodedView.GetPlane(Av1Plane.Y).DangerousGetRowSpan(y); + Span referenceRow = reference.Frame.CodedView.GetPlane(Av1Plane.Y).DangerousGetRowSpan(y); + if (isIntraBlockCopy) + { + // Completed blocks provide a repeated reconstructed reference before the current superblock. + // The target differs from it, so pixel search must supply the candidate without an exact hash match. + Span reconstructionRow = reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y).DangerousGetRowSpan(y); + for (int x = 0; x < frameWidth; x++) + { + int phase = x % Width; + int sample = 64 + (((phase * 37) + (y * 53) + (phase * y * 19)) % 128); + sourceRow[x] = reconstructionRow[x] = createSample(sample << (bitDepthValue - 8)); + } + } + + for (int x = 0; x < Width; x++) + { + int index = (y * Width) + x; + int sample = 64 + (((x * 37) + (y * 53) + (x * y * 19)) % 128); + int difference = (((x * 13) + (y * 7) + (x * y * 3)) % ((amplitude * 2) + 1)) - amplitude; + int precisionShift = bitDepthValue - 8; + sample <<= precisionShift; + difference = (difference << precisionShift) + (precisionShift > 0 && index % 3 == 0 ? 1 : 0); + prediction[index] = referenceRow[x] = createSample(sample); + sourceRow[blockOrigin.X + x] = createSample(sample + difference); + residual[index] = (short)difference; + squaredError += difference * difference; + } + } + + source.Frame.ExtendBorders(); + reference.Frame.ExtendBorders(); + using Av1EncoderModeInfoBuffer modeInfoBuffer = new(Configuration.Default, frameWidth, Height, disallow4x4AllFrames: true); + Av1PictureControlSet template = CreatePicture(modeInfoBuffer, colorConfig, use128x128Superblock: false, qIndex); + template.Parent.FrameHeader.FrameType = isIntraBlockCopy ? ObuFrameType.KeyFrame : ObuFrameType.InterFrame; + template.Parent.FrameHeader.AllowIntraBlockCopy = isIntraBlockCopy; + template.Parent.FrameHeader.AllowScreenContentTools = isIntraBlockCopy; + template.Parent.FrameHeader.FrameSize.FrameWidth = frameWidth; + template.Parent.FrameHeader.FrameSize.FrameHeight = Height; + template.Parent.FrameHeader.TransformMode = Av1TransformMode.Largest; + template.Parent.FrameHeader.InterpolationFilter = Av1InterpolationFilter.Regular; + template.Parent.FrameHeader.ReferenceMode = ObuReferenceMode.SingleReference; + using Av1EncoderPictureBuffer pictureBuffer = new( + Configuration.Default, + template.Sequence.SequenceHeader, + template.Parent.FrameHeader, + frameWidth, + Height, + disallow4x4AllFrames: true); + + Av1PictureControlSet picture = pictureBuffer.Picture; + if (isIntraBlockCopy) + { + picture.IntraBlockCopySearch.Initialize(source.Frame.View.GetPlane(Av1Plane.Y)); + } + + using Av1EncoderCoefficientBuffer coefficients = new(Configuration.Default, template.Sequence.SequenceHeader, frameWidth, Height); + using Av1EncoderSuperblockWorkspace superblockWorkspace = new(Configuration.Default); + using Av1EncoderBlockWorkspace blockWorkspace = new(Configuration.Default, allocateInterMotionCosts: !isIntraBlockCopy); + using Av1SymbolEncoder writer = new(Configuration.Default, 4096, qIndex, updateCdf: true); + if (!isIntraBlockCopy) + { + writer.FillMotionVectorCosts(blockWorkspace.GetMotionVectorCosts(picture.Parent.FrameHeader.MotionVectorPrecision)); + } + + Av1Superblock superblock = new() + { + Workspace = superblockWorkspace, + TileInfo = new Av1TileInfo(0, 0, picture.Parent.FrameHeader), + Index = superblockIndex + }; + + Av1IntraSuperblockEncoder.Prepare(picture, superblock, blockOrigin); + picture.MapModeInfoBlock(modeInfoPosition, Av1BlockSize.Block8x8); + Av1MacroBlockD macroBlock = new() { Tile = superblock.TileInfo }; + Av1TileWriter.SetModeInfoRowAndColumn( + picture, + macroBlock, + superblock.TileInfo, + modeInfoPosition, + Av1BlockSize.Block8x8, + picture.Parent.Common.ModeInfoStride, + picture.Parent.Common.ModeInfoRowCount, + picture.Parent.Common.ModeInfoColumnCount); + + if (!isIntraBlockCopy) + { + picture.Parent.MotionSearchSettings = new Av1MotionSearchSettings( + HeifEncodingSpeed.Level0, false, new Size(frameWidth, Height), qIndex, false, false); + + picture.Parent.MotionSearchStepParameter = Av1MotionSearchBase.GetInitialStepParameter(Math.Max(frameWidth, Height)); + + ref Av1ReferenceMotionVectors references = ref blockWorkspace.ReferenceMotionVectors; + references.Build( + picture, + macroBlock, + modeInfoPosition, + Av1BlockSize.Block8x8, + Av1PartitionType.None, + template.Sequence.SequenceHeader, + picture.Parent.FrameHeader, + Av1ReferenceFrameType.Last); + + // Thirty-two preceding global-motion symbols make the zero global predictor the cheapest + // mode. Keep every competing mode enabled so this fixture tests residual skipping independently + // of mode-search restrictions, while checking exact syntax rates, distortion, and reconstruction. + for (int index = 0; index < 32; index++) + { + writer.WriteInterMode(Av1PredictionMode.GlobalMotionVector, references.ModeContext); + } + + int globalRate = writer.GetInterModeCost(Av1PredictionMode.GlobalMotionVector, references.ModeContext); + Assert.True(globalRate < writer.GetInterModeCost(Av1PredictionMode.NearestMotionVector, references.ModeContext)); + Assert.True(globalRate < writer.GetInterModeCost(Av1PredictionMode.NearMotionVector, references.ModeContext)); + Assert.True(globalRate < writer.GetInterModeCost(Av1PredictionMode.NewMotionVector, references.ModeContext)); + } + + int multiplier = isIntraBlockCopy + ? Av1RateDistortion.GetKeyFrameRateMultiplier(qIndex, bitDepth) + : Av1RateDistortion.GetInterFrameRateMultiplier(qIndex, bitDepth); + int skipContext = Av1TileWriter.GetSkipContext(macroBlock); + int skipRate = writer.GetSkipCost(true, skipContext); + int squaredPrecisionScale = 1 << ((bitDepthValue - 8) * 2); + long expectedDistortion = ((squaredError + (squaredPrecisionScale / 2)) / squaredPrecisionScale) * 16; + long skipCost = ((((long)skipRate * multiplier) + 256) / 512) + (expectedDistortion * 128); + long bestCodedCost = long.MaxValue; + bool allTransformsAreNonzero = true; + Av1TransformSetType transformSet = Av1SymbolContextHelper.GetExtendedTransformSetType( + Av1TransformSize.Size8x8, + isInter: true, + picture.Parent.FrameHeader.UseReducedTransformSet); + + // Evaluate residual coding independently of the block decision. Qualifying cases must have + // nonzero coefficients in every legal transform, yet cost more than prediction-only reconstruction. + for (Av1TransformType transformType = Av1TransformType.DctDct; + transformType < Av1TransformType.AllTransformTypes; + transformType++) + { + if (!transformType.IsExtendedSetUsed(transformSet)) + { + continue; + } + + Av1EncoderTransformBlockState state = default; + long distortion = TOperator.EncodePredictionCandidate( + blockWorkspace, + source.Frame.CodedView.GetPlane(Av1Plane.Y), + blockOrigin, + prediction, + residual, + trialReconstruction, + Width, + trialCoefficients, + Av1TransformSize.Size8x8, + transformType, + Av1Plane.Y, + qIndex, + 0, + 0, + bitDepth, + ref state); + + allTransformsAreNonzero &= state.EndOfBlock != 0; + int rate = writer.GetSkipCost(false, skipContext) + writer.GetCoefficientCost( + Av1TransformSize.Size8x8, + transformType, + isIntraBlockCopy ? Av1PredictionMode.DC : Av1PredictionMode.GlobalMotionVector, + trialCoefficients, + Av1ComponentType.Luminance, + default, + state.EndOfBlock, + picture.Parent.FrameHeader.UseReducedTransformSet, + Av1FilterIntraMode.AllFilterIntraModes, + usesInterTransformSet: true); + + long cost = ((((long)rate * multiplier) + 256) / 512) + (distortion * 128); + bestCodedCost = Math.Min(bestCodedCost, cost); + } + + if (!allTransformsAreNonzero || skipCost > bestCodedCost) + { + continue; + } + + Av1IntraSuperblockEncoder.ModeDecision decision = new( + source.Frame, + reference.Frame, + reconstruction.Frame, + picture, + superblock, + coefficients, + blockWorkspace, + effort: 0); + + ref Av1MacroBlockModeInfo modeInfo = ref picture.GetMacroBlockModeInfo(modeInfoPosition); + Av1EncoderBlockStruct block = default; + Av1EncoderPaletteInfo palette = default; + Av1MotionVector displacementReference = default; + if (isIntraBlockCopy) + { + displacementReference = Av1IntraBlockCopy.FindReference( + picture, + macroBlock, + modeInfoPosition, + Av1BlockSize.Block8x8, + Av1PartitionType.None, + new Av1MotionVector[8], + new int[8]); + } + + decision.EncodeBlock(writer, macroBlock, blockOrigin, 0, ref modeInfo, ref block, ref palette); + Assert.Equal(isIntraBlockCopy, modeInfo.Block.UseIntraBlockCopy); + if (!isIntraBlockCopy) + { + Assert.Equal(Av1ReferenceFrameType.Last, modeInfo.Block.ReferenceFrame); + } + + Assert.True(modeInfo.Block.Skip, $"qIndex={qIndex}, amplitude={amplitude}, skipCost={skipCost}, codedCost={bestCodedCost}"); + Assert.Equal(isIntraBlockCopy ? Av1PredictionMode.DC : Av1PredictionMode.GlobalMotionVector, modeInfo.Block.Mode); + Assert.Equal(expectedDistortion, decision.SelectedBlockStatistics.Distortion); + int expectedRate = skipRate + (isIntraBlockCopy + ? writer.GetUseIntraBlockCopyCost(true) + + writer.GetDisplacementVectorCost(picture.GetDisplacementVector(modeInfoPosition), displacementReference) + : writer.GetIsInterCost(true, Av1TileWriter.GetIntraInterContext(macroBlock)) + + writer.GetSingleReferenceCost(Av1ReferenceFrameType.Last, new byte[Av1Constants.ReferenceFrameCount]) + + writer.GetInterModeCost(Av1PredictionMode.GlobalMotionVector, blockWorkspace.ReferenceMotionVectors.ModeContext)); + + Assert.Equal(expectedRate, decision.SelectedBlockStatistics.Rate); + Assert.Equal( + ((((long)expectedRate * multiplier) + 256) / 512) + (expectedDistortion * 128), + decision.SelectedBlockStatistics.Cost); + Assert.Equal((ushort)0, coefficients.GetTransformBlockSpan(superblockIndex, Av1Plane.Y)[0].EndOfBlock); + Assert.Equal(Av1TransformType.DctDct, coefficients.GetTransformBlockSpan(superblockIndex, Av1Plane.Y)[0].TransformType); + Assert.Equal((byte)0, coefficients.GetTransformBlockSpan(superblockIndex, Av1Plane.Y)[0].EntropyContext); + Assert.All(coefficients.GetPlaneSpan(superblockIndex, Av1Plane.Y)[..64].ToArray(), value => Assert.Equal(0, value)); + for (int y = 0; y < Height; y++) + { + Assert.Equal( + MemoryMarshal.AsBytes(prediction.AsSpan(y * Width, Width)), + MemoryMarshal.AsBytes(reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y).DangerousGetRowSpan(y).Slice(blockOrigin.X, Width))); + } + + verifiedCases++; + } + } + + Assert.True(verifiedCases > 0, "The input set must exercise skipping despite nonzero coefficients in every legal transform."); + } + [Theory] [InlineData(true)] [InlineData(false)] @@ -622,7 +997,8 @@ public class Av1IntraSuperblockEncoderTests 8, colorFormat, 1, - 1); + 1, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -631,7 +1007,8 @@ public class Av1IntraSuperblockEncoderTests 8, colorFormat, 1, - 1); + 1, + lumaBorder: 64); FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.Y), (byte)128); ClearPlane(reconstruction.Luma); @@ -713,7 +1090,8 @@ public class Av1IntraSuperblockEncoderTests 8, colorFormat, 1, - 1); + 1, + lumaBorder: 64); ClearPlane(liveReconstruction.Luma); if (!isMonochrome) @@ -775,7 +1153,8 @@ public class Av1IntraSuperblockEncoderTests 12, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -784,7 +1163,8 @@ public class Av1IntraSuperblockEncoderTests 12, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourcePlane = source.Frame.CodedView.GetPlane(Av1Plane.Y); for (int y = 0; y < sourcePlane.Height; y++) @@ -847,7 +1227,8 @@ public class Av1IntraSuperblockEncoderTests 12, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); ClearPlane(tileReconstruction.Luma); using Av1EncoderPictureBuffer tilePicture = new( @@ -907,8 +1288,8 @@ public class Av1IntraSuperblockEncoderTests BitDepth = Av1BitDepth.EightBit }; - using Av1EncoderFrameBuffer source = new(Configuration.Default, Width, Height, 8, colorFormat, 1, 1); - using Av1EncoderFrameBuffer reconstruction = new(Configuration.Default, Width, Height, 8, colorFormat, 1, 1); + using Av1EncoderFrameBuffer source = new(Configuration.Default, Width, Height, 8, colorFormat, 1, 1, lumaBorder: 64); + using Av1EncoderFrameBuffer reconstruction = new(Configuration.Default, Width, Height, 8, colorFormat, 1, 1, lumaBorder: 64); int planeCount = isMonochrome ? 1 : 3; for (int planeIndex = 0; planeIndex < planeCount; planeIndex++) { @@ -1422,7 +1803,8 @@ public class Av1IntraSuperblockEncoderTests 8, colorFormat, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -1431,7 +1813,8 @@ public class Av1IntraSuperblockEncoderTests 8, colorFormat, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion lumaSource = source.Frame.CodedView.GetPlane(Av1Plane.Y); Buffer2DRegion blueSource = source.Frame.CodedView.GetPlane(Av1Plane.U); @@ -1657,7 +2040,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -1666,7 +2050,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourcePlane = source.Frame.CodedView.GetPlane(Av1Plane.Y); @@ -1823,7 +2208,8 @@ public class Av1IntraSuperblockEncoderTests 8, colorFormat, chromaSubsamplingX, - chromaSubsamplingY); + chromaSubsamplingY, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -1832,7 +2218,8 @@ public class Av1IntraSuperblockEncoderTests 8, colorFormat, chromaSubsamplingX, - chromaSubsamplingY); + chromaSubsamplingY, + lumaBorder: 64); FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.Y), (byte)128); FillChromaModeSelectionPlane( @@ -1986,7 +2373,8 @@ public class Av1IntraSuperblockEncoderTests bitDepth, colorFormat, chromaSubsamplingX, - chromaSubsamplingY); + chromaSubsamplingY, + lumaBorder: 64); using Av1EncoderFrameBuffer pilotReconstruction = new( Configuration.Default, @@ -1995,7 +2383,8 @@ public class Av1IntraSuperblockEncoderTests bitDepth, colorFormat, chromaSubsamplingX, - chromaSubsamplingY); + chromaSubsamplingY, + lumaBorder: 64); Buffer2DRegion pilotLuma = pilotSource.Frame.CodedView.GetPlane(Av1Plane.Y); for (int y = 0; y < pilotLuma.Height; y++) @@ -2081,7 +2470,8 @@ public class Av1IntraSuperblockEncoderTests bitDepth, colorFormat, chromaSubsamplingX, - chromaSubsamplingY); + chromaSubsamplingY, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -2090,7 +2480,8 @@ public class Av1IntraSuperblockEncoderTests bitDepth, colorFormat, chromaSubsamplingX, - chromaSubsamplingY); + chromaSubsamplingY, + lumaBorder: 64); for (int y = 0; y < pilotLuma.Height; y++) { @@ -2335,7 +2726,8 @@ public class Av1IntraSuperblockEncoderTests bitDepth, Av1ColorFormat.Yuv400, 1, - 1); + 1, + lumaBorder: 64); using Av1EncoderFrameBuffer pilotReconstruction = new( Configuration.Default, @@ -2344,7 +2736,8 @@ public class Av1IntraSuperblockEncoderTests bitDepth, Av1ColorFormat.Yuv400, 1, - 1); + 1, + lumaBorder: 64); Buffer2DRegion pilotLuma = pilotSource.Frame.CodedView.GetPlane(Av1Plane.Y); for (int y = 0; y < pilotLuma.Height; y++) @@ -2497,7 +2890,8 @@ public class Av1IntraSuperblockEncoderTests bitDepth, Av1ColorFormat.Yuv400, 1, - 1); + 1, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -2506,7 +2900,8 @@ public class Av1IntraSuperblockEncoderTests bitDepth, Av1ColorFormat.Yuv400, 1, - 1); + 1, + lumaBorder: 64); Buffer2DRegion sourceLuma = source.Frame.CodedView.GetPlane(Av1Plane.Y); for (int y = 0; y < pilotLuma.Height; y++) @@ -2659,7 +3054,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -2668,7 +3064,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourcePlane = source.Frame.CodedView.GetPlane(Av1Plane.Y); FillPlane(sourcePlane, (byte)128); @@ -2830,7 +3227,8 @@ public class Av1IntraSuperblockEncoderTests bitDepthValue, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -2839,7 +3237,8 @@ public class Av1IntraSuperblockEncoderTests bitDepthValue, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourcePlane = source.Frame.CodedView.GetPlane(Av1Plane.Y); for (int row = 0; row < Height; row++) @@ -2941,7 +3340,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv420, 1, - 1); + 1, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -2950,7 +3350,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv420, 1, - 1); + 1, + lumaBorder: 64); Buffer2DRegion lumaSource = source.Frame.CodedView.GetPlane(Av1Plane.Y); for (int row = 0; row < Height; row++) @@ -3110,7 +3511,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -3119,7 +3521,8 @@ public class Av1IntraSuperblockEncoderTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); FillPlane(source.Frame.CodedView.GetPlane(Av1Plane.Y), 251, 29); ClearPlane(reconstruction.Luma); @@ -3201,8 +3604,8 @@ public class Av1IntraSuperblockEncoderTests BitDepth = Av1BitDepth.EightBit }; - using Av1EncoderFrameBuffer source = new(Configuration.Default, size, size, 8, Av1ColorFormat.Yuv400, 0, 0); - using Av1EncoderFrameBuffer reconstruction = new(Configuration.Default, size, size, 8, Av1ColorFormat.Yuv400, 0, 0); + using Av1EncoderFrameBuffer source = new(Configuration.Default, size, size, 8, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); + using Av1EncoderFrameBuffer reconstruction = new(Configuration.Default, size, size, 8, Av1ColorFormat.Yuv400, 0, 0, lumaBorder: 64); Buffer2DRegion sourcePlane = source.Frame.CodedView.GetPlane(Av1Plane.Y); for (int y = 0; y < size; y++) { @@ -3274,6 +3677,143 @@ public class Av1IntraSuperblockEncoderTests } } + [Theory] + [InlineData((int)Av1ColorFormat.Yuv400)] + [InlineData((int)Av1ColorFormat.Yuv420)] + [InlineData((int)Av1ColorFormat.Yuv422)] + [InlineData((int)Av1ColorFormat.Yuv444)] + public void ProductionDeblockingPreservesEightBitReconstruction(int colorFormatValue) + => VerifyProductionDeblocking( + colorFormatValue, + 8, + static (writer, source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new(writer, source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace, effort: 5)); + + [Theory] + [InlineData((int)Av1ColorFormat.Yuv400, 10)] + [InlineData((int)Av1ColorFormat.Yuv400, 12)] + [InlineData((int)Av1ColorFormat.Yuv420, 10)] + [InlineData((int)Av1ColorFormat.Yuv420, 12)] + [InlineData((int)Av1ColorFormat.Yuv422, 10)] + [InlineData((int)Av1ColorFormat.Yuv422, 12)] + [InlineData((int)Av1ColorFormat.Yuv444, 10)] + [InlineData((int)Av1ColorFormat.Yuv444, 12)] + public void ProductionDeblockingPreservesHighBitDepthReconstruction(int colorFormatValue, int bitDepth) + => VerifyProductionDeblocking( + colorFormatValue, + bitDepth, + static (writer, source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace) => + new(writer, source, reconstruction, picture, coefficients, superblockWorkspace, blockWorkspace, effort: 5)); + + /// + /// Verifies retained reconstruction and exports the same encoded stream for an independent decoder comparison. + /// + /// The component sample type. + /// The component layout. + /// The component precision. + /// The typed production tile constructor. + private static void VerifyProductionDeblocking(int colorFormatValue, int bitDepth, TileWriterFactory createWriter) + where TSample : unmanaged, IBinaryInteger + { + const int Width = 33; + const int Height = 137; + const int QIndex = 37; + Av1ColorFormat colorFormat = (Av1ColorFormat)colorFormatValue; + ObuColorConfig colorConfig = new() + { + IsMonochrome = colorFormat == Av1ColorFormat.Yuv400, + SubSamplingX = colorFormat is Av1ColorFormat.Yuv400 or Av1ColorFormat.Yuv420 or Av1ColorFormat.Yuv422, + SubSamplingY = colorFormat is Av1ColorFormat.Yuv400 or Av1ColorFormat.Yuv420, + BitDepth = bitDepth == 8 ? Av1BitDepth.EightBit : bitDepth == 10 ? Av1BitDepth.TenBit : Av1BitDepth.TwelveBit + }; + + using Av1EncoderFrameBuffer source = new(Configuration.Default, Width, Height, bitDepth, colorFormat, 1, 1, lumaBorder: 64); + using Av1EncoderFrameBuffer reconstruction = new(Configuration.Default, Width, Height, bitDepth, colorFormat, 1, 1, lumaBorder: 64); + int planeCount = colorConfig.PlaneCount; + TSample[][] unfiltered = new TSample[planeCount][]; + for (int planeIndex = 0; planeIndex < planeCount; planeIndex++) + { + Buffer2DRegion plane = source.Frame.View.GetPlane((Av1Plane)planeIndex); + unfiltered[planeIndex] = new TSample[plane.Width * plane.Height]; + for (int y = 0; y < plane.Height; y++) + { + Span row = plane.DangerousGetRowSpan(y); + for (int x = 0; x < row.Length; x++) + { + // Small discontinuities at coding boundaries activate deblocking. Odd dimensions and a + // height above 128 exercise chroma ownership, coded padding, and intersecting row bands. + int value = 96 + (4 * ((x / 8) + (y / 8))) + (planeIndex * 8); + row[x] = TSample.CreateChecked(value << (bitDepth - 8)); + } + } + } + + source.Frame.ExtendBorders(); + using Av1EncoderModeInfoBuffer modeInfo = new(Configuration.Default, Width, Height, disallow4x4AllFrames: true); + Av1PictureControlSet template = CreatePicture(modeInfo, colorConfig, use128x128Superblock: false, QIndex); + ObuFrameHeader header = template.Parent.FrameHeader; + header.FrameSize.FrameWidth = Width; + header.FrameSize.FrameHeight = Height; + using Av1EncoderPictureBuffer picture = new( + Configuration.Default, template.Sequence.SequenceHeader, header, Width, Height, disallow4x4AllFrames: true); + + using Av1EncoderCoefficientBuffer coefficients = new(Configuration.Default, template.Sequence.SequenceHeader, Width, Height); + using Av1EncoderSuperblockWorkspace superblockWorkspace = new(Configuration.Default); + using Av1EncoderBlockWorkspace blockWorkspace = new(Configuration.Default); + using Av1SymbolEncoder symbolEncoder = CreateTileSymbolEncoder(picture.Picture, 8192); + _ = createWriter(symbolEncoder, source.Frame, reconstruction.Frame, picture.Picture, coefficients, superblockWorkspace, blockWorkspace); + for (int planeIndex = 0; planeIndex < planeCount; planeIndex++) + { + Buffer2DRegion plane = reconstruction.Frame.View.GetPlane((Av1Plane)planeIndex); + for (int y = 0; y < plane.Height; y++) + { + plane.DangerousGetRowSpan(y).CopyTo(unfiltered[planeIndex].AsSpan(y * plane.Width, plane.Width)); + } + } + + header.LoopFilterParameters.FilterLevel[0] = 63; + header.LoopFilterParameters.FilterLevel[1] = 37; + header.LoopFilterParameters.FilterLevelU = 31; + header.LoopFilterParameters.FilterLevelV = 47; + header.LoopFilterParameters.SharpnessLevel = 3; + header.LoopFilterParameters.ReferenceDeltaModeEnabled = true; + picture.Reset(header); + Av1TileEncoder tileWriter = createWriter( + symbolEncoder, source.Frame, reconstruction.Frame, picture.Picture, coefficients, superblockWorkspace, blockWorkspace); + + byte[] payload = WriteCompleteTileObu(picture.Picture, tileWriter, Width, Height); + using Av1Decoder decoder = new(Configuration.Default); + using Av1FrameBuffer decodedFrame = decoder.DecodeFrameBuffer(payload, null, null, out _); + int changedSamples = 0; + string directory = Path.Combine(TestEnvironment.ActualOutputDirectoryFullPath, "Heif", "Av1", "ProductionDeblocking"); + Directory.CreateDirectory(directory); + string name = $"{bitDepth}-{colorFormat}"; + File.WriteAllBytes(Path.Combine(directory, name + ".obu"), payload); + using FileStream raw = File.Create(Path.Combine(directory, name + ".retained.yuv")); + for (int planeIndex = 0; planeIndex < planeCount; planeIndex++) + { + Av1Plane plane = (Av1Plane)planeIndex; + Buffer2DRegion retained = reconstruction.Frame.View.GetPlane(plane); + int subX = plane == Av1Plane.Y ? 0 : reconstruction.Frame.ChromaSubsamplingX; + int subY = plane == Av1Plane.Y ? 0 : reconstruction.Frame.ChromaSubsamplingY; + Buffer2DRegion decoded = decodedFrame.DeriveBlockPointer(plane, subX, subY); + for (int y = 0; y < retained.Height; y++) + { + ReadOnlySpan row = retained.DangerousGetRowSpan(y); + ReadOnlySpan decodedRow = MemoryMarshal.Cast(decoded.DangerousGetRowSpan(y)); + Assert.Equal(row, decodedRow); + for (int x = 0; x < row.Length; x++) + { + changedSamples += row[x] != unfiltered[planeIndex][(y * retained.Width) + x] ? 1 : 0; + } + + raw.Write(MemoryMarshal.AsBytes(row)); + } + } + + Assert.True(changedSamples > 0); + } + private static byte[] WriteCompleteTileObu( Av1PictureControlSet pictureTemplate, IAv1TileWriter tileWriter, @@ -3421,7 +3961,8 @@ public class Av1IntraSuperblockEncoderTests bitDepthValue, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstruction = new( Configuration.Default, @@ -3430,7 +3971,8 @@ public class Av1IntraSuperblockEncoderTests bitDepthValue, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); Buffer2DRegion sourcePlane = source.Frame.CodedView.GetPlane(Av1Plane.Y); for (int row = 0; row < sourcePlane.Height; row++) @@ -3815,6 +4357,8 @@ public class Av1IntraSuperblockEncoderTests this.Count = 0; } + public static bool UsesRetainedDecisions => false; + /// /// Gets the number of final blocks visited by the writer. /// @@ -3891,6 +4435,8 @@ public class Av1IntraSuperblockEncoderTests this.Count = 0; } + public static bool UsesRetainedDecisions => false; + /// /// Gets the number of final blocks visited by the writer. /// diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TransformBlockEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TransformBlockEncoderTests.cs index 5d5beede63..cf7c36fb37 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TransformBlockEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TransformBlockEncoderTests.cs @@ -59,7 +59,8 @@ public class Av1TransformBlockEncoderTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); using Av1EncoderFrameBuffer reconstructionFrame = new( Configuration.Default, @@ -68,7 +69,8 @@ public class Av1TransformBlockEncoderTests 8, Av1ColorFormat.Yuv400, 0, - 0); + 0, + lumaBorder: 64); reconstructionFrame.Luma.DangerousGetSingleSpan().Fill(PaddingSentinel); Buffer2DRegion sourcePlane = sourceFrame.Frame.CodedView.GetPlane(Av1Plane.Y); @@ -596,8 +598,10 @@ public class Av1TransformBlockEncoderTests /// /// Verifies that the block workspace uses one exact-size allocator owner and returns it exactly once. /// - [Fact] - public void BlockWorkspaceUsesOneExactSizeOwner() + [Theory] + [InlineData(false, 0)] + [InlineData(true, 135836)] + public void BlockWorkspaceUsesOneExactSizeOwner(bool allocateInterMotionCosts, int additionalLength) { TestMemoryAllocator allocator = new(); allocator.EnableNonThreadSafeLogging(); @@ -605,12 +609,12 @@ public class Av1TransformBlockEncoderTests configuration.MemoryAllocator = allocator; TestMemoryAllocator.AllocationRequest allocation; - using (Av1EncoderBlockWorkspace workspace = new(configuration)) + using (Av1EncoderBlockWorkspace workspace = new(configuration, allocateInterMotionCosts)) { allocation = Assert.Single(allocator.AllocationLog); Assert.Empty(allocator.ReturnLog); Assert.Equal(typeof(int), allocation.ElementType); - Assert.Equal(Av1EncoderBlockWorkspace.StorageLength, allocation.Length); + Assert.Equal(Av1EncoderBlockWorkspace.StorageLength + additionalLength, allocation.Length); Assert.Equal(Av1EncoderBlockWorkspace.MaximumResidualCount, workspace.Residual.Length); Assert.Equal(Av1EncoderBlockWorkspace.MaximumCoefficientCount, workspace.TransformCoefficients.Length); Assert.Equal(Av1EncoderBlockWorkspace.MaximumCoefficientCount, workspace.DequantizedCoefficients.Length);