From 578ec34d9c1661f3cf30d6c1b01b6d50f202ca9d Mon Sep 17 00:00:00 2001 From: James Jackson-South Date: Sat, 5 Sep 2026 14:58:59 +1000 Subject: [PATCH] Fix AV1 intra reference extents and partition context --- HEIF_IMPLEMENTATION_PLAN.md | 163 ++++++++++++++++-- ...traSuperblockEncoder.ChromaModeDecision.cs | 75 +++++--- .../Av1IntraSuperblockEncoder.ModeDecision.cs | 133 +++++--------- .../Formats/Heif/Av1/Av1EncoderFrameTests.cs | 92 ++++++++++ .../Av1/Av1IntraSuperblockEncoderTests.cs | 89 ++++++++++ 5 files changed, 423 insertions(+), 129 deletions(-) diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index c465ca7276..da795a1af3 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -6,9 +6,136 @@ Complete a production-quality, fully managed AV1 codec and its bounded AVIF/HEIF This plan is the authoritative delivery checklist. A source file, unit test, build, self-roundtrip, or local implementation is not completion evidence by itself. +## Takeover audit: 2026-09-05 + +The earlier checked boxes and measurements below are historical checkpoint reports, not accepted conclusions about +the current encoder or complete decoder. The fresh production-path audit is still in progress. No benchmark has +been run during this investigation, and the complete 738-file upstream diff has not yet received a line-by-line audit. + +### Reference and worktree evidence + +- Live `git ls-remote` identifies official `https://aomedia.googlesource.com/aom` main as + `d565eec60f084421fa34fc0534b760c6452b6a6c`. The export at `D:\GitHub\ynse01\aom-d565eec6-source` + was compared with that revision's official archive: 1,522 files present, 18 byte-identical, 1,504 differing only + by CRLF versus LF, and zero remaining content differences. The archive is temporary and outside this repository. +- Live ImageSharp main and local `upstream/main` both resolve to `adb982081a7e89a824f873f7f286f517e04f80dd`. + The starting HEAD is `a7f0fca6b01d0d498862aba66e024942393c1614`; the index is empty. The initial worktree + contains 13 modified files and two untracked benchmark files. Preserve all existing work while correcting demonstrated defects. +- The optimized native build is `D:\GitHub\ynse01\aom-d565eec6-build-x64-release`: Ninja, MSVC x64, + Release `/O2 /Ob2 /DNDEBUG`, runtime CPU dispatch, decoder, encoder, and high-bit-depth support. + Generated `config/aom_config.h` enables SSE2, SSE4.1, AVX2, and AVX512. The cache's zero-valued HAVE entries + do not describe the generated configuration. The version header reports 3.15.0 but does not independently identify a commit. +- Commit `a7f0fca6b` already contains temporary native integration: + `tests/ImageSharp.Benchmarks/Codecs/Heif/Native/aom_benchmark.c`, its `CMakeLists.txt`, + `LibaomBenchmarkEncoder.cs`, and the referencing sequence benchmark. The current untracked decoder benchmark + and adapter wrapper are also temporary integration. Do not stage or commit them. Removal from existing commits + or deletion of local reference files requires a separate, concrete proposal; no history rewrite or deletion is authorized here. + Existing generated conformance fixtures also require classification before any cleanup proposal. + +### Confirmed implementation deviations + +Managed paths below are relative to the repository; reference paths are relative to the verified libaom export. +Line numbers describe the inspected starting tree, before subsequent corrections. + +| Class | Managed evidence | Official reference evidence | Finding | +| --- | --- | --- | --- | +| Missing functionality | `Av1FrameEncoder.cs:372-408`, under `src/ImageSharp/Formats/Heif/Av1/Pipeline` | `av1/encoder/encoder.c:641-646`; `av1/av1_cx_iface.c:287-288,1284-1286,1561-1562` | Sequence setup unconditionally disables CDEF, restoration, and intra-edge filtering. These are not equivalent to the reference's configured tool decisions. | +| Architectural deviation | `Av1FrameEncoder.cs:508-542,1551-1557` | `av1/encoder/encode_strategy.c:168-230,1664-1669` | Every frame is error resilient, refreshes all slots, disables frame-end CDF publication, and resets probabilities. The reference selects retained primary-reference state. | +| Missing functionality | `Av1IntraSuperblockEncoder.ModeDecision.cs:185-245`; `Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:526-566` | `av1/encoder/partition_search.c:3320` onward; `av1/encoder/rdopt.c:6196-6236` | Inter frames retain a fixed 8x8 partition tree and search only LAST. Larger partitions and additional reference roles are not implemented by this path. | +| Architectural deviation | `Av1IntraSuperblockEncoder.ModeDecision.cs:597-609,662-672,824-834` | `av1/encoder/rdopt.c:111-142,6186-6236`; `av1/encoder/intra_mode_search.c:1291-1344` | Managed coding finishes intra search before inter evaluation. Reference inter search has its own ordered candidates, pruning state, bounds, and later intra evaluation. | +| Architectural deviation | `Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:576-617` | `av1/encoder/rdopt.c:111-142` | Managed single-reference mode order is NEAREST, NEAR, GLOBAL, NEW. The reference default order is NEAREST, NEW, NEAR, GLOBAL across eligible references. | +| Architectural deviation | `Av1IntraSuperblockEncoder.ReferenceModeDecision.cs:1278-1414` | `av1/encoder/mcomp.c`; caller policy in `av1/encoder/motion_search_facade.c` | The managed radius is an effort-shifted value capped by its border; each scale visits eight offsets once. This is a simplified search controller whose full reference-policy reconciliation remains open. | +| Missing functionality | `Av1TransformBlockEncoder.cs:1047-1097` | `av1/encoder/encodemb.c:842-885` | Lossy transform coding ends at fast quantization. Reference coding selects quantization with trellis policy and can optimize coefficients before reconstruction. Native primitive arithmetic alone does not establish encoder parity. | +| Architectural deviation | `Av1IntraSuperblockEncoder.ModeDecision.cs:1581-1655` | `av1/encoder/intra_mode_search_utils.h:622-657`; `av1/encoder/intra_mode_search.c:467-491,1597-1619` | The 8x8 SATD screen imports only the 1.5-best threshold. The reference also maintains ranked candidates and quantizer/neighbor-dependent pruning, and owns the surrounding mode/transform decisions. Contrary to an initial audit hypothesis, this revision's `intra_model_rd` does return raw SATD. No modeled-RD-versus-SATD numerical defect is established. | +| Missing functionality | `Av1TransformBlockEncoder.cs:695-837` | `av1/common/reconintra.c:958-986,1204-1243` | Encoder directional prediction bypasses edge preparation and upsampling. Flipping the sequence flag alone would make encoder reconstruction disagree with its emitted syntax. | +| Verification gap | `Av1InverseTransformerFactory.cs:47-60,98-112`; `Av1InverseTransformTests.cs:402-532` | Sparse inverse dispatch in `av1/common/idct.c` and `av1/common/x86` | The uncommitted decoder path specializes DC-only DCT; other lossy EOB values still use full transforms. Its new tests compare against the managed full transform, not an independent native oracle. Complete sparse dispatch and SIMD reconciliation remain open. | + +Two reconstruction-input defects were established and corrected during this audit: + +- Rectangular intra transforms require width plus height samples on each extended edge. The starting encoder + prepared twice the width above and twice the height to the left + (`Av1IntraSuperblockEncoder.ModeDecision.cs:1466-1534,2583-2677`; + `Av1IntraSuperblockEncoder.ChromaModeDecision.cs:179-182,1153-1223`). + This could expose unprepared scratch samples to directional prediction. Reference + `av1/common/reconintra.c:1149-1184,1451-1488,1817-1820` copies the available adjacent edge and repeats its endpoint + through the width-plus-height extent. Luma and chroma now reuse the existing edge-preparation method, and tiled + candidate preparation follows the same extent without adding storage or an allocation. +- Trial and final geometry discarded the enclosing partition, and all encoder directional edge-availability calls + passed `None` (`Av1IntraSuperblockEncoder.ModeDecision.cs:397-418,858-921,1434-1464,2557-2581` in the starting tree). + Reference `av1/common/reconintra.c:158-192,343-379` selects different availability tables for mixed vertical + partitions; `av1/decoder/decodeframe.c:1392-1416` distinguishes split child nodes from mixed-partition leaves. + Both trial and final setup now retain the terminal partition in existing mode information, and luma, chroma, + and tiled prediction consume it. Split children retain their own implicit `None` leaf state. + +The isolated 8x8 Hadamard pruning hook has been removed from production mode selection. Its primitive and existing +component tests remain uncommitted in the worktree. Removing the unsupported hook does not complete the remaining +encoder controller or establish a quality or performance improvement. + +Disabled tools, limited search, and different decision order can also change reconstructed samples without +producing an invalid bitstream. They are separate from the reconstruction-input defects above. + +### Architecture and verification findings + +Fresh correction verification on 2026-09-05: + +- Final Release .NET 11 build after removing the screening hook: zero errors and zero reported warnings + on that incremental build. The preceding test compilation reported 1,009 existing warnings. +- Visual Studio VSTest 18.9, .NET 11 preview 7, serialized collections, one test thread, stop-on-failure: + 251/251 cases passed in `Av1EncoderFrameTests`, `Av1IntraSuperblockEncoderTests`, and `HeifEncoderTests`. + The final report is `D:\GitHub\ynse01\av1-takeover-20260905\no-screen-final.trx`. +- `RectangularIntraReferencesExtendTheLastAvailableSample` checks explicit reference-edge samples at + 8/10/12 bits for both rectangle orientations and available/unavailable extensions, including output sentinels. +- `ProductionMixedPartitionsPreserveReconstructionOrder` requires actual mixed partitions and compares the live + mapped encoder partition state and retained reconstruction with production decoding. Its two emitted 32x32 + monochrome streams also match the optimized libaom decoder: 2,048 luma samples, maximum error 0, + and zero samples exceeding one. +- The twelve freshly regenerated two-frame color streams cover 8/10/12-bit 4:2:0, 4:2:2, and 4:4:4: + managed and optimized native decoding agree on all 21,348 Y/U/V samples, maximum error 0, zero exceeding one. + Per-frame and per-plane counts are in `D:\GitHub\ynse01\av1-takeover-20260905\decoder-comparison.json`. +- These are bounded reconstruction and same-bitstream decoder checks. They do not prove separately encoded + output parity, complete encoder control flow, or decoder-wide conformance. No benchmark was run. + Temporary launch scripts, native comparison output, logs, and reports remain local and are excluded from commits. + +- PNG resolves options before converted metadata and sanitizes incompatible output combinations + (`src/ImageSharp/Formats/Png/PngEncoderCore.cs:1632-1675`). TIFF follows the same precedence and converts + unsupported combinations (`src/ImageSharp/Formats/Tiff/TiffEncoderCore.cs:111-175,371-457`). + HEIF's generic pixel input must retain that conversion contract; source pixel type is not an eligibility gate. +- JPEG's closed `JpegColorConverter` owns traversal and calls semantic static operator arithmetic + (`src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.Operator.cs:144-250`). + Shared prediction work must follow that family boundary and existing ownership APIs. +- Encoder block scratch already shares mode and inter storage by their non-overlapping lifetimes + (`Av1EncoderBlockWorkspace.cs:31-103`). The sequence constructor retains picture, coefficient, block, + entropy, conversion, and frame state (`Av1FrameEncoder.cs:1492-1646`). Exact sizing, failure unwinding, + alignment, reference parity, and allocation attribution still need the complete native comparison. +- Two decoders agreeing on one managed bitstream establishes only decoding agreement for that bitstream. + The benchmark's photographic setup checks that agreement; it does not compare separately encoded outputs. + Its native output uses ImageSharp color conversion, so it is not an independent RGB conversion oracle. +- Historical separate-encoder Y/U/V maxima of 40/30/53 fail the required one-component-unit limit. + Counts exceeding one were not supplied with those historical figures. Neither those figures nor the recorded + 1,363.09/71.73 ms timing pair is a new measurement of a subsequently edited tree. +- VSTest logs identify Visual Studio 18.9 x64 and `.NETCoreApp,Version=v11.0`. Future runs must deduplicate + child environment keys case-insensitively, disable collection parallelism, stop on failure, and run serially. + Absence of an observed dialog is not evidence that no popup occurred. + +### Required completion gates + +- [ ] Finish the full production-path and complete upstream-diff audit, including conversion, animation, + ownership, filters, decoder SIMD, and independent validity of the claimed tests. +- [ ] Reconcile frame configuration and encoder decision policy with the reference before isolated pruning changes. +- [ ] Implement missing tools and complete reference, partition, motion, transform, coefficient, and winner decisions. +- [ ] Compare separately encoded results from identical source samples with explicitly reconciled settings. + Report maximum absolute error and counts exceeding one for every decoded output component and every frame. + The acceptance limit is one component unit per sample; PSNR and average error cannot replace it. +- [ ] Verify each final relevant edit with focused serialized Release .NET 11 Visual Studio VSTest and independent + native production-output checks. Compilation and component tests do not close codec completeness. +- [ ] Run equivalent end-to-end benchmarks only after the relevant source comparison justifies the next change. + Retain output sizes, absolute times, per-sample errors, memory units, reference configuration, and limitations. +- [ ] Inspect the staged diff before every verified checkpoint commit and exclude all temporary native integration, + codec sources, binaries, build directories, and generated comparison artifacts. Do not push. + ## Source authority -- AV1 codec syntax, tables, fixed-point arithmetic, prediction, transforms, entropy behavior, filters, encoder decisions, and lifecycle behavior must be ported and checked only against the current `main` branch of the official libaom checkout at `D:\GitHub\AOMediaCodec\aom`. +- AV1 codec syntax, tables, fixed-point arithmetic, prediction, transforms, entropy behavior, filters, encoder decisions, and lifecycle behavior must be ported and checked only against the current `main` branch of the official libaom checkout at the verified `D:\GitHub\ynse01\aom-d565eec6-source` export. - Libaom is the sole external codec implementation source. Do not use HM, libheif, FFmpeg, GPAC, SVT-AV1, dav1d, libgav1, or any other codec implementation as an algorithm, arithmetic, output, or architecture reference. - Existing ImageSharp and JPEG code is authoritative only for ImageSharp architecture, allocator ownership, SIMD dispatch, pixel conversion, and test API patterns. It is not an alternate AV1 algorithm source. - Production code must not load, invoke, install, or fall back to a native codec. @@ -44,6 +171,11 @@ Reconciled with the worktree on 2026-09-05. Current interpolation-search checkpoint, implemented on 2026-09-05 with focused verification in progress: +- [x] Complete elementary-stream decoder comparison now includes parsing, reconstruction, output allocation, RGB conversion, and disposal for both ImageSharp and optimized current-main libaom. Eight-bit Kodak and ten-bit Cosmos inputs match every `Rgb48` sample before timing. Initial warmed means are 15.130 versus 5.423 ms and 27.655 versus 10.246 ms respectively. These expose an open decoder gap; they are not container-load measurements. Reproduction and limitations are recorded in the benchmark README. +- [x] Shared inverse reconstruction now uses EOB to select a DC-only DCT path, preserving both axis roundings, rectangular normalization, input clamps, and final clipping through the existing semantic output operators. Vector512 output was added to that operator contract. Every transform size, signed boundary, padded separate/in-place destination, and supported sample precision is checked against the full transform. The post-change photographic encoder payload remained byte-identical; decoder RGB output remained exact against libaom. Short decoder measurements do not yet establish a statistically significant improvement. +- [~] Fixed 8x8 lossy luma search now screens candidates with the reference's unnormalized Hadamard magnitude and 1.5x best-model threshold. Existing tensor and transpose APIs vectorize the operation over frame-reused scratch; no per-candidate allocation or custom hardware-width operator was introduced. A five-warmup, ten-measurement repeat records 1,363.09 ms versus native 71.73 ms, about 34% faster than the initial managed baseline but still approximately 19x behind native. Output increased slightly to 11.577 KiB and aggregate native-plane PSNR declined from 37.124 to 37.069 dB. This tradeoff does not close the performance/compression gate. Larger-block screening, top-ranked pruning, transform bounds, winner refinement, filter decisions, and allocation attribution remain open. Evidence: `artifacts/BenchmarkDotNet/av1-screen-verified-short-20260905/20260905-140358`. +- [x] The current screening/DC/native-profile/moving-color subset passes 25 cases in each of three separately configured VSTest hardware tiers. Normal-path verification passes 84 screening/DC/superblock cases and 165 frame/public encoder cases. The photographic benchmark now additionally requires exact ImageSharp/libaom RGB agreement across all three dependent frames before timing. No expected image or golden output was changed. Release test/benchmark builds have zero errors and their existing 1,009/39 warning baselines. Evidence and the corrected process-local VSTest environment handling are recorded in `tests/ImageSharp.Benchmarks/Codecs/Heif/README.md` and `artifacts/TestResults/av1-screen-20260905`. + - [x] The reference benchmark now measures identical RGB-to-OBU boundaries, including pixel conversion for every frame on both sides. A benchmark-only C adapter links the optimized current-main libaom build in-process and borrows the existing converted plane storage; production remains fully managed. Setup and file I/O are excluded equally. Native quantizer bounds are fixed to the managed base index, with independent speed settings and explicit size/quality reporting. Reproduction, native build provenance, lifetime documentation, and exact output hashes are in `tests/ImageSharp.Benchmarks/Codecs/Heif/README.md`. - [ ] The corrected benchmark exposes a substantial remaining performance and compression gap. For three photographic 256x256 frames, ImageSharp effort seven takes 2,053.31 ms and writes 11.54 KiB at 37.124 dB aggregate native YUV PSNR; current-main libaom cpu-used six takes 70.67 ms and writes 8.27 KiB at 38.942 dB. Both include RGB conversion and both measured outputs decode to all three complete frames. This is fixed-base-quantizer evidence, not equal-quality evidence. The managed path records 9.38 MiB of managed allocations per operation, which still requires attribution; native memory is not measured by that counter. The Short-run evidence is `artifacts/BenchmarkDotNet/av1-sequence-rgb-fixed-q-short-20260905/20260905-131149`. Power-plan and CPU-query warnings remain documented. The new benchmark files build in net11.0 Release and have no Roslyn compiler/analyzer diagnostics. Do not close the interpolation-performance gate or advance to additional reference tools until the gap is addressed. - [x] The requested in-progress tree was committed as `433afd1a9` before further encoder work. That commit is a checkpoint, not a claim of completed interpolation or codec delivery. @@ -804,15 +936,15 @@ SIMD traversal consistency evidence on 2026-09-02: Decoder exit gate: -- [x] Every supported native format and AV1 tool has exact current-main libaom production-path evidence. -- [x] Every supported presentation behavior has established reference-image evidence at the correct output precision. -- [x] No decoder path relies on a native codec, copied plane, per-block allocation, or contiguous memory-group accident. -- [x] All allocator ownership is deterministic and exactly once. -- [x] Full focused Release verification is recorded with no false coverage claims. +- [ ] Re-establish current-main native evidence for every supported AV1 tool through the complete production path. +- [ ] Re-establish presentation behavior against independent reference images at the correct output precision. +- [ ] Complete the decoder audit for managed execution, copies, per-block allocation, and segmented memory. +- [ ] Verify allocator ownership and exceptional-path disposal throughout the decoder. +- [ ] Record focused Release verification of the final tree; earlier checkpoint results do not close these gates. ## AV1 encoder implementation -Writer primitives are not an encoder. The public encoder remains incomplete until it produces independently decodable AV1 payloads and AVIF containers for every exposed option. +Writer primitives are not an encoder. The public encoder remains incomplete until its complete decision and reconstruction paths follow the reference, every exposed option has production-path verification, and separately encoded output satisfies the one-unit per-sample acceptance limit. ### 5. Define and enforce the encoder contract @@ -879,10 +1011,10 @@ Encoder verification contract: - [~] Luma and chroma palette-color coding now matches current libaom's neighbor-cache flags, sorted delta representation, wrapped V-plane deltas, strict delta-versus-raw V selection, and fixed-point color-rate model at 8, 10, and 12 bits. Encoder costing and emission use only fixed stack spans, including explicitly initialized cache-membership state, and steady-state color costing allocates zero managed bytes. The decoder consumes the same bounded color-syntax primitive after the tile reader derives its neighbor cache, removing duplicated color parsing without changing retained palette ownership. Nine focused syntax, exact palette decode, constrained-allocation, truncation, presentation, and allocation cases pass; all 1,933 entropy cases and all 8,983 HEIF/AV1 cases pass direct net11 Release VSTest. The exact Release rebuild remains at 1,005 warnings and zero errors. Retained encoder palette colors, neighbor caches, color-index maps, candidate generation, and production palette selection remain incomplete, and the compact 8-byte frame mode entries were not enlarged. - [~] Palette color-index map coding now shares the exact current-libaom neighbor weights, stable color ordering, five context classes, first-index uniform code, and diagonal wavefront between encoder costing, encoder writing, and decoder parsing. The decoder's stack-allocated context scores are explicitly cleared before accumulation, removing an invalid dependency on uninitialized stack contents. Costing and writing use a closed generic operation while the shared driver owns traversal and context derivation, so the semantic operations remain independent of map layout and tail handling. The path adds no retained state or per-call managed allocation. Its allocation regression now runs one complete unmeasured hot-path window before measuring an independent 1,000-call steady-state window, so tiered-runtime transitions cannot make the full parallel suite report a one-time allocation as a recurring operation cost. Twelve focused map, exact palette decode, padding, trailing-bit, and allocation cases pass; all 1,941 entropy cases and all 8,991 HEIF/AV1 cases pass direct net11 Release VSTest. The exact Release rebuild remains at 1,005 warnings and zero errors. Production payloads remain unchanged because palette selection is still disabled; retained colors, neighbor caches, index-map storage, candidate generation, and production palette mode decision remain incomplete. - [~] Retained encoder palette state and production palette writing now mirror current libaom's 50-byte palette-mode contents, separate luma and shared-chroma sizes, three eight-color planes, above-and-left sorted cache, 64-sample above-cache boundary, mode contexts, palette colors, color-index maps, and syntax order. The current block keeps one inline value in the reusable superblock workspace; only the 4x4-granularity top and left picture edges retain copies for later blocks. For a 3840x2160 tile these edges occupy about 73.4 KiB instead of about 6.2 MiB for a 50-byte palette value attached to every 8x8 mode allocation. Luma and chroma index maps occupy a fixed 32 KiB region of the single 40.3 KiB superblock-workspace owner. That owner is allocated with encoder state, matching libaom's compressor-state lifetime while removing libaom's separate palette allocation and cleanup path. The compact final-block decision region remains about 8.3 KiB. The writer caps map traversal to the coded plane count, writes maps before transform syntax, and publishes palette edges only after the current block has consumed preceding contexts. The previous eight focused size, alignment, ownership, cache-boundary, round-trip, map-consumption, and edge-publication cases passed with all 114 palette cases, all 1,942 entropy cases, and all 8,996 HEIF/AV1 cases through direct foreground net11 Release VSTest. The current one-owner refactor has zero Roslynk compiler errors; runtime verification remains pending. The current source reference is official libaom main at `d565eec60f084421fa34fc0534b760c6452b6a6c`. -- [~] Luma palette clustering now follows current libaom's one-dimensional search primitive exactly: equal-interval midpoint initialization, first-color tie order, rounded centroid means, deterministic empty-cluster replacement, the 50-iteration limit, and retention of the preceding state when distortion increases. Nearest-color assignment improves on libaom's AVX2 implementation by dispatching Vector512, Vector256, Vector128, then scalar through ImageSharp's shared vector-count helpers. The primitive uses only bounded stack scratch and introduces no allocator rent, managed array, or per-row copy. Three independent tests cover exact centroid convergence, initialization order, 12-bit nearest-color distortion, destination bounds, and every hardware-intrinsic tier. The complete AVIF set passes 8,930 of 8,930 cases and the HEIF set passes 230 of 230 cases through direct foreground net11 Release VSTest. The exact net11 Release rebuild reports 1,050 solution warnings and zero errors, and Roslynk reports zero compiler errors. Candidate enumeration, palette-cache snapping, transform RD selection, and production activation remain in the open luma-palette checkpoint. -- [~] Live luma palette selection now follows current libaom's dominant-color and one-dimensional K-means candidate families, cache-bias threshold, sorted duplicate removal, active-edge map extension, and strict winner tie order. It improves on speed-configured libaom by evaluating both candidate families at every legal 2-through-8 size without early pruning, then exhaustively evaluates every legal transform using the existing SIMD prediction, residual, transform, quantization, and reconstruction operators. Candidate storage remains bounded stack memory; the reusable maps come from the fixed encoder-lifetime superblock workspace, so palette search cannot introduce a first-use allocation. A production tile test proves that full 8x8 and clipped 5x3 blocks at 8 and 12 bits select exact colors and indices, extend the visible edges through coded padding, reconstruct every sample without coefficients, and emit a nonempty tile. The complete 57-case intra-superblock set, 8,931-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains gated until chroma palette mode and its rate accounting are complete. -- [~] Paired chroma palette clustering now preserves current libaom's squared two-component distance, first-centroid tie order, independently rounded U/V means, paired deterministic empty-cluster replacement, preceding-state retention on increased distortion, and 50-iteration limit. Keeping the source planes separate avoids interleave/deinterleave copies and improves on libaom's AVX2 ceiling with Vector512, Vector256, Vector128, then scalar dispatch through ImageSharp's shared vector-count helpers. Three independent tests cover exact paired convergence, midpoint initialization, 12-bit distance and index parity, untouched destination bounds, and every intrinsic tier. The exact Release test-project build reports 1,992 baseline warnings and zero errors; the focused three-case set, complete 8,934-case AVIF set, and complete 230-case HEIF set pass direct foreground net11 Release VSTest. Roslynk reports zero compiler errors and no touched-file analyzer warnings. Candidate integration and production activation remain in the open chroma-palette checkpoint. -- [~] Live paired chroma palette selection now follows current libaom's complete 2-through-8 color-size search, U-plane neighbor-cache snapping, stable U-ordered color pairs, shared U/V index map, implicit DCT-DCT transform, and strict rate-distortion winner replacement. It improves on speed-configured libaom by applying no early header-cost pruning, keeps planar U/V source data separate, and reuses the SIMD-first prediction, residual, transform, quantization, and reconstruction operators without allocator-backed candidate storage. The production tile regression proves both palette-mode probability branches, exact paired colors and indices, coefficient-free reconstruction, and nonempty syntax. The complete 58-case intra-superblock set, 8,935-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains the next checkpoint. +- [~] Luma palette clustering now follows current libaom's one-dimensional search primitive exactly: equal-interval midpoint initialization, first-color tie order, rounded centroid means, deterministic empty-cluster replacement, the 50-iteration limit, and retention of the preceding state when distortion increases. Nearest-color assignment dispatches Vector512, Vector256, Vector128, then scalar through ImageSharp's shared vector-count helpers. Wider dispatch alone does not establish an end-to-end performance improvement. The primitive uses only bounded stack scratch and introduces no allocator rent, managed array, or per-row copy. Three independent tests cover exact centroid convergence, initialization order, 12-bit nearest-color distortion, destination bounds, and every hardware-intrinsic tier. The complete AVIF set passes 8,930 of 8,930 cases and the HEIF set passes 230 of 230 cases through direct foreground net11 Release VSTest. The exact net11 Release rebuild reports 1,050 solution warnings and zero errors, and Roslynk reports zero compiler errors. Candidate enumeration, palette-cache snapping, transform RD selection, and production activation remain in the open luma-palette checkpoint. +- [~] Live luma palette selection now follows current libaom's dominant-color and one-dimensional K-means candidate families, cache-bias threshold, sorted duplicate removal, active-edge map extension, and strict winner tie order. It evaluates both candidate families at every legal 2-through-8 size without the reference's speed-dependent pruning, then evaluates every legal transform. This is a controller deviation; evaluating more candidates does not establish improved quality, speed, or reference parity. Candidate storage remains bounded stack memory; the reusable maps come from the fixed encoder-lifetime superblock workspace, so palette search cannot introduce a first-use allocation. A production tile test proves that full 8x8 and clipped 5x3 blocks at 8 and 12 bits select exact colors and indices, extend the visible edges through coded padding, reconstruct every sample without coefficients, and emit a nonempty tile. The complete 57-case intra-superblock set, 8,931-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains gated until chroma palette mode and its rate accounting are complete. +- [~] Paired chroma palette clustering now preserves current libaom's squared two-component distance, first-centroid tie order, independently rounded U/V means, paired deterministic empty-cluster replacement, preceding-state retention on increased distortion, and 50-iteration limit. The source planes remain separate, with Vector512, Vector256, Vector128, then scalar dispatch through ImageSharp's shared vector-count helpers. An end-to-end improvement over the native implementation has not been established. Three independent tests cover exact paired convergence, midpoint initialization, 12-bit distance and index parity, untouched destination bounds, and every intrinsic tier. The exact Release test-project build reports 1,992 baseline warnings and zero errors; the focused three-case set, complete 8,934-case AVIF set, and complete 230-case HEIF set pass direct foreground net11 Release VSTest. Roslynk reports zero compiler errors and no touched-file analyzer warnings. Candidate integration and production activation remain in the open chroma-palette checkpoint. +- [~] Live paired chroma palette selection now follows current libaom's complete 2-through-8 color-size search, U-plane neighbor-cache snapping, stable U-ordered color pairs, shared U/V index map, implicit DCT-DCT transform, and strict rate-distortion winner replacement. It omits the reference's early header-cost pruning, keeps planar U/V source data separate, and reuses the existing prediction, residual, transform, quantization, and reconstruction operators. Omitting pruning is an unresolved decision-policy deviation, not an established improvement. The production tile regression proves both palette-mode probability branches, exact paired colors and indices, coefficient-free reconstruction, and nonempty syntax. The complete 58-case intra-superblock set, 8,935-case AVIF set, and 230-case HEIF set pass direct foreground net11 Release VSTest. The exact Release test-project build reports 1,992 baseline warnings and zero errors; Roslynk reports zero compiler errors and no touched-file analyzer warnings. Production frame activation remains the next checkpoint. - [x] Production screen-content activation now matches current libaom's default good-quality detector: it scans only complete 16x16 luma blocks, normalizes palette samples to eight bits, admits 2-through-4-color blocks, and uses the reference's strict greater-than-ten-percent frame-area threshold. The same pass accumulates centered sums and squared sums at native precision, applies libaom's exact 10-bit and 12-bit variance rounding, and enables intra-block copy only when positive rounded per-pixel variance exceeds its strict one-twelfth frame-area threshold. A 256-bit stack bitset and fifth-color early exit replace libaom's larger per-block histogram without a second source scan or allocation. The adaptive sequence flag remains enabled and both frame flags are fixed before picture-state allocation. Focused regressions prove strict palette-threshold equality, high-bit-depth normalization, the exact variance rounding boundary, five-color rejection, emitted frame-header activation, actual production IBC selection, and production decode. The exact Release test-project build reports 1,992 baseline warnings and zero errors; all 9,242 non-HEVC HEIF/AV1 cases pass direct foreground net11 Release VSTest. Current-main `aomdec` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` accepts all 31 payloads regenerated by the current test tree, including an actual IBC-coded 328x16 stream with decoded MD5 `677435e5af39c930af1178f91c34af6a`. Roslynk reports zero compiler errors and no touched-file analyzer warnings. - [~] Intra-block-copy rate accounting now uses the live frame-local flag and displacement-vector distributions without copying or adapting either context during candidate measurement. Displacement-vector costing and writing share one closed symbol operation over the exact current-libaom joint, sign, magnitude-class, class-zero, and integer-offset syntax; final mode evaluation applies libaom's 120/128 displacement-rate weight with nearest-integer rounding. Independent fixed costs cover all four joint states, both signs, class zero, and large offset classes before adaptive writes, followed by an encoder/decoder round trip through the same sequence. Encoder and decoder reference-vector derivation now share the exact eight-candidate spatial scan, independent nearest and outer-region ranking, top-right partition geometry, clamping, and tile-relative fallback. Selected vectors use a naturally aligned pair of signed 16-bit components packed into the existing picture-state owner only when intra-block copy is permitted; a 3840x2160 frame retains 130,560 vectors in 510 KiB while leaving the compact 8-byte mode allocation unchanged. The tile writer derives the same reference and emits the retained vector without another allocation or copy. Coefficient costing and writing now select the inter transform sets and frame-local probability tables required by intra-block copy; independent tests verify every legal symbol against the exact default inter distribution and round-trip full and reduced sets from 4x4 through 32x32. Legal 8x8 hash discovery now indexes every visible source origin, including unaligned origins, in libaom's coarse-to-fine insertion order with the same 256-candidate bucket cap. A separable rolling hash fills one packed picture-lifetime workspace before reconstruction, then reuses that workspace for integer candidate links; exact wide or SIMD block comparison rejects hash collisions, and SIMD variance uses libaom's eight-bit normalization at 8, 10, and 12 bits. Power-of-two bucket arrays scale down with small images and stop at the reference's 16-bit limit, avoiding libaom's fixed six-size pointer table; the 3840x2160 search index occupies about 32.2 MiB and introduces no additional owner or frame copy. Above and left search rectangles, integer displacement legality, strict tie order, and live raw displacement rate follow current libaom. Motion-candidate ranking uses libaom's undiscounted probability cost and exact variance-domain error-per-bit scaling, separately from the later 120/128 final-mode discount. The allocation-free full-pixel core now follows current libaom's NSTEP search: it clamps the spatial reference to each legal region, traverses the fixed 15-stage radii and site order, skips equivalent centered 210-pixel stages, repeats progressively shorter paths, and compares their winners in the normalized variance domain. Paths above the speed-zero screen-content threshold continue through libaom's 256-pixel, one-pixel-step exhaustive mesh. Four adjacent byte or high-bit-depth candidates share each SIMD source load, strict row-major tie ordering is retained, and the final legal tail column remains searchable where libaom's current four-wide remainder loop omits it. Byte and high-bit-depth operators compute each 8x8 absolute difference with Vector128 before scalar fallback; high-bit-depth SAD remains in its native sample scale while its quantizer-derived rate multiplier uses libaom's normalized AC step. Production mode decision now derives the same spatial displacement reference used by the writer, deduplicates hash and full-pixel finalists in search order, and evaluates every surviving vector through complete luma and chroma transform RD. This intentionally improves on libaom's preliminary-error pruning by permitting a hash and pixel finalist from the same search region to compete using their final syntax and reconstruction costs. Prediction is prepared once per plane and vector, including integer or half-sample chroma phase, then reused across every legal inter transform without an allocator rent or frame copy. The joint comparison includes the live intra-block-copy flag, discounted displacement rate, skip flag, coefficient syntax, and normalized Y/U/V distortion; an empty transform alternative can win only when its complete skip cost is strictly lower, while conventional intra and earlier vectors retain tie precedence. Winning reconstruction, coefficients, transform state, DC modes, cleared palette/filter/CfL state, and displacement are copied once into the existing retained stores. Production regressions force the path at 8 and 12 bits and force 4:2:0 horizontal half-sample chroma with an unaligned reference. The former bulk local workspace occupied 2.75 KiB for byte samples or 3.375 KiB for high-bit-depth samples. Prediction, candidate, winning reconstruction, residual, and coefficient scratch now occupy one naturally aligned 3.125 KiB extension of the existing frame-reused block-workspace owner, matching libaom's reusable macroblock-scratch lifetime without adding an allocation; only the 128-byte reference, weight, and finalist arrays remain on the stack. The net11 Release solution build reports zero errors; all 2,082 focused transform, entropy, intra-block-copy, intra-superblock, and frame-encoder cases and all 9,242 non-HEVC HEIF/AV1 cases pass through direct foreground VSTest, with tiered compilation disabled only for the full allocation-sensitive suite. Adaptive production activation is complete, and the emitted frame flag remains authoritative for the complete frame rather than being invalidated after tile coding. - [x] The expanded checkpoint exposed a pre-existing transform-block test that asserted uninitialized pooled padding was zero. The test now initializes the complete physical luma plane with a sentinel and proves the block operation leaves both adjacent padding samples unchanged. The exact net11 Release rebuild remains at 1,005 baseline warnings and zero errors, the focused allocator-order set passes 30 of 30 cases, and the complete HEIF/AV1 namespace passes 8,859 of 8,859 direct VSTest cases with zero failures or skips. @@ -921,13 +1053,14 @@ Encoder verification contract: Encoder exit gate: -- [x] Current-main libaom accepts every currently produced AV1 payload. -- [x] Lossless output is exact at public pixel and direct native-plane precision for 8-, 10-, and 12-bit output. -- [ ] Lossy output demonstrates recorded quality and effort tradeoffs with absolute size, quality, timing, and allocation evidence. +- [ ] Current-main libaom accepts the payloads regenerated from the final tree. +- [ ] Reverify lossless output at public pixel and native-plane precision for 8-, 10-, and 12-bit output. +- [ ] Separately encoded lossy outputs differ by no more than one unit at every decoded output sample with reconciled settings; report maxima and counts exceeding one. +- [ ] Record equivalent end-to-end absolute timing, output size, quality, and allocation evidence after the source audit. - [ ] 8, 10, and 12-bit monochrome, 4:2:0, 4:2:2, and 4:4:4 outputs pass. - [ ] Alpha, grids, metadata, color profiles, transforms, and bounded sequences pass. - [ ] ImageSharp decode of its own output is supplemental coverage only, never the sole oracle. -- [x] Public encoding no longer throws for a supported AV1 request. +- [ ] Verify every exposed encoding combination through established ImageSharp conversion behavior. - [ ] Focused Release and FeatureTestRunner verification passes with exact recorded evidence. ## Architecture rules diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs index fe3b448777..27db92d751 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ChromaModeDecision.cs @@ -116,6 +116,7 @@ internal static partial class Av1IntraSuperblockEncoder bool rightAvailable = modeInfoColumn + (transformSize.Get4x4WideCount() << subsamplingX) < macroBlock.Tile.ModeInfoColumnEnd; bool bottomAvailable = modeInfoRow + (transformSize.Get4x4HighCount() << subsamplingY) < macroBlock.Tile.ModeInfoRowEnd; + Av1PartitionType partitionType = modeInfo.Block.PartitionType; bool hasTopRight = Av1IntraReferenceAvailability.HasTopRight( this.picture.Sequence.SequenceHeader.SuperblockSize, blockSize, @@ -123,7 +124,7 @@ internal static partial class Av1IntraSuperblockEncoder modeInfoColumn, hasAbove, rightAvailable, - Av1PartitionType.None, + partitionType, transformSize, 0, 0, @@ -137,7 +138,7 @@ internal static partial class Av1IntraSuperblockEncoder modeInfoColumn, bottomAvailable, hasLeft, - Av1PartitionType.None, + partitionType, transformSize, 0, 0, @@ -152,7 +153,7 @@ internal static partial class Av1IntraSuperblockEncoder Span blueLeftStorage = workspace.GetReferenceSamples(1); Span redAboveStorage = workspace.GetReferenceSamples(2); Span redLeftStorage = workspace.GetReferenceSamples(3); - this.PrepareReferenceSamples( + PrepareReferenceSamples( blueReconstruction, chromaOrigin, width, @@ -161,10 +162,11 @@ internal static partial class Av1IntraSuperblockEncoder hasAbove, hasTopRight, hasBottomLeft, + this.bitDepth, blueAboveStorage, blueLeftStorage); - this.PrepareReferenceSamples( + PrepareReferenceSamples( redReconstruction, chromaOrigin, width, @@ -173,13 +175,14 @@ internal static partial class Av1IntraSuperblockEncoder hasAbove, hasTopRight, hasBottomLeft, + this.bitDepth, redAboveStorage, redLeftStorage); - ReadOnlySpan blueAbove = blueAboveStorage.Slice(1, width * 2); - ReadOnlySpan blueLeft = blueLeftStorage.Slice(1, height * 2); - ReadOnlySpan redAbove = redAboveStorage.Slice(1, width * 2); - ReadOnlySpan redLeft = redLeftStorage.Slice(1, height * 2); + ReadOnlySpan blueAbove = blueAboveStorage.Slice(1, width + height); + ReadOnlySpan blueLeft = blueLeftStorage.Slice(1, width + height); + ReadOnlySpan redAbove = redAboveStorage.Slice(1, width + height); + ReadOnlySpan redLeft = redLeftStorage.Slice(1, width + height); Av1TransformBlockContext blueContext = Av1TileWriter.GetTransformBlockContexts( Av1ComponentType.Chroma, this.picture.CbDcSignLevelCoefficientNeighbors[tileIndex], @@ -873,8 +876,8 @@ internal static partial class Av1IntraSuperblockEncoder source, transformOrigin, prediction, - aboveStorage.Slice(1, transformWidth * 2), - leftStorage.Slice(1, transformHeight * 2), + aboveStorage.Slice(1, transformWidth + transformHeight), + leftStorage.Slice(1, transformWidth + transformHeight), hasLeft, hasAbove, predictionMode, @@ -1150,7 +1153,21 @@ internal static partial class Av1IntraSuperblockEncoder return Av1RateDistortion.GetCost(this.rateMultiplier, rate, distortion); } - private void PrepareReferenceSamples( + /// + /// Prepares the shared corner and extended top and left edges for intra prediction. + /// + /// The previously reconstructed plane. + /// The prediction block's origin in plane samples. + /// The transform width in samples. + /// The transform height in samples. + /// Whether the left edge is available. + /// Whether the top edge is available. + /// Whether the adjacent top-right block is reconstructed. + /// Whether the adjacent bottom-left block is reconstructed. + /// The sample precision used for unavailable edges. + /// The corner followed by at least width plus height top-edge samples. + /// The corner followed by at least width plus height left-edge samples. + public static void PrepareReferenceSamples( Buffer2DRegion reconstructionPlane, Point blockOrigin, int width, @@ -1159,11 +1176,14 @@ internal static partial class Av1IntraSuperblockEncoder bool hasAbove, bool hasTopRight, bool hasBottomLeft, + Av1BitDepth bitDepth, Span aboveStorage, Span leftStorage) { - Span above = aboveStorage.Slice(1, width * 2); - Span left = leftStorage.Slice(1, height * 2); + // A directional ray can reach width + height - 1 on either edge, including on rectangles. + // Only one adjacent block supplies extension samples; the rest repeat its final sample. + Span above = aboveStorage.Slice(1, width + height); + Span left = leftStorage.Slice(1, width + height); if (hasAbove) { reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y - 1).Slice(blockOrigin.X, width).CopyTo(above[..width]); @@ -1177,7 +1197,7 @@ internal static partial class Av1IntraSuperblockEncoder } } - int midpoint = 128 << (this.bitDepth.GetBitCount() - 8); + int midpoint = 128 << (bitDepth.GetBitCount() - 8); if (!hasAbove) { above[..width].Fill(hasLeft ? left[0] : TOperator.CreateSample(midpoint - 1)); @@ -1188,27 +1208,26 @@ internal static partial class Av1IntraSuperblockEncoder left[..height].Fill(hasAbove ? above[0] : TOperator.CreateSample(midpoint + 1)); } + int topRightCount = hasTopRight ? Math.Min(width, height) : 0; if (hasTopRight) { - reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y - 1).Slice(blockOrigin.X + width, width).CopyTo(above[width..]); - } - else - { - above[width..].Fill(above[width - 1]); + reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y - 1) + .Slice(blockOrigin.X + width, topRightCount) + .CopyTo(above[width..]); } - if (hasBottomLeft) - { - for (int row = height; row < height * 2; row++) - { - left[row] = reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y + row)[blockOrigin.X - 1]; - } - } - else + int topCount = width + topRightCount; + above[topCount..].Fill(above[topCount - 1]); + + int bottomLeftCount = hasBottomLeft ? Math.Min(height, width) : 0; + for (int row = height; row < height + bottomLeftCount; row++) { - left[height..].Fill(left[height - 1]); + left[row] = reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y + row)[blockOrigin.X - 1]; } + int leftCount = height + bottomLeftCount; + left[leftCount..].Fill(left[leftCount - 1]); + // Zone-two projection and Paeth address the common corner immediately before both edges. // Missing edges derive it from the closest coded sample or the bit-depth midpoint. TSample corner = hasAbove && hasLeft diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs index fa2cae059a..2986908368 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1IntraSuperblockEncoder.ModeDecision.cs @@ -376,6 +376,7 @@ internal static partial class Av1IntraSuperblockEncoder leafOrigin, tileIndex, leafSize, + partitionType == Av1PartitionType.Split ? Av1PartitionType.None : partitionType, publishContexts); } @@ -412,7 +413,13 @@ internal static partial class Av1IntraSuperblockEncoder if (this.IsBlockOriginInsideFrame(leafOrigin)) { - this.SetBlockGeometry(leafOrigin, leafSize, Av1PartitionType.None); + // Mixed vertical partitions reconstruct square leaves in a different order. + // Retain the parent decision so prediction uses the same edge availability as the decoder. + // Split children own another partition node; a terminal 4x4 child implicitly owns NONE. + this.SetBlockGeometry( + leafOrigin, + leafSize, + partitionType == Av1PartitionType.Split ? Av1PartitionType.None : partitionType); } } } @@ -861,9 +868,11 @@ internal static partial class Av1IntraSuperblockEncoder Point blockOrigin, ushort tileIndex, Av1BlockSize blockSize, + Av1PartitionType partitionType, bool publishContexts) { - this.SetBlockGeometry(blockOrigin, blockSize, Av1PartitionType.None); + // Trial leaves must use the same reconstruction order as final leaves of this partition. + this.SetBlockGeometry(blockOrigin, blockSize, partitionType); Point modeInfoPosition = blockOrigin >> Av1Constants.ModeInfoSizeLog2; Av1TileWriter.SetModeInfoRowAndColumn( this.picture, @@ -1435,6 +1444,7 @@ internal static partial class Av1IntraSuperblockEncoder int modeInfoColumn = blockOrigin.X >> Av1Constants.ModeInfoSizeLog2; bool rightAvailable = modeInfoColumn + transformSize.Get4x4WideCount() < macroBlock.Tile.ModeInfoColumnEnd; bool bottomAvailable = modeInfoRow + transformSize.Get4x4HighCount() < macroBlock.Tile.ModeInfoRowEnd; + Av1PartitionType partitionType = macroBlock.GetRelativeModeInfo(0).Block.PartitionType; bool hasTopRight = Av1IntraReferenceAvailability.HasTopRight( this.picture.Sequence.SequenceHeader.SuperblockSize, blockSize, @@ -1442,7 +1452,7 @@ internal static partial class Av1IntraSuperblockEncoder modeInfoColumn, hasAbove, rightAvailable, - Av1PartitionType.None, + partitionType, transformSize, 0, 0, @@ -1456,7 +1466,7 @@ internal static partial class Av1IntraSuperblockEncoder modeInfoColumn, bottomAvailable, hasLeft, - Av1PartitionType.None, + partitionType, transformSize, 0, 0, @@ -1464,74 +1474,22 @@ internal static partial class Av1IntraSuperblockEncoder 0); Span aboveStorage = workspace.GetReferenceSamples(0); - Span above = aboveStorage[1..]; Span leftStorage = workspace.GetReferenceSamples(1); - Span left = leftStorage[1..]; - - if (hasAbove) - { - reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y - 1) - .Slice(blockOrigin.X, blockWidth) - .CopyTo(above[..blockWidth]); - } - - if (hasLeft) - { - for (int row = 0; row < blockHeight; row++) - { - left[row] = reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y + row)[blockOrigin.X - 1]; - } - } - - int midpoint = 128 << (this.bitDepth.GetBitCount() - 8); - - // A missing edge repeats the closest perpendicular sample. Only a block with neither edge - // available uses the asymmetric midpoint offsets that distinguish top from left. - if (!hasAbove) - { - above[..blockWidth].Fill(hasLeft ? left[0] : TOperator.CreateSample(midpoint - 1)); - } - - if (!hasLeft) - { - left[..blockHeight].Fill(hasAbove ? above[0] : TOperator.CreateSample(midpoint + 1)); - } - - if (hasTopRight) - { - reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y - 1) - .Slice(blockOrigin.X + blockWidth, blockWidth) - .CopyTo(above.Slice(blockWidth, blockWidth)); - } - else - { - above.Slice(blockWidth, blockWidth).Fill(above[blockWidth - 1]); - } - - if (hasBottomLeft) - { - for (int row = blockHeight; row < blockHeight * 2; row++) - { - left[row] = reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y + row)[blockOrigin.X - 1]; - } - } - else - { - left.Slice(blockHeight, blockHeight).Fill(left[blockHeight - 1]); - } - - // Zone-two projection and Paeth address the common corner immediately before both prepared edges. - // When an edge is unavailable AV1 derives that corner from the closest coded edge. - TSample corner = hasAbove && hasLeft - ? reconstructionPlane.DangerousGetRowSpan(blockOrigin.Y - 1)[blockOrigin.X - 1] - : hasAbove - ? above[0] - : hasLeft - ? left[0] - : TOperator.CreateSample(midpoint); + PrepareReferenceSamples( + reconstructionPlane, + blockOrigin, + blockWidth, + blockHeight, + hasLeft, + hasAbove, + hasTopRight, + hasBottomLeft, + this.bitDepth, + aboveStorage, + leftStorage); - aboveStorage[0] = corner; - leftStorage[0] = corner; + ReadOnlySpan above = aboveStorage.Slice(1, blockWidth + blockHeight); + ReadOnlySpan left = leftStorage.Slice(1, blockWidth + blockHeight); Av1TransformBlockContext blockContext = Av1TileWriter.GetTransformBlockContexts( Av1ComponentType.Luminance, @@ -2536,6 +2494,7 @@ internal static partial class Av1IntraSuperblockEncoder ((transformRow4x4 + transformSize.Get4x4HighCount()) << subsamplingY) < macroBlock.Tile.ModeInfoRowEnd; + Av1PartitionType partitionType = macroBlock.GetRelativeModeInfo(0).Block.PartitionType; bool hasTopRight = Av1IntraReferenceAvailability.HasTopRight( this.picture.Sequence.SequenceHeader.SuperblockSize, blockSize, @@ -2543,7 +2502,7 @@ internal static partial class Av1IntraSuperblockEncoder modeInfoColumn, hasAbove, rightAvailable, - Av1PartitionType.None, + partitionType, transformSize, transformRow4x4, transformColumn4x4, @@ -2557,15 +2516,17 @@ internal static partial class Av1IntraSuperblockEncoder modeInfoColumn, bottomAvailable, hasLeft, - Av1PartitionType.None, + partitionType, transformSize, transformRow4x4, transformColumn4x4, subsamplingX, subsamplingY); - Span above = aboveStorage.Slice(1, transformWidth * 2); - Span left = leftStorage.Slice(1, transformHeight * 2); + // Rectangular transforms project as far as width + height - 1 on either edge. + // The existing reference storage already holds this maximum; no additional scratch is needed. + Span above = aboveStorage.Slice(1, transformWidth + transformHeight); + Span left = leftStorage.Slice(1, transformWidth + transformHeight); if (hasAbove) { if (transformRow > 0) @@ -2613,6 +2574,7 @@ internal static partial class Av1IntraSuperblockEncoder left[..transformHeight].Fill(hasAbove ? above[0] : TOperator.CreateSample(midpoint + 1)); } + int topRightCount = hasTopRight ? Math.Min(transformWidth, transformHeight) : 0; if (hasTopRight) { if (transformRow > 0) @@ -2620,26 +2582,26 @@ internal static partial class Av1IntraSuperblockEncoder candidateReconstruction .Slice( ((rowOffset - 1) * planeBlockWidth) + columnOffset + transformWidth, - transformWidth) + topRightCount) .CopyTo(above[transformWidth..]); } else { reconstructionPlane.DangerousGetRowSpan(planeBlockOrigin.Y - 1) - .Slice(planeBlockOrigin.X + columnOffset + transformWidth, transformWidth) + .Slice(planeBlockOrigin.X + columnOffset + transformWidth, topRightCount) .CopyTo(above[transformWidth..]); } } - else - { - above[transformWidth..].Fill(above[transformWidth - 1]); - } + int topCount = transformWidth + topRightCount; + above[topCount..].Fill(above[topCount - 1]); + + int bottomLeftCount = hasBottomLeft ? Math.Min(transformHeight, transformWidth) : 0; if (hasBottomLeft) { if (transformColumn > 0) { - for (int row = transformHeight; row < transformHeight * 2; row++) + for (int row = transformHeight; row < transformHeight + bottomLeftCount; row++) { left[row] = candidateReconstruction[ ((rowOffset + row) * planeBlockWidth) + columnOffset - 1]; @@ -2647,17 +2609,16 @@ internal static partial class Av1IntraSuperblockEncoder } else { - for (int row = transformHeight; row < transformHeight * 2; row++) + for (int row = transformHeight; row < transformHeight + bottomLeftCount; row++) { left[row] = reconstructionPlane .DangerousGetRowSpan(planeBlockOrigin.Y + rowOffset + row)[planeBlockOrigin.X - 1]; } } } - else - { - left[transformHeight..].Fill(left[transformHeight - 1]); - } + + int leftCount = transformHeight + bottomLeftCount; + left[leftCount..].Fill(left[leftCount - 1]); // Only an interior transform corner belongs to decision scratch. Boundary corners continue // to read the already reconstructed neighboring block so candidate trials remain isolated. diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs index 3fce5999c3..ee667719f3 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs @@ -29,6 +29,98 @@ public class Av1EncoderFrameTests private const int Yuv422 = (int)Av1ColorFormat.Yuv422; private const int Yuv444 = (int)Av1ColorFormat.Yuv444; + [Theory] + [InlineData(EightBit, false, false)] + [InlineData(EightBit, false, true)] + [InlineData(EightBit, true, false)] + [InlineData(EightBit, true, true)] + [InlineData(TenBit, false, false)] + [InlineData(TenBit, false, true)] + [InlineData(TenBit, true, false)] + [InlineData(TenBit, true, true)] + [InlineData(TwelveBit, false, false)] + [InlineData(TwelveBit, false, true)] + [InlineData(TwelveBit, true, false)] + [InlineData(TwelveBit, true, true)] + public void RectangularIntraReferencesExtendTheLastAvailableSample(int bitDepthValue, bool transpose, bool extensionAvailable) + { + Av1BitDepth bitDepth = (Av1BitDepth)bitDepthValue; + if (bitDepth == Av1BitDepth.EightBit) + { + AssertRectangularIntraReferences(bitDepth, transpose, extensionAvailable); + } + else + { + AssertRectangularIntraReferences(bitDepth, transpose, extensionAvailable); + } + } + + private static void AssertRectangularIntraReferences( + Av1BitDepth bitDepth, + bool transpose, + bool extensionAvailable) + where TSample : unmanaged + where TOperator : struct, Av1IntraSuperblockEncoder.IBlockEncodingOperator + { + int width = transpose ? 16 : 4; + int height = transpose ? 4 : 16; + int scale = 1 << (bitDepth.GetBitCount() - 8); + using Buffer2D plane = Configuration.Default.MemoryAllocator.Allocate2D(33, 33); + for (int i = 0; i < 32; i++) + { + plane.DangerousGetRowSpan(0)[i + 1] = TOperator.CreateSample((10 + i) * scale); + plane.DangerousGetRowSpan(i + 1)[0] = TOperator.CreateSample((50 + i) * scale); + } + + plane.DangerousGetRowSpan(0)[0] = TOperator.CreateSample(100 * scale); + + // Native reconintra.c extends a four-sample edge through its four-sample neighbor, then + // repeats sample seven to cover the twenty samples required by a 4x16 directional ray. + // These explicit offsets also distinguish unavailable neighbors from available extension. + int[] shortEdge = extensionAvailable + ? [0, 1, 2, 3, 4, 5, 6, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7] + : [0, 1, 2, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3, 3]; + + int[] longEdge = extensionAvailable + ? [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19] + : [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 15, 15, 15, 15]; + + int[] expectedAbove = transpose ? longEdge : shortEdge; + int[] expectedLeft = transpose ? shortEdge : longEdge; + TSample poison = TOperator.CreateSample((1 << bitDepth.GetBitCount()) - 1); + TSample[] above = new TSample[23]; + TSample[] left = new TSample[23]; + above.AsSpan().Fill(poison); + left.AsSpan().Fill(poison); + + // The exact-sized interior includes the corner and twenty projected samples. Sentinel samples + // on either side detect writes outside the reference view, including the former 2*long-edge span. + Av1IntraSuperblockEncoder.ModeDecision.PrepareReferenceSamples( + plane.GetRegion(), + new Point(1, 1), + width, + height, + true, + true, + extensionAvailable, + extensionAvailable, + bitDepth, + above.AsSpan(1, 21), + left.AsSpan(1, 21)); + + Assert.Equal(poison, above[0]); + Assert.Equal(poison, left[0]); + Assert.Equal(poison, above[^1]); + Assert.Equal(poison, left[^1]); + Assert.Equal(TOperator.CreateSample(100 * scale), above[1]); + Assert.Equal(TOperator.CreateSample(100 * scale), left[1]); + for (int i = 0; i < 20; i++) + { + Assert.Equal(TOperator.CreateSample((10 + expectedAbove[i]) * scale), above[i + 2]); + Assert.Equal(TOperator.CreateSample((50 + expectedLeft[i]) * scale), left[i + 2]); + } + } + [Fact] public void EncodeUsesMultipleTilesWhenSingleTileWidthLimitIsExceeded() { diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs index 0e4a1eddea..93d16a8e6a 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1IntraSuperblockEncoderTests.cs @@ -2911,6 +2911,95 @@ public class Av1IntraSuperblockEncoderTests Assert.NotEqual(0, tileWriter.GetTileData(0).Length); } + /// + /// Verifies that mixed partition trials and final writing retain the decoder's reconstruction order. + /// + [Theory] + [InlineData(false)] + [InlineData(true)] + public void ProductionMixedPartitionsPreserveReconstructionOrder(bool transpose) + { + const int Size = 32; + const int QIndex = 4; + ObuColorConfig colorConfig = new() + { + IsMonochrome = true, + SubSamplingX = true, + SubSamplingY = true, + BitDepth = Av1BitDepth.EightBit + }; + + using Av1EncoderFrameBuffer source = new(Configuration.Default, Size, Size, 8, Av1ColorFormat.Yuv400, 0, 0); + using Av1EncoderFrameBuffer reconstruction = new(Configuration.Default, Size, Size, 8, Av1ColorFormat.Yuv400, 0, 0); + Buffer2DRegion sourcePlane = source.Frame.CodedView.GetPlane(Av1Plane.Y); + for (int y = 0; y < Size; y++) + { + for (int x = 0; x < Size; x++) + { + // The lower-right quadrant contains two different square surfaces beside one vertical + // surface. Transposition exercises the corresponding horizontal reconstruction order. + int value = x < 16 && y < 16 ? 128 + : y < 16 ? 16 + ((x - 16) * 12) + : x < 16 ? 16 + ((y - 16) * 12) + : x >= 24 ? 16 + ((x - 16) * 12) + : y < 24 ? 16 + ((x + y - 31) * 12) + : 16 + ((x - 8) * 12); + + sourcePlane.DangerousGetRowSpan(transpose ? x : y)[transpose ? y : x] = (byte)value; + } + } + + ClearPlane(reconstruction.Luma); + using Av1EncoderModeInfoBuffer modeInfo = new(Configuration.Default, Size, Size, disallow4x4AllFrames: false); + Av1PictureControlSet template = CreatePicture(modeInfo, colorConfig, use128x128Superblock: false, QIndex); + using Av1EncoderPictureBuffer picture = new( + Configuration.Default, template.Sequence.SequenceHeader, template.Parent.FrameHeader, Size, Size, disallow4x4AllFrames: false); + + using Av1EncoderCoefficientBuffer coefficients = new(Configuration.Default, template.Sequence.SequenceHeader, Size, Size); + using Av1EncoderSuperblockWorkspace superblockWorkspace = new(Configuration.Default); + using Av1EncoderBlockWorkspace blockWorkspace = new(Configuration.Default); + using Av1SymbolEncoder symbolEncoder = CreateTileSymbolEncoder(picture.Picture, 8192); + Av1TileEncoder tileWriter = new( + symbolEncoder, source.Frame, reconstruction.Frame, picture.Picture, coefficients, superblockWorkspace, blockWorkspace, effort: 9); + + byte[] payload = WriteCompleteTileObu(picture.Picture, tileWriter, Size, Size); + using Av1Decoder decoder = new(Configuration.Default); + decoder.DecodeSequenceReference(payload, null, null); + Av1FrameInfo decodedInfo = Assert.IsType(decoder.FrameInfo); + Av1FrameBuffer decodedFrame = Assert.IsType>(decoder.FrameBuffer); + Buffer2DRegion decodedPlane = decodedFrame.DeriveBlockPointer(Av1Plane.Y, 0, 0); + Buffer2DRegion retainedPlane = reconstruction.Frame.CodedView.GetPlane(Av1Plane.Y); + bool hasMixedPartition = false; + for (int y = 0; y < Size; y++) + { + Assert.Equal(retainedPlane.DangerousGetRowSpan(y).ToArray(), decodedPlane.DangerousGetRowSpan(y).ToArray()); + for (int x = 0; x < Size; x += 4) + { + Point position = new(x >> 2, y >> 2); + Av1PartitionType partition = decodedInfo.GetModeInfoAt(position).PartitionType; + + // Interior 4x4 entries alias the block origin through the live grid; unused allocation + // slots may still contain rejected trial data and are not retained block state. + int allocationIndex = picture.Picture.ModeInfoGrid.Span[(position.Y * picture.Picture.ModeInfoStride) + position.X]; + Assert.Equal(partition, picture.Picture.ModeInfoAllocation.Span[allocationIndex].Block.PartitionType); + hasMixedPartition |= partition is Av1PartitionType.HorizontalA or Av1PartitionType.HorizontalB + or Av1PartitionType.VerticalA or Av1PartitionType.VerticalB; + } + } + + Assert.True(hasMixedPartition); + string directory = Path.Combine( + TestEnvironment.ActualOutputDirectoryFullPath, "Heif", "Av1", nameof(this.ProductionMixedPartitionsPreserveReconstructionOrder)); + + Directory.CreateDirectory(directory); + File.WriteAllBytes(Path.Combine(directory, $"{transpose}.obu"), payload); + using FileStream raw = File.Create(Path.Combine(directory, $"{transpose}.retained.yuv")); + for (int y = 0; y < Size; y++) + { + raw.Write(retainedPlane.DangerousGetRowSpan(y)); + } + } + private static byte[] WriteCompleteTileObu( Av1PictureControlSet pictureTemplate, IAv1TileWriter tileWriter,