diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index 43a21f8f04..b925710b33 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -754,6 +754,21 @@ Final decoder stream, presentation, and public-registration evidence on 2026-09- `git diff --check` passes, and `.gitattributes` is unchanged. Every VSTest invocation returned normally with no surviving test host and no Windows application-error dialog. +SIMD traversal consistency evidence on 2026-09-02: + +- [x] The shared `Numerics` vector-count helpers now cover same-lane spans and all fixed hardware widths. + AV1 decoder and current encoder hot paths use those helpers for complete-vector traversal instead of + repeating local modulo or last-vector calculations. Reverse-source indexing and algorithm-specific + partial-output groups remain explicit because they are not vector-count calculations. +- [x] Forward quantization, palette prediction, and scaled inter prediction construct width-specific SIMD + constants only when at least one vector batch will execute. Narrower dispatch tiers consume only the + remainder left by wider tiers before the scalar tail. +- [x] The net11.0 Release production assembly builds with zero warnings and zero errors. The test project + builds with zero errors while retaining the existing repository warning set. Roslynk reports zero + compiler errors, `git diff --check` passes, and `.gitattributes` is unchanged. + Foreground VSTest passes 63 of 63 focused quantizer, forward-transform, CDEF, restoration, palette, + intra, inter, film-grain, and super-resolution cases. + Decoder exit gate: - [x] Every supported native format and AV1 tool has exact current-main libaom production-path evidence. @@ -768,28 +783,49 @@ Writer primitives are not an encoder. The public encoder remains incomplete unti ### 5. Define and enforce the encoder contract +- [x] Use official libaom `main` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` as the encoder syntax, probability-model, transform, quantization, filtering, and bitstream reference. +- [x] Use the existing PNG, TIFF, and JPEG encoders as the ImageSharp architecture reference: generic `Image` input, encoder options taking precedence over converted format metadata and codec defaults, allocator-owned temporary storage, and deterministic disposal. +- [x] Treat source pixel type, source alpha representation, and decoded source bit depth as conversion inputs, never as output-eligibility checks. Do not pre-scan pixels before encoding. +- [x] Resolve output configuration once from explicit encoder options, converted `HeifMetadata`, and AV1 defaults in that order. Sanitize only combinations that cannot describe a legal requested output, and never write resolved values back to source metadata. - [ ] Finalize observable options for quality, effort, lossless mode, bit depth, chroma subsampling, alpha quality, metadata, and bounded sequences. - [ ] Preserve high-bit-depth source precision through 16-bit RGB and native 10/12-bit component planes. -- [ ] Reject unsupported combinations at the public boundary before writing output. +- [ ] Reject only genuinely unsupported output combinations at the public boundary before writing output. - [ ] Register only capabilities that the completed encoder proves. +Encoder data-flow contract: + +1. Resolve immutable frame and sequence output settings before allocating codec state. +2. Convert each generic `ImageFrame` once through `PixelOperations` and the SIMD-first HEIF planar converter into native 8, 10, or 12-bit planes. Alpha is encoded as an auxiliary image when requested by the resolved output contract; it is not discarded through a source scan. +3. Reuse allocator-owned plane, row, block, transform, quantization, entropy, and reconstruction workspaces for the complete frame. No active path may allocate per row, block, transform, scanline, or SIMD tail. +4. Analyze and encode tiles directly from those planes, retaining reconstructed reference frames only for the bounded sequence lifetime. +5. Stream OBUs and container extents through allocator-backed chunked storage. Every ownership transfer is explicit, every owner is disposed exactly once, and no `ToArray` or file-sized copy crosses a layer boundary. +6. Iterate image frames using ImageSharp frame metadata and format-connecting metadata. Root-frame-only behavior is permitted only for an explicitly static output contract. + +Encoder verification contract: + +- Exercise source pixel formats independently from requested AV1 bit depth, chroma subsampling, alpha, and lossless/lossy mode. +- Run every SIMD operator through FeatureTestRunner at Vector512, Vector256, Vector128, and scalar tiers against an independent scalar oracle shaped from the same libaom revision. +- Cover discontiguous allocator buffers, constrained memory groups, cancellation, non-seekable output, multiple extents, auxiliary alpha, and bounded sequences. +- Validate produced AV1 payloads with current-main libaom and compare native planes before using ImageSharp self-decode as supplemental container coverage. + ### 6. Build the complete AV1 frame encoder - [~] SIMD-first RGB-to-native-plane conversion exists locally. - [~] Forward transform families and transform workspace exist locally. - [~] Symbol writer, coefficient writer, and tile writer fragments exist locally. -- [ ] Connect a frame-owned encoder lifecycle using ImageSharp allocators and pools. -- [ ] Write compliant temporal delimiter, sequence header, frame header, tile group, metadata, and padding OBUs as required. +- [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. Allocator-backed luma and 4:2:0 chroma extension passed direct net11 VSTest; the containing encode operation still needs to connect matching plane rents with ordinary `using` lifetimes. +- [~] Temporal delimiter, sequence header, frame header, and combined-frame tile-group writing exist locally. The remaining required metadata, padding, and encoder-wide syntax paths are not complete. - [ ] Implement superblock and partition analysis for every permitted block size and partition. - [ ] Implement intra mode search, chroma mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. - [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. -- [ ] Implement transform-size/type search, forward transform, quantization, coefficient optimization, and lossless behavior. +- [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain. - [ ] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. - [ ] Implement tile-local entropy coding and CDF update behavior. - [ ] Implement legal deblocking, CDEF, restoration, super-resolution, and film-grain signaling decisions. -- [ ] Remove per-transform and per-block managed allocations from active encoder paths. -- [ ] Use descending SIMD dispatch: Vector512, Vector256, Vector128, then scalar. -- [ ] Verify every SIMD operator with FeatureTestRunner and an independent scalar oracle shaped from the same current-main libaom behavior. +- [~] The coefficient symbol encoder now reuses tile-lifetime level and context workspaces instead of allocating per transform. Every remaining encoder fragment must be audited before it becomes active. +- [~] The planar conversion, forward transform, and forward quantizer use descending SIMD dispatch: Vector512, Vector256, Vector128, then scalar. Apply the same rule to every later hot-path family. +- [~] Forward-quantizer FeatureTestRunner and zero-allocation tests compare every hardware tier with an independent scan-order scalar oracle shaped from current-main libaom. Both passed direct net11 VSTest in Release. +- [~] The combined-frame writer now completes the byte-counted uncompressed frame header before starting the optional multi-tile tile-group flag, matching current libaom's separate frame-header and tile-group writers. A non-uniform two-tile round trip verifies the explicit boundaries, both tile payloads, and complete stream consumption through direct net11 VSTest in Release. ### 7. Write complete AVIF output diff --git a/src/ImageSharp/Common/Helpers/Numerics.cs b/src/ImageSharp/Common/Helpers/Numerics.cs index e5a6b45493..b1c6f29f19 100644 --- a/src/ImageSharp/Common/Helpers/Numerics.cs +++ b/src/ImageSharp/Common/Helpers/Numerics.cs @@ -1024,6 +1024,46 @@ internal static class Numerics where TVector : struct => (uint)span.Length / (uint)Vector512.Count; + /// + /// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type. + /// + /// The type of the span elements and vector lanes. + /// The given span. + /// Count of vectors that safely fit into the span. + public static nuint Vector128Count(this ReadOnlySpan span) + where TVector : struct + => (uint)span.Length / (uint)Vector128.Count; + + /// + /// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type. + /// + /// The type of the span elements and vector lanes. + /// The given span. + /// Count of vectors that safely fit into the span. + public static nuint Vector256Count(this ReadOnlySpan span) + where TVector : struct + => (uint)span.Length / (uint)Vector256.Count; + + /// + /// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type. + /// + /// The type of the span elements and vector lanes. + /// The given span. + /// Count of vectors that safely fit into the span. + public static nuint Vector512Count(this ReadOnlySpan span) + where TVector : struct + => (uint)span.Length / (uint)Vector512.Count; + + /// + /// Gets the count of vectors that safely fit into the given length. + /// + /// The type of the vector. + /// The given length. + /// Count of vectors that safely fit into the length. + public static nuint Vector128Count(int length) + where TVector : struct + => (uint)length / (uint)Vector128.Count; + /// /// Gets the count of vectors that safely fit into length. /// diff --git a/src/ImageSharp/Formats/Heif/Av1/Av1BitStreamWriter.cs b/src/ImageSharp/Formats/Heif/Av1/Av1BitStreamWriter.cs index 81d64c25e7..a7e193b11c 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Av1BitStreamWriter.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Av1BitStreamWriter.cs @@ -187,10 +187,11 @@ internal ref struct Av1BitStreamWriter } else { - uint extraBit = ((value + m) >> 1) - value; - uint k = (value + m - extraBit) >> 1; + // libaom partitions the upper values into a shorter prefix followed by the low bit of the offset from m. + uint offset = value - m; + uint k = m + (offset >> 1); this.WriteLiteral(k, w - 1); - this.WriteLiteral(extraBit, 1); + this.WriteLiteral(offset & 1, 1); } } @@ -236,7 +237,7 @@ internal ref struct Av1BitStreamWriter DebugGuard.IsTrue(Av1Math.Modulus8(this.BitPosition) == 0, "Writing of Tile Data only allowed on byte alignment"); int wordPosition = this.BitPosition >> 3; - if (this.span.Length <= wordPosition + tileData.Length) + if (this.span.Length < wordPosition + tileData.Length) { this.memory.GetSpan(wordPosition + tileData.Length); this.span = this.memory.GetEntireSpan(); diff --git a/src/ImageSharp/Formats/Heif/Av1/Color/Av1PresentationSampleBuffer.cs b/src/ImageSharp/Formats/Heif/Av1/Color/Av1PresentationSampleBuffer.cs index 68660a47b3..64b1904836 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Color/Av1PresentationSampleBuffer.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Color/Av1PresentationSampleBuffer.cs @@ -475,7 +475,9 @@ internal sealed class Av1PresentationSampleBuffer : IDisposabl ref ushort bottomSourceBase = ref MemoryMarshal.GetReference(bottomSource); ref ushort topDestinationBase = ref MemoryMarshal.GetReference(topDestination); ref ushort bottomDestinationBase = ref MemoryMarshal.GetReference(bottomDestination); - for (; x + Vector128.Count <= lastSource; x += Vector128.Count) + nuint vectorCount = topSource[..lastSource].Vector128Count(); + + for (; vectorCount > 0; vectorCount--, x += Vector128.Count) { Vector128 top0 = Vector128.LoadUnsafe(ref topSourceBase, (nuint)x); Vector128 top1 = Vector128.LoadUnsafe(ref topSourceBase, (nuint)(x + 1)); diff --git a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs index 7fa1e1e190..d9f96129c1 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs @@ -16,6 +16,11 @@ namespace SixLabors.ImageSharp.Formats.Heif.Av1.Entropy; /// internal class Av1SymbolEncoder : IDisposable { + /// + /// The largest coefficient-context plane required after AV1 removes the uncoded half of 64-point transforms. + /// + private const int MaximumCoefficientContextCount = (Av1Constants.MaxTransformSize / 2) * (Av1Constants.MaxTransformSize / 2); + /// /// The tile-adaptive intra-block-copy distribution. /// @@ -132,9 +137,14 @@ internal class Av1SymbolEncoder : IDisposable private bool isDisposed; /// - /// The configuration providing output and coefficient-context memory. + /// The reusable padded coefficient levels used to derive entropy contexts. + /// + private readonly Av1LevelBuffer levels; + + /// + /// The reusable raster-order coefficient contexts for one transform. /// - private readonly Configuration configuration; + private readonly IMemoryOwner coefficientContexts; /// /// The range writer producing the current tile payload. @@ -178,7 +188,8 @@ internal class Av1SymbolEncoder : IDisposable this.coefficientsBaseEndOfBlock = Av1DefaultDistributions.GetBaseEndOfBlock(qIndex); this.dcSign = Av1DefaultDistributions.GetDcSign(qIndex); this.endOfBlockExtra = Av1DefaultDistributions.GetEndOfBlockExtra(qIndex); - this.configuration = configuration; + this.levels = new(configuration); + this.coefficientContexts = configuration.MemoryAllocator.Allocate(MaximumCoefficientContextCount); this.writer = new(configuration, initialSize, updateCdf); this.baseQIndex = qIndex; } @@ -275,9 +286,11 @@ internal class Av1SymbolEncoder : IDisposable ref Av1SymbolWriter w = ref this.writer; - // AV1 omits high-frequency coefficients beyond 32 samples on every 64-point transform dimension. - using Av1LevelBuffer levels = new(this.configuration, new Size(width, height)); - Span coefficientContexts = new sbyte[width * height]; + // AV1 omits high-frequency coefficients beyond 32 samples on every 64-point transform dimension. The tile + // owns maximum-sized workspaces so repeated transform coding changes only their active views. + this.levels.Reset(new Size(width, height)); + Span coefficientContexts = this.coefficientContexts.Memory.Span[..(width * height)]; + coefficientContexts.Clear(); Guard.MustBeLessThan((int)transformSizeContext, (int)Av1TransformSize.AllSizes, nameof(transformSizeContext)); @@ -288,7 +301,7 @@ internal class Av1SymbolEncoder : IDisposable return 0; } - levels.Initialize(coefficientBuffer); + this.levels.Initialize(coefficientBuffer); if (componentType == Av1ComponentType.Luminance) { this.WriteTransformType(transformType, transformSize, useReducedTransformSet, this.baseQIndex, filterIntraMode, intraDirection); @@ -296,14 +309,14 @@ internal class Av1SymbolEncoder : IDisposable this.WriteEndOfBlockPosition(endOfBlock, componentType, transformClass, transformSize, transformSizeContext); - Av1SymbolContextHelper.GetNzMapContexts(levels, scan, endOfBlock, transformSize, transformClass, coefficientContexts); + Av1SymbolContextHelper.GetNzMapContexts(this.levels, scan, endOfBlock, transformSize, transformClass, coefficientContexts); int limitedTransformSizeContext = Math.Min((int)transformSizeContext, (int)Av1TransformSize.Size32x32); for (c = endOfBlock - 1; c >= 0; --c) { short pos = scan[c]; int v = coefficientBuffer[pos]; short coeffContext = coefficientContexts[pos]; - Point position = levels.GetPosition(pos); + Point position = this.levels.GetPosition(pos); int level = Math.Abs(v); if (c == endOfBlock - 1) @@ -319,7 +332,7 @@ internal class Av1SymbolEncoder : IDisposable { // Base-range symbols extend levels above the two base levels in fixed-size chunks. int baseRange = level - 1 - Av1Constants.BaseLevelsCount; - int baseRangeContext = Av1SymbolContextHelper.GetBaseRangeContext(levels, position, transformClass); + int baseRangeContext = Av1SymbolContextHelper.GetBaseRangeContext(this.levels, position, transformClass); for (int idx = 0; idx < Av1Constants.CoefficientBaseRange; idx += Av1Constants.BaseRangeSizeMinus1) { int k = Math.Min(baseRange - idx, Av1Constants.BaseRangeSizeMinus1); @@ -429,6 +442,8 @@ internal class Av1SymbolEncoder : IDisposable { if (!this.isDisposed) { + this.coefficientContexts.Dispose(); + this.levels.Dispose(); this.writer.Dispose(); this.isDisposed = true; } diff --git a/src/ImageSharp/Formats/Heif/Av1/IAv1TileWriter.cs b/src/ImageSharp/Formats/Heif/Av1/IAv1TileWriter.cs index 1e2552b8b3..ade1acffa3 100644 --- a/src/ImageSharp/Formats/Heif/Av1/IAv1TileWriter.cs +++ b/src/ImageSharp/Formats/Heif/Av1/IAv1TileWriter.cs @@ -9,11 +9,11 @@ namespace SixLabors.ImageSharp.Formats.Heif.Av1; internal interface IAv1TileWriter { /// - /// Write the information for a single tile. + /// Gets the encoded bytes for a single tile. /// - /// The index of the tile that is to be read. + /// The index of the encoded tile. /// /// The bytes of encoded data in the bitstream dedicated to this tile. /// - Span WriteTile(int tileNum); + ReadOnlySpan GetTileData(int tileNum); } diff --git a/src/ImageSharp/Formats/Heif/Av1/OpenBitstreamUnit/ObuWriter.cs b/src/ImageSharp/Formats/Heif/Av1/OpenBitstreamUnit/ObuWriter.cs index a592685ac1..00244d3216 100644 --- a/src/ImageSharp/Formats/Heif/Av1/OpenBitstreamUnit/ObuWriter.cs +++ b/src/ImageSharp/Formats/Heif/Av1/OpenBitstreamUnit/ObuWriter.cs @@ -29,7 +29,7 @@ internal class ObuWriter // The allocation expands when necessary; this initial size avoids repeated growth for // the small headers and tiles produced by the current still-image encoder. int initialBufferSize = 2000; - AutoExpandingMemory buffer = new(configuration, initialBufferSize); + using AutoExpandingMemory buffer = new(configuration, initialBufferSize); Av1BitStreamWriter writer = new(buffer); WriteObuHeaderAndSize(stream, ObuType.TemporalDelimiter, []); @@ -173,9 +173,6 @@ internal class ObuWriter { // AV1 fixes this RGB identity-matrix combination to full-range 4:4:4 and omits // the range and subsampling fields used by YUV configurations. - colorConfig.ColorRange = true; - colorConfig.SubSamplingX = false; - colorConfig.SubSamplingY = false; } else { @@ -335,13 +332,18 @@ internal class ObuWriter else { int startSuperBlock = 0; - int i = 0; - for (; startSuperBlock < superblockColumnCount; i++) + for (int i = 0; i < tileInfo.TileColumnCount; i++) { - uint widthInSuperBlocks = (uint)((tileInfo.TileColumnStartModeInfo[i] >> superblockShift) - startSuperBlock); + int endSuperBlock = i == tileInfo.TileColumnCount - 1 + ? superblockColumnCount + : tileInfo.TileColumnStartModeInfo[i + 1] >> superblockShift; + + // The stored terminal boundary is clipped to the visible mode-info width. libaom retains the exact + // superblock endpoint, so the final tile uses the derived frame-wide superblock count instead. + uint widthInSuperBlocks = (uint)(endSuperBlock - startSuperBlock); uint maxWidth = (uint)Math.Min(superblockColumnCount - startSuperBlock, tileInfo.MaxTileWidthSuperblock); writer.WriteNonSymmetric(widthInSuperBlocks - 1, maxWidth); - startSuperBlock += (int)widthInSuperBlocks; + startSuperBlock = endSuperBlock; } if (startSuperBlock != superblockColumnCount) @@ -350,12 +352,17 @@ internal class ObuWriter } startSuperBlock = 0; - for (i = 0; startSuperBlock < superblockRowCount; i++) + for (int i = 0; i < tileInfo.TileRowCount; i++) { - uint heightInSuperBlocks = (uint)((tileInfo.TileRowStartModeInfo[i] >> superblockShift) - startSuperBlock); + int endSuperBlock = i == tileInfo.TileRowCount - 1 + ? superblockRowCount + : tileInfo.TileRowStartModeInfo[i + 1] >> superblockShift; + + // As with columns, the final visible mode-info boundary may end inside its containing superblock. + uint heightInSuperBlocks = (uint)(endSuperBlock - startSuperBlock); uint maxHeight = (uint)Math.Min(superblockRowCount - startSuperBlock, tileInfo.MaxTileHeightSuperblock); writer.WriteNonSymmetric(heightInSuperBlocks - 1, maxHeight); - startSuperBlock += (int)heightInSuperBlocks; + startSuperBlock = endSuperBlock; } if (startSuperBlock != superblockRowCount) @@ -521,6 +528,11 @@ internal class ObuWriter private static void WriteTileGroup(ref Av1BitStreamWriter writer, ObuTileGroupHeader tileInfo, IAv1TileWriter tileWriter) { int tileCount = tileInfo.TileColumnCount * tileInfo.TileRowCount; + + // libaom starts the tile-group header at the next byte after the uncompressed frame header. This + // boundary is required before the optional flag because the flag belongs to tile_group_obu syntax. + AlignToByteBoundary(ref writer); + if (tileCount > 1) { // A combined OBU_FRAME has implicit complete-frame tile bounds. The reference decoder still @@ -544,7 +556,7 @@ internal class ObuWriter int tileCount = tileInfo.TileColumnCount * tileInfo.TileRowCount; for (int tileNum = 0; tileNum < tileCount; tileNum++) { - Span tileData = tileWriter.WriteTile(tileNum); + ReadOnlySpan tileData = tileWriter.GetTileData(tileNum); if (tileNum != tileCount - 1 && tileCount > 1) { writer.WriteLittleEndian((uint)tileData.Length - 1U, tileInfo.TileSizeBytes); diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs new file mode 100644 index 0000000000..75c12fdacd --- /dev/null +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs @@ -0,0 +1,334 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using SixLabors.ImageSharp.Formats.Heif.Components; +using SixLabors.ImageSharp.Memory; + +namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; + +/// +/// Provides non-owning visible and coded views over operation-scoped AV1 component planes. +/// +/// The native unsigned sample storage type. +internal readonly struct Av1EncoderFrame + where TSample : unmanaged +{ + /// + /// The base-two alignment exponent applied to coded frame dimensions. + /// + private const int CodedDimensionAlignmentLog2 = 3; + + /// + /// The base-two alignment exponent applied to the physical luma row stride. + /// + private const int LumaStrideAlignmentLog2 = 5; + + /// + /// The physical luma border required by non-resized all-intra encoding. + /// + public const int LumaBorder = 64; + + /// + /// Initializes a new instance of the struct for a monochrome frame. + /// + /// The coded luma region inside the bordered plane owned by the encode operation. + /// The visible luma width. + /// The visible luma height. + /// The native component precision. + public Av1EncoderFrame(Buffer2DRegion luma, int width, int height, int bitDepth) + : this(luma, default, default, width, height, bitDepth, Av1ColorFormat.Yuv400, 0, 0) + { + } + + /// + /// Initializes a new instance of the struct for a color frame. + /// + /// The coded luma region inside the bordered plane owned by the encode operation. + /// The coded blue-difference region inside its bordered plane. + /// The coded red-difference region inside its bordered plane. + /// The visible luma width. + /// The visible luma height. + /// The native component precision. + /// The native luma and chroma sampling layout. + /// The horizontal chroma position in half-luma-sample units. + /// The vertical chroma position in half-luma-sample units. + public Av1EncoderFrame( + Buffer2DRegion luma, + Buffer2DRegion chromaBlue, + Buffer2DRegion chromaRed, + int width, + int height, + int bitDepth, + Av1ColorFormat colorFormat, + int chromaPositionX, + int chromaPositionY) + { + this.Width = width; + this.Height = height; + this.LumaBitDepth = bitDepth; + this.ChromaBitDepth = bitDepth; + this.IsMonochrome = colorFormat == Av1ColorFormat.Yuv400; + this.ChromaSubsamplingX = colorFormat is Av1ColorFormat.Yuv420 or Av1ColorFormat.Yuv422 ? 1 : 0; + this.ChromaSubsamplingY = colorFormat == Av1ColorFormat.Yuv420 ? 1 : 0; + this.ChromaPositionX = chromaPositionX; + this.ChromaPositionY = chromaPositionY; + this.CodedWidth = luma.Width; + this.CodedHeight = luma.Height; + int visibleChromaWidth = (width + this.ChromaSubsamplingX) >> this.ChromaSubsamplingX; + int visibleChromaHeight = (height + this.ChromaSubsamplingY) >> this.ChromaSubsamplingY; + Buffer2DRegion visibleChromaBlue = this.IsMonochrome + ? default + : chromaBlue.GetSubRegion(0, 0, visibleChromaWidth, visibleChromaHeight); + + Buffer2DRegion visibleChromaRed = this.IsMonochrome + ? default + : chromaRed.GetSubRegion(0, 0, visibleChromaWidth, visibleChromaHeight); + + this.View = new PlanarView( + luma.GetSubRegion(0, 0, width, height), + visibleChromaBlue, + visibleChromaRed, + width, + height, + bitDepth, + colorFormat, + chromaPositionX, + chromaPositionY); + + this.CodedView = new PlanarView( + luma, + chromaBlue, + chromaRed, + this.CodedWidth, + this.CodedHeight, + bitDepth, + colorFormat, + chromaPositionX, + chromaPositionY); + } + + /// + /// Gets the writable component-plane view used by closed generic conversion and coding operations. + /// + public PlanarView View { get; } + + /// + /// Gets the writable coded component planes used by block coding and reconstruction. + /// + public PlanarView CodedView { get; } + + /// + /// Gets the visible luma width. + /// + public int Width { get; } + + /// + /// Gets the visible luma height. + /// + public int Height { get; } + + /// + /// Gets the luma width rounded up to the fixed coding-block boundary. + /// + public int CodedWidth { get; } + + /// + /// Gets the luma height rounded up to the fixed coding-block boundary. + /// + public int CodedHeight { get; } + + /// + /// Gets the native luma sample precision. + /// + public int LumaBitDepth { get; } + + /// + /// Gets the native chroma sample precision. + /// + public int ChromaBitDepth { get; } + + /// + /// Gets a value indicating whether the frame contains only luma samples. + /// + public bool IsMonochrome { get; } + + /// + /// Gets the horizontal chroma subsampling shift. + /// + public int ChromaSubsamplingX { get; } + + /// + /// Gets the vertical chroma subsampling shift. + /// + public int ChromaSubsamplingY { get; } + + /// + /// Gets the horizontal chroma position in half-luma-sample units. + /// + public int ChromaPositionX { get; } + + /// + /// Gets the vertical chroma position in half-luma-sample units. + /// + public int ChromaPositionY { get; } + + /// + /// Calculates the physical dimensions required for an all-intra component plane. + /// + /// The visible luma width. + /// The visible luma height. + /// The plane's horizontal subsampling shift. + /// The plane's vertical subsampling shift. + /// The physical plane dimensions, including its complete border and row padding. + public static Size GetPlaneBufferSize(int width, int height, int subsamplingX, int subsamplingY) + { + int codedWidth = Av1Math.AlignPowerOf2(width, CodedDimensionAlignmentLog2); + int codedHeight = Av1Math.AlignPowerOf2(height, CodedDimensionAlignmentLog2); + + // libaom aligns the complete luma row before deriving a subsampled plane's stride. + // Aligning chroma independently would produce a different physical layout for narrow or odd-sized frames. + int lumaStride = Av1Math.AlignPowerOf2(codedWidth + (2 * LumaBorder), LumaStrideAlignmentLog2); + int planeStride = lumaStride >> subsamplingX; + int planeBorderHeight = LumaBorder >> subsamplingY; + return new Size(planeStride, (codedHeight >> subsamplingY) + (2 * planeBorderHeight)); + } + + /// + /// Extends the visible edge samples through the coded padding. + /// + public void ExtendBorders() + => this.CodedView.ExtendBorders(this.Width, this.Height); + + /// + /// Replicates the visible edge samples through a plane's complete physical border. + /// + private static void ExtendPlane(Buffer2DRegion plane, int visibleWidth, int visibleHeight) + { + Buffer2D buffer = plane.Buffer; + Rectangle bounds = plane.Bounds; + for (int y = 0; y < visibleHeight; y++) + { + Span row = buffer.DangerousGetRowSpan(bounds.Y + y); + + // libaom fills both physical borders and the right-hand coded alignment from the nearest visible sample. + row[..bounds.X].Fill(row[bounds.X]); + row[(bounds.X + visibleWidth)..].Fill(row[bounds.X + visibleWidth - 1]); + } + + // Horizontal extension runs first so copying the first and last visible rows also initializes both corners. + ReadOnlySpan firstVisibleRow = buffer.DangerousGetRowSpan(bounds.Y); + for (int y = 0; y < bounds.Y; y++) + { + firstVisibleRow.CopyTo(buffer.DangerousGetRowSpan(y)); + } + + ReadOnlySpan finalVisibleRow = buffer.DangerousGetRowSpan(bounds.Y + visibleHeight - 1); + for (int y = bounds.Y + visibleHeight; y < buffer.Height; y++) + { + finalVisibleRow.CopyTo(buffer.DangerousGetRowSpan(y)); + } + } + + /// + /// Provides a non-owning component-plane view for generic hot-path operations. + /// + internal readonly struct PlanarView : IHeifPlanarSampleBuffer + { + /// + /// The writable luma plane. + /// + private readonly Buffer2DRegion luma; + + /// + /// The writable blue-difference plane, or the default region for monochrome frames. + /// + private readonly Buffer2DRegion chromaBlue; + + /// + /// The writable red-difference plane, or the default region for monochrome frames. + /// + private readonly Buffer2DRegion chromaRed; + + /// + /// Initializes a new instance of the struct. + /// + public PlanarView( + Buffer2DRegion luma, + Buffer2DRegion chromaBlue, + Buffer2DRegion chromaRed, + int width, + int height, + int bitDepth, + Av1ColorFormat colorFormat, + int chromaPositionX, + int chromaPositionY) + { + this.luma = luma; + this.chromaBlue = chromaBlue; + this.chromaRed = chromaRed; + this.Width = width; + this.Height = height; + this.LumaBitDepth = bitDepth; + this.ChromaBitDepth = bitDepth; + this.IsMonochrome = colorFormat == Av1ColorFormat.Yuv400; + this.ChromaSubsamplingX = colorFormat is Av1ColorFormat.Yuv420 or Av1ColorFormat.Yuv422 ? 1 : 0; + this.ChromaSubsamplingY = colorFormat == Av1ColorFormat.Yuv420 ? 1 : 0; + this.ChromaPositionX = chromaPositionX; + this.ChromaPositionY = chromaPositionY; + } + + /// + public int Width { get; } + + /// + public int Height { get; } + + /// + public int LumaBitDepth { get; } + + /// + public int ChromaBitDepth { get; } + + /// + public bool IsMonochrome { get; } + + /// + public int ChromaSubsamplingX { get; } + + /// + public int ChromaSubsamplingY { get; } + + /// + public int ChromaPositionX { get; } + + /// + public int ChromaPositionY { get; } + + /// + /// Replicates the visible component edges through the coded padding. + /// + /// The visible luma width. + /// The visible luma height. + public void ExtendBorders(int visibleWidth, int visibleHeight) + { + ExtendPlane(this.luma, visibleWidth, visibleHeight); + + if (!this.IsMonochrome) + { + int visibleChromaWidth = (visibleWidth + this.ChromaSubsamplingX) >> this.ChromaSubsamplingX; + int visibleChromaHeight = (visibleHeight + this.ChromaSubsamplingY) >> this.ChromaSubsamplingY; + ExtendPlane(this.chromaBlue, visibleChromaWidth, visibleChromaHeight); + ExtendPlane(this.chromaRed, visibleChromaWidth, visibleChromaHeight); + } + } + + /// + public Span GetLumaRowSpan(int row) => this.luma.DangerousGetRowSpan(row); + + /// + public Span GetChromaBlueRowSpan(int row) => this.chromaBlue.DangerousGetRowSpan(row); + + /// + public Span GetChromaRedRowSpan(int row) => this.chromaRed.DangerousGetRowSpan(row); + } +} diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Cdef/Av1CdefFilter.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Cdef/Av1CdefFilter.cs index 35f378a588..2eabaac131 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Cdef/Av1CdefFilter.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Cdef/Av1CdefFilter.cs @@ -60,7 +60,8 @@ internal static partial class Av1CdefFilter int firstDestinationRow = destinationOffset + (row * destinationStride); int secondDestinationRow = firstDestinationRow + destinationStride; int column = 0; - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 first = Vector128.LoadUnsafe(ref sourceBase, (nuint)(firstSourceRow + column)); Vector128 second = Vector128.LoadUnsafe(ref sourceBase, (nuint)(secondSourceRow + column)); diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainNoise.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainNoise.cs index 847204f1b1..24f5dd87ed 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainNoise.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainNoise.cs @@ -264,11 +264,11 @@ internal static class Av1FilmGrainNoise int sampleRowOffset = row * sampleStride; int grainRowOffset = row * grainStride; int column = 0; - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); // Samples, grain, and scale indices share the same lane coordinate. No permutation is required between // the lookup, fixed-point multiply, clipping, and native-sample store. - for (; column <= vectorEnd; column += Vector256.Count) + for (; column < vectorEnd; column += Vector256.Count) { ref TSample destination = ref Unsafe.Add(ref sampleBase, sampleRowOffset + column); Vector256 source = Av1FilmGrainSampleOperations.Load8(ref destination); @@ -319,11 +319,11 @@ internal static class Av1FilmGrainNoise int sampleRowOffset = row * sampleStride; int grainRowOffset = row * grainStride; int column = 0; - int vectorEnd = width - Vector128.Count; + int vectorEnd = (int)(Numerics.Vector128Count(width) * (nuint)Vector128.Count); // Four scalar table reads assemble the scale vector; the rest of the normative grain equation remains // lane-wise, including interpolation for 10- and 12-bit coordinates. - for (; column <= vectorEnd; column += Vector128.Count) + for (; column < vectorEnd; column += Vector128.Count) { ref TSample destination = ref Unsafe.Add(ref sampleBase, sampleRowOffset + column); Vector128 source = Av1FilmGrainSampleOperations.Load4(ref destination); @@ -570,11 +570,11 @@ internal static class Av1FilmGrainNoise int chromaRowOffset = row * chromaStride; int grainRowOffset = row * grainStride; int column = 0; - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); // Each lane represents one chroma coordinate and its corresponding reconstructed-luma coordinate. The Q6 // luma/chroma blend is clamped to a legal sample code before it becomes a scaling-table index. - for (; column <= vectorEnd; column += Vector256.Count) + for (; column < vectorEnd; column += Vector256.Count) { ref TSample lumaSource = ref Unsafe.Add(ref lumaRow, column << subsamplingX); Vector256 averageLuma = Av1FilmGrainSampleOperations.LoadChromaLuma8(ref lumaSource, subsamplingX); @@ -701,11 +701,11 @@ internal static class Av1FilmGrainNoise int chromaRowOffset = row * chromaStride; int grainRowOffset = row * grainStride; int column = 0; - int vectorEnd = width - Vector128.Count; + int vectorEnd = (int)(Numerics.Vector128Count(width) * (nuint)Vector128.Count); // The four-lane path preserves the same coordinate alignment and Q6 scaling-index arithmetic. Only the // table read changes from a hardware gather to four scalar reads assembled into a vector. - for (; column <= vectorEnd; column += Vector128.Count) + for (; column < vectorEnd; column += Vector128.Count) { ref TSample lumaSource = ref Unsafe.Add(ref lumaRow, column << subsamplingX); Vector128 averageLuma = Av1FilmGrainSampleOperations.LoadChromaLuma4(ref lumaSource, subsamplingX); diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainOverlap.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainOverlap.cs index a6aaacc50b..3cd2be0b66 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainOverlap.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainOverlap.cs @@ -210,8 +210,8 @@ internal static class Av1FilmGrainOverlap Vector512 rounding = Vector512.Create(16); Vector512 minima = Vector512.Create(minimum); Vector512 maxima = Vector512.Create(maximum); - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 leftValues = Vector512.LoadUnsafe(ref leftBase, (nuint)column); Vector512 rightValues = Vector512.LoadUnsafe(ref rightBase, (nuint)column); @@ -257,8 +257,8 @@ internal static class Av1FilmGrainOverlap Vector256 rounding = Vector256.Create(16); Vector256 minima = Vector256.Create(minimum); Vector256 maxima = Vector256.Create(maximum); - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 leftValues = Vector256.LoadUnsafe(ref leftBase, (nuint)column); Vector256 rightValues = Vector256.LoadUnsafe(ref rightBase, (nuint)column); @@ -304,8 +304,8 @@ internal static class Av1FilmGrainOverlap Vector128 rounding = Vector128.Create(16); Vector128 minima = Vector128.Create(minimum); Vector128 maxima = Vector128.Create(maximum); - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 leftValues = Vector128.LoadUnsafe(ref leftBase, (nuint)column); Vector128 rightValues = Vector128.LoadUnsafe(ref rightBase, (nuint)column); diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopRestoration/Av1SelfGuidedFilter.Operations.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopRestoration/Av1SelfGuidedFilter.Operations.cs index efe950622c..b269bf9915 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopRestoration/Av1SelfGuidedFilter.Operations.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopRestoration/Av1SelfGuidedFilter.Operations.cs @@ -252,8 +252,8 @@ internal static partial class Av1SelfGuidedFilter Vector256 squareCarry = Vector256.Zero; Vector256 sumCarry = Vector256.Zero; int column = 0; - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { // Eight packed 16-bit samples become eight 32-bit lanes. The prefix scans mirror // the reference decoder's scan_32, and the replicated carry joins consecutive vector batches. @@ -323,8 +323,8 @@ internal static partial class Av1SelfGuidedFilter Vector128 squareCarry = Vector128.Zero; Vector128 sumCarry = Vector128.Zero; int column = 0; - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { // Loading through Vector64 avoids reading beyond the four samples owned by this // batch. Widening is normalized by the runtime for both x86 and Arm64 targets. @@ -827,11 +827,11 @@ internal static partial class Av1SelfGuidedFilter int roundingBits = SelfGuidedBits + ((row & 1) == 0 ? 5 : 4) - RestorationBits; Vector256 rounding = Vector256.Create(1 << (roundingBits - 1)); int column = 0; - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); // Filtered signals use Q4 precision. Projection applies the signaled Q7 weights to their difference from // the unfiltered Q4 sample, then performs the combined Q11 rounding shift once before clipping. - for (; column <= vectorEnd; column += Vector256.Count) + for (; column < vectorEnd; column += Vector256.Count) { Vector256 factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, row, vector); Vector256 means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, row, vector); @@ -886,11 +886,11 @@ internal static partial class Av1SelfGuidedFilter int roundingBits = SelfGuidedBits + ((row & 1) == 0 ? 5 : 4) - RestorationBits; Vector128 rounding = Vector128.Create(1 << (roundingBits - 1)); int column = 0; - int vectorEnd = width - Vector128.Count; + int vectorEnd = (int)(Numerics.Vector128Count(width) * (nuint)Vector128.Count); // The 128-bit path uses the same Q4/Q7 projection equation. The four-sample load and store are deliberately // 64 bits wide so a tightly strided destination row never requires writable padding. - for (; column <= vectorEnd; column += Vector128.Count) + for (; column < vectorEnd; column += Vector128.Count) { Vector128 factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, row, vector); Vector128 means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, row, vector); @@ -946,8 +946,8 @@ internal static partial class Av1SelfGuidedFilter int filteredRowOffset = row * width; int coefficientRowOffset = bufferOrigin + (row * bufferStride); int column = 0; - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, vector); Vector256 means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, vector); @@ -1002,8 +1002,8 @@ internal static partial class Av1SelfGuidedFilter int filteredRowOffset = row * width; int coefficientRowOffset = bufferOrigin + (row * bufferStride); int column = 0; - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, vector); Vector128 means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, vector); @@ -1255,8 +1255,8 @@ internal static partial class Av1SelfGuidedFilter int destinationRowOffset = row * destinationStride; int filteredRowOffset = row * width; int column = 0; - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector128 packed = Vector128.LoadUnsafe(ref sourceBase, (nuint)(sourceRowOffset + column)); Vector256 samples = Vector256.WidenLower(Vector256.Create(packed, Vector128.Zero)).AsInt32(); @@ -1348,8 +1348,8 @@ internal static partial class Av1SelfGuidedFilter int destinationRowOffset = row * destinationStride; int filteredRowOffset = row * width; int column = 0; - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { ref ushort sourceReference = ref Unsafe.Add(ref sourceBase, sourceRowOffset + column); Vector64 packed = Unsafe.As>(ref sourceReference); diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.Operator.cs new file mode 100644 index 0000000000..114b69d60e --- /dev/null +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.Operator.cs @@ -0,0 +1,205 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using System.Runtime.Intrinsics; + +namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; + +/// +/// Defines the arithmetic operators used by . +/// +internal static partial class Av1ForwardQuantizer +{ + /// + /// Defines one AV1 forward-quantization arithmetic contract across hardware widths. + /// + internal interface IForwardQuantizationOperator + { + /// + /// Quantizes four raster-order transform coefficients. + /// + /// The signed transform coefficients. + /// The positive rounding constant after transform-size scaling. + /// The Q16 reciprocal quantizer. + /// The Q3 reconstruction quantizer. + /// The transform-size quantization scale. + /// The signed reconstruction coefficients. + /// The signed entropy-coding coefficients. + public static abstract Vector128 Quantize( + Vector128 coefficients, + Vector128 rounding, + Vector128 quantizer, + Vector128 dequantizer, + int logScale, + out Vector128 dequantizedCoefficients); + + /// + /// Quantizes eight raster-order transform coefficients. + /// + /// The signed transform coefficients. + /// The positive rounding constant after transform-size scaling. + /// The Q16 reciprocal quantizer. + /// The Q3 reconstruction quantizer. + /// The transform-size quantization scale. + /// The signed reconstruction coefficients. + /// The signed entropy-coding coefficients. + public static abstract Vector256 Quantize( + Vector256 coefficients, + Vector256 rounding, + Vector256 quantizer, + Vector256 dequantizer, + int logScale, + out Vector256 dequantizedCoefficients); + + /// + /// Quantizes sixteen raster-order transform coefficients. + /// + /// The signed transform coefficients. + /// The positive rounding constant after transform-size scaling. + /// The Q16 reciprocal quantizer. + /// The Q3 reconstruction quantizer. + /// The transform-size quantization scale. + /// The signed reconstruction coefficients. + /// The signed entropy-coding coefficients. + public static abstract Vector512 Quantize( + Vector512 coefficients, + Vector512 rounding, + Vector512 quantizer, + Vector512 dequantizer, + int logScale, + out Vector512 dequantizedCoefficients); + + /// + /// Quantizes one transform coefficient. + /// + /// The signed transform coefficient. + /// The positive rounding constant after transform-size scaling. + /// The Q16 reciprocal quantizer. + /// The Q3 reconstruction quantizer. + /// The transform-size quantization scale. + /// The signed reconstruction coefficient. + /// The signed entropy-coding coefficient. + public static abstract int Quantize( + int coefficient, + int rounding, + int quantizer, + int dequantizer, + int logScale, + out int dequantizedCoefficient); + } + + /// + /// Implements libaom's fast no-matrix quantizer for lossy transform blocks. + /// + /// + /// Every SIMD overload preserves the scalar operation order: magnitude threshold, saturating round, Q16 reciprocal + /// multiply, transform-size shift, dequantization, and sign restoration. Each lane owns one raster-order coefficient. + /// + internal readonly struct FastQuantizationOperator : IForwardQuantizationOperator + { + /// + public static Vector128 Quantize( + Vector128 coefficients, + Vector128 rounding, + Vector128 quantizer, + Vector128 dequantizer, + int logScale, + out Vector128 dequantizedCoefficients) + { + Vector128 coefficientSign = Vector128.ShiftRightArithmetic(coefficients, 31); + Vector128 magnitude = Vector128.Abs(coefficients); + + // libaom retains equality at the dequantizer threshold. Reversing the comparison and complementing its mask + // expresses scaledMagnitude >= dequantizer with the vector operations available for every supported ISA. + Vector128 thresholdMask = ~Vector128.GreaterThan(dequantizer, magnitude << (1 + logScale)); + + // Clamp the rounded magnitude to 32,767 so the following Q16 product remains within a signed 32-bit lane. + Vector128 rounded = Vector128.Min(magnitude + rounding, Vector128.Create((int)short.MaxValue)); + Vector128 quantizedMagnitude = ((rounded * quantizer) >> (16 - logScale)) & thresholdMask; + + // XOR followed by subtraction restores the original sign without a lane-wise branch. + Vector128 quantized = (quantizedMagnitude ^ coefficientSign) - coefficientSign; + Vector128 dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; + dequantizedCoefficients = (dequantizedMagnitude ^ coefficientSign) - coefficientSign; + return quantized; + } + + /// + public static Vector256 Quantize( + Vector256 coefficients, + Vector256 rounding, + Vector256 quantizer, + Vector256 dequantizer, + int logScale, + out Vector256 dequantizedCoefficients) + { + Vector256 coefficientSign = Vector256.ShiftRightArithmetic(coefficients, 31); + Vector256 magnitude = Vector256.Abs(coefficients); + + // Preserve threshold equality by complementing dequantizer > scaledMagnitude. + Vector256 thresholdMask = ~Vector256.GreaterThan(dequantizer, magnitude << (1 + logScale)); + + // Clamp the rounded magnitude to 32,767 so the following Q16 product remains within a signed 32-bit lane. + Vector256 rounded = Vector256.Min(magnitude + rounding, Vector256.Create((int)short.MaxValue)); + Vector256 quantizedMagnitude = ((rounded * quantizer) >> (16 - logScale)) & thresholdMask; + + // Apply the input sign to both coded and reconstructed magnitudes without branching. + Vector256 quantized = (quantizedMagnitude ^ coefficientSign) - coefficientSign; + Vector256 dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; + dequantizedCoefficients = (dequantizedMagnitude ^ coefficientSign) - coefficientSign; + return quantized; + } + + /// + public static Vector512 Quantize( + Vector512 coefficients, + Vector512 rounding, + Vector512 quantizer, + Vector512 dequantizer, + int logScale, + out Vector512 dequantizedCoefficients) + { + Vector512 coefficientSign = Vector512.ShiftRightArithmetic(coefficients, 31); + Vector512 magnitude = Vector512.Abs(coefficients); + + // Preserve threshold equality by complementing dequantizer > scaledMagnitude. + Vector512 thresholdMask = ~Vector512.GreaterThan(dequantizer, magnitude << (1 + logScale)); + + // Clamp the rounded magnitude to 32,767 so the following Q16 product remains within a signed 32-bit lane. + Vector512 rounded = Vector512.Min(magnitude + rounding, Vector512.Create((int)short.MaxValue)); + Vector512 quantizedMagnitude = ((rounded * quantizer) >> (16 - logScale)) & thresholdMask; + + // Apply the input sign to both coded and reconstructed magnitudes without branching. + Vector512 quantized = (quantizedMagnitude ^ coefficientSign) - coefficientSign; + Vector512 dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; + dequantizedCoefficients = (dequantizedMagnitude ^ coefficientSign) - coefficientSign; + return quantized; + } + + /// + public static int Quantize( + int coefficient, + int rounding, + int quantizer, + int dequantizer, + int logScale, + out int dequantizedCoefficient) + { + int coefficientSign = coefficient >> 31; + int magnitude = (coefficient ^ coefficientSign) - coefficientSign; + int quantizedMagnitude = 0; + + // The scalar tail keeps the same threshold, clamp, and fixed-point operation order as every vector lane. + if (((long)magnitude << (1 + logScale)) >= dequantizer) + { + int rounded = Math.Min(magnitude + rounding, short.MaxValue); + quantizedMagnitude = (rounded * quantizer) >> (16 - logScale); + } + + int quantized = (quantizedMagnitude ^ coefficientSign) - coefficientSign; + int dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; + dequantizedCoefficient = (dequantizedMagnitude ^ coefficientSign) - coefficientSign; + return quantized; + } + } +} diff --git a/src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.cs b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.cs new file mode 100644 index 0000000000..6395598892 --- /dev/null +++ b/src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.cs @@ -0,0 +1,209 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using System.Runtime.CompilerServices; +using System.Runtime.InteropServices; +using System.Runtime.Intrinsics; +using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; + +namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; + +/// +/// Applies AV1 forward quantization to raster-order transform coefficients. +/// +internal static partial class Av1ForwardQuantizer +{ + /// + /// Quantizes one lossy transform block with libaom's fast no-matrix arithmetic. + /// + /// The raster-order forward-transform coefficients. + /// The raster-order entropy-coding coefficients. + /// The raster-order reconstruction coefficients. + /// The transform-block dimensions. + /// The compound transform type. + /// The segment quantizer index. + /// The plane DC quantizer adjustment. + /// The plane AC quantizer adjustment. + /// The coded sample bit depth. + /// The one-based end position in coefficient scan order. + public static ushort QuantizeLossy( + ReadOnlySpan coefficients, + Span quantizedCoefficients, + Span dequantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth) + => QuantizeLossy( + coefficients, + quantizedCoefficients, + dequantizedCoefficients, + transformSize, + transformType, + qIndex, + dcDeltaQ, + acDeltaQ, + bitDepth); + + /// + /// Applies a closed generic quantization operator across the widest available hardware widths. + /// + internal static ushort QuantizeLossy( + ReadOnlySpan coefficients, + Span quantizedCoefficients, + Span dequantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth) + where TOperator : struct, IForwardQuantizationOperator + { + int coefficientCount = transformSize.GetAdjusted().GetSize2d(); + int logScale = transformSize.GetScale(); + int dcDequantizer = Av1QuantizationLookup.GetDcQuant(qIndex, dcDeltaQ, bitDepth); + int acDequantizer = Av1QuantizationLookup.GetAcQuant(qIndex, acDeltaQ, bitDepth); + int dcQuantizer = (1 << 16) / dcDequantizer; + int acQuantizer = (1 << 16) / acDequantizer; + int dcRounding = RoundPowerOfTwo((64 * dcDequantizer) >> 7, logScale); + int acRounding = RoundPowerOfTwo((64 * acDequantizer) >> 7, logScale); + + ref int sourceBase = ref MemoryMarshal.GetReference(coefficients); + ref int quantizedBase = ref MemoryMarshal.GetReference(quantizedCoefficients); + ref int dequantizedBase = ref MemoryMarshal.GetReference(dequantizedCoefficients); + + // Raster coefficient zero is the only DC coefficient, so it is encoded once with the plane's DC constants + // before the AC-only SIMD traversal begins. + Unsafe.Add(ref quantizedBase, 0) = TOperator.Quantize( + Unsafe.Add(ref sourceBase, 0), + dcRounding, + dcQuantizer, + dcDequantizer, + logScale, + out Unsafe.Add(ref dequantizedBase, 0)); + + int i = 1; + + // Raster traversal keeps loads and stores contiguous. Each narrower tier resumes at the shared offset left by + // the previous tier, retaining vector execution for the widest possible remainder without overlapping lanes. + if (Vector512.IsHardwareAccelerated) + { + nuint vectorCount = coefficients[i..coefficientCount].Vector512Count(); + + if (vectorCount > 0) + { + // Width-specific constants are created only when at least one complete vector remains. + Vector512 rounding = Vector512.Create(acRounding); + Vector512 quantizer = Vector512.Create(acQuantizer); + Vector512 dequantizer = Vector512.Create(acDequantizer); + + for (; vectorCount > 0; vectorCount--, i += Vector512.Count) + { + Vector512 source = Unsafe.As>(ref Unsafe.Add(ref sourceBase, i)); + Vector512 quantized = TOperator.Quantize( + source, + rounding, + quantizer, + dequantizer, + logScale, + out Vector512 dequantized); + + Unsafe.As>(ref Unsafe.Add(ref quantizedBase, i)) = quantized; + Unsafe.As>(ref Unsafe.Add(ref dequantizedBase, i)) = dequantized; + } + } + } + + if (Vector256.IsHardwareAccelerated) + { + nuint vectorCount = coefficients[i..coefficientCount].Vector256Count(); + + if (vectorCount > 0) + { + // The shared offset exposes only the remainder left by wider lanes, so no coefficient is revisited. + Vector256 rounding = Vector256.Create(acRounding); + Vector256 quantizer = Vector256.Create(acQuantizer); + Vector256 dequantizer = Vector256.Create(acDequantizer); + + for (; vectorCount > 0; vectorCount--, i += Vector256.Count) + { + Vector256 source = Unsafe.As>(ref Unsafe.Add(ref sourceBase, i)); + Vector256 quantized = TOperator.Quantize( + source, + rounding, + quantizer, + dequantizer, + logScale, + out Vector256 dequantized); + + Unsafe.As>(ref Unsafe.Add(ref quantizedBase, i)) = quantized; + Unsafe.As>(ref Unsafe.Add(ref dequantizedBase, i)) = dequantized; + } + } + } + + if (Vector128.IsHardwareAccelerated) + { + nuint vectorCount = coefficients[i..coefficientCount].Vector128Count(); + + if (vectorCount > 0) + { + // The final SIMD tier consumes complete four-lane groups and leaves fewer than four coefficients. + Vector128 rounding = Vector128.Create(acRounding); + Vector128 quantizer = Vector128.Create(acQuantizer); + Vector128 dequantizer = Vector128.Create(acDequantizer); + + for (; vectorCount > 0; vectorCount--, i += Vector128.Count) + { + Vector128 source = Unsafe.As>(ref Unsafe.Add(ref sourceBase, i)); + Vector128 quantized = TOperator.Quantize( + source, + rounding, + quantizer, + dequantizer, + logScale, + out Vector128 dequantized); + + Unsafe.As>(ref Unsafe.Add(ref quantizedBase, i)) = quantized; + Unsafe.As>(ref Unsafe.Add(ref dequantizedBase, i)) = dequantized; + } + } + } + + // On SIMD-capable systems this loop receives only the final zero-to-three AC coefficients. + for (; i < coefficientCount; i++) + { + Unsafe.Add(ref quantizedBase, i) = TOperator.Quantize( + Unsafe.Add(ref sourceBase, i), + acRounding, + acQuantizer, + acDequantizer, + logScale, + out Unsafe.Add(ref dequantizedBase, i)); + } + + ReadOnlySpan scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan; + + // Quantized coefficients remain in raster order for reconstruction and entropy coding. A reverse scan finds + // the final nonzero position without another buffer, and normally exits on its first iteration at high quality. + for (int scanIndex = coefficientCount - 1; scanIndex >= 0; scanIndex--) + { + if (Unsafe.Add(ref quantizedBase, scan[scanIndex]) != 0) + { + return (ushort)(scanIndex + 1); + } + } + + return 0; + } + + /// + /// Applies libaom's positive round-power-of-two operation to one quantizer constant. + /// + [MethodImpl(MethodImplOptions.AggressiveInlining)] + private static int RoundPowerOfTwo(int value, int shift) + => shift == 0 ? value : (value + (1 << (shift - 1))) >> shift; +} diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DcIntraPredictor.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DcIntraPredictor.Operator.cs index 8dedd0fc4c..51ff071fde 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DcIntraPredictor.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DcIntraPredictor.Operator.cs @@ -316,8 +316,8 @@ internal static class Av1DcIntraPredictor // length without a separate dispatch tree and leaves only an incomplete final vector to scalar code. if (Vector512.IsHardwareAccelerated) { - int oneVectorFromEnd = samples.Length - Vector512.Count; - for (; index <= oneVectorFromEnd; index += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(samples.Length - index); + for (; vectorCount > 0; vectorCount--, index += Vector512.Count) { sum += TOperator.Sum(Vector512.LoadUnsafe(ref samplesBase, (nuint)index)); } @@ -325,8 +325,8 @@ internal static class Av1DcIntraPredictor if (Vector256.IsHardwareAccelerated) { - int oneVectorFromEnd = samples.Length - Vector256.Count; - for (; index <= oneVectorFromEnd; index += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(samples.Length - index); + for (; vectorCount > 0; vectorCount--, index += Vector256.Count) { sum += TOperator.Sum(Vector256.LoadUnsafe(ref samplesBase, (nuint)index)); } @@ -334,8 +334,8 @@ internal static class Av1DcIntraPredictor if (Vector128.IsHardwareAccelerated) { - int oneVectorFromEnd = samples.Length - Vector128.Count; - for (; index <= oneVectorFromEnd; index += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(samples.Length - index); + for (; vectorCount > 0; vectorCount--, index += Vector128.Count) { sum += TOperator.Sum(Vector128.LoadUnsafe(ref samplesBase, (nuint)index)); } @@ -362,8 +362,8 @@ internal static class Av1DcIntraPredictor if (Vector512.IsHardwareAccelerated) { - int oneVectorFromEnd = samples.Length - Vector512.Count; - for (; index <= oneVectorFromEnd; index += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(samples.Length - index); + for (; vectorCount > 0; vectorCount--, index += Vector512.Count) { sum += TOperator.Sum(Vector512.LoadUnsafe(ref samplesBase, (nuint)index)); } @@ -371,8 +371,8 @@ internal static class Av1DcIntraPredictor if (Vector256.IsHardwareAccelerated) { - int oneVectorFromEnd = samples.Length - Vector256.Count; - for (; index <= oneVectorFromEnd; index += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(samples.Length - index); + for (; vectorCount > 0; vectorCount--, index += Vector256.Count) { sum += TOperator.Sum(Vector256.LoadUnsafe(ref samplesBase, (nuint)index)); } @@ -380,8 +380,8 @@ internal static class Av1DcIntraPredictor if (Vector128.IsHardwareAccelerated) { - int oneVectorFromEnd = samples.Length - Vector128.Count; - for (; index <= oneVectorFromEnd; index += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(samples.Length - index); + for (; vectorCount > 0; vectorCount--, index += Vector128.Count) { sum += TOperator.Sum(Vector128.LoadUnsafe(ref samplesBase, (nuint)index)); } diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DirectionalIntraPredictor.Operations.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DirectionalIntraPredictor.Operations.cs index 9cc792a1a9..7266b8b02f 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DirectionalIntraPredictor.Operations.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DirectionalIntraPredictor.Operations.cs @@ -204,8 +204,8 @@ internal static partial class Av1DirectionalIntraPredictor // left by the wider path, so the row is written once without requiring padded destination storage. if (Vector512.IsHardwareAccelerated) { - int oneVectorFromEnd = validCount - Vector512.Count; - for (; index <= oneVectorFromEnd; index += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(validCount - index); + for (; vectorCount > 0; vectorCount--, index += Vector512.Count) { Vector512 left = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector512 right = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); @@ -215,8 +215,8 @@ internal static partial class Av1DirectionalIntraPredictor if (Vector256.IsHardwareAccelerated) { - int oneVectorFromEnd = validCount - Vector256.Count; - for (; index <= oneVectorFromEnd; index += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(validCount - index); + for (; vectorCount > 0; vectorCount--, index += Vector256.Count) { Vector256 left = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector256 right = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); @@ -226,8 +226,8 @@ internal static partial class Av1DirectionalIntraPredictor if (Vector128.IsHardwareAccelerated) { - int oneVectorFromEnd = validCount - Vector128.Count; - for (; index <= oneVectorFromEnd; index += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(validCount - index); + for (; vectorCount > 0; vectorCount--, index += Vector128.Count) { Vector128 left = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector128 right = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); @@ -312,8 +312,8 @@ internal static partial class Av1DirectionalIntraPredictor // weighted sum. The largest supported 12-bit sample therefore cannot overflow an intermediate lane. if (Vector512.IsHardwareAccelerated) { - int oneVectorFromEnd = validCount - Vector512.Count; - for (; index <= oneVectorFromEnd; index += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(validCount - index); + for (; vectorCount > 0; vectorCount--, index += Vector512.Count) { Vector512 left = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector512 right = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); @@ -323,8 +323,8 @@ internal static partial class Av1DirectionalIntraPredictor if (Vector256.IsHardwareAccelerated) { - int oneVectorFromEnd = validCount - Vector256.Count; - for (; index <= oneVectorFromEnd; index += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(validCount - index); + for (; vectorCount > 0; vectorCount--, index += Vector256.Count) { Vector256 left = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector256 right = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); @@ -334,8 +334,8 @@ internal static partial class Av1DirectionalIntraPredictor if (Vector128.IsHardwareAccelerated) { - int oneVectorFromEnd = validCount - Vector128.Count; - for (; index <= oneVectorFromEnd; index += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(validCount - index); + for (; vectorCount > 0; vectorCount--, index += Vector128.Count) { Vector128 left = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector128 right = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1NonDirectionalIntraPredictor.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1NonDirectionalIntraPredictor.Operator.cs index 59c5097077..25688ecc54 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1NonDirectionalIntraPredictor.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1NonDirectionalIntraPredictor.Operator.cs @@ -226,7 +226,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase // narrower paths consume any complete vectors left before the scalar tail handles the final columns. if (Vector512.IsHardwareAccelerated) { - int vectorizedColumns = width - (width % Vector512.Count); + int vectorizedColumns = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); if (vectorizedColumns > 0) { Vector512 topLeftVector = usesTopLeft ? Vector512.Create(topLeft) : default; @@ -255,7 +255,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase if (Vector256.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector256.Count); + int vectorizedColumns = (int)(Numerics.Vector256Count(remainingColumns) * (nuint)Vector256.Count); if (vectorizedColumns > 0) { Vector256 topLeftVector = usesTopLeft ? Vector256.Create(topLeft) : default; @@ -285,7 +285,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase if (Vector128.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector128.Count); + int vectorizedColumns = (int)(Numerics.Vector128Count(remainingColumns) * (nuint)Vector128.Count); if (vectorizedColumns > 0) { Vector128 topLeftVector = usesTopLeft ? Vector128.Create(topLeft) : default; @@ -355,7 +355,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase // column, so the same width-progressive traversal is valid without inter-lane packing or saturation. if (Vector512.IsHardwareAccelerated) { - int vectorizedColumns = width - (width % Vector512.Count); + int vectorizedColumns = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); if (vectorizedColumns > 0) { Vector512 topLeftVector = usesTopLeft ? Vector512.Create(topLeft) : default; @@ -384,7 +384,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase if (Vector256.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector256.Count); + int vectorizedColumns = (int)(Numerics.Vector256Count(remainingColumns) * (nuint)Vector256.Count); if (vectorizedColumns > 0) { Vector256 topLeftVector = usesTopLeft ? Vector256.Create(topLeft) : default; @@ -414,7 +414,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase if (Vector128.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector128.Count); + int vectorizedColumns = (int)(Numerics.Vector128Count(remainingColumns) * (nuint)Vector128.Count); if (vectorizedColumns > 0) { Vector128 topLeftVector = usesTopLeft ? Vector128.Create(topLeft) : default; diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PalettePredictor.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PalettePredictor.Operator.cs index f1cd3fd209..9317d8ee3e 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PalettePredictor.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PalettePredictor.Operator.cs @@ -212,33 +212,43 @@ internal static class Av1PalettePredictor if (Vector512.IsHardwareAccelerated) { - Vector256 palette256 = Vector256.Create(palette128, palette128); - Vector512 palette512 = Vector512.Create(palette256, palette256); - int oneVectorFromEnd = width - Vector512.Count; + nuint vectorCount = Numerics.Vector512Count(width - column); - for (; column <= oneVectorFromEnd; column += Vector512.Count) + if (vectorCount > 0) { - Vector512 indices = Vector512.LoadUnsafe(ref mapRow, (nuint)column); - TOperator.Predict(palette512, indices).StoreUnsafe(ref destinationRow, (nuint)column); + // Replicate the lookup table only when the row has a complete 64-lane batch. + Vector256 palette256 = Vector256.Create(palette128, palette128); + Vector512 palette512 = Vector512.Create(palette256, palette256); + + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) + { + Vector512 indices = Vector512.LoadUnsafe(ref mapRow, (nuint)column); + TOperator.Predict(palette512, indices).StoreUnsafe(ref destinationRow, (nuint)column); + } } } if (Vector256.IsHardwareAccelerated) { - Vector256 palette256 = Vector256.Create(palette128, palette128); - int oneVectorFromEnd = width - Vector256.Count; + nuint vectorCount = Numerics.Vector256Count(width - column); - for (; column <= oneVectorFromEnd; column += Vector256.Count) + if (vectorCount > 0) { - Vector256 indices = Vector256.LoadUnsafe(ref mapRow, (nuint)column); - TOperator.Predict(palette256, indices).StoreUnsafe(ref destinationRow, (nuint)column); + // The narrower table is likewise materialized only for a complete 32-lane remainder. + Vector256 palette256 = Vector256.Create(palette128, palette128); + + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) + { + Vector256 indices = Vector256.LoadUnsafe(ref mapRow, (nuint)column); + TOperator.Predict(palette256, indices).StoreUnsafe(ref destinationRow, (nuint)column); + } } } if (Vector128.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector128.Count; - for (; column <= oneVectorFromEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 indices = Vector128.LoadUnsafe(ref mapRow, (nuint)column); TOperator.Predict(palette128, indices).StoreUnsafe(ref destinationRow, (nuint)column); @@ -296,35 +306,45 @@ internal static class Av1PalettePredictor if (Vector512.IsHardwareAccelerated) { - Vector256 palette256 = Vector256.Create(palette128, palette128); - Vector512 palette512 = Vector512.Create(palette256, palette256); - int oneVectorFromEnd = width - Vector512.Count; + nuint vectorCount = Numerics.Vector512Count(width - column); - for (; column <= oneVectorFromEnd; column += Vector512.Count) + if (vectorCount > 0) { - (Vector256 lower, Vector256 upper) = Vector256.Widen(Vector256.LoadUnsafe(ref mapRow, (nuint)column)); - Vector512 indices = Vector512.Create(lower, upper); - TOperator.Predict(palette512, indices).StoreUnsafe(ref destinationRow, (nuint)column); + // High-bit-depth output has half as many lanes, so gate table replication with that lane count. + Vector256 palette256 = Vector256.Create(palette128, palette128); + Vector512 palette512 = Vector512.Create(palette256, palette256); + + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) + { + (Vector256 lower, Vector256 upper) = Vector256.Widen(Vector256.LoadUnsafe(ref mapRow, (nuint)column)); + Vector512 indices = Vector512.Create(lower, upper); + TOperator.Predict(palette512, indices).StoreUnsafe(ref destinationRow, (nuint)column); + } } } if (Vector256.IsHardwareAccelerated) { - Vector256 palette256 = Vector256.Create(palette128, palette128); - int oneVectorFromEnd = width - Vector256.Count; + nuint vectorCount = Numerics.Vector256Count(width - column); - for (; column <= oneVectorFromEnd; column += Vector256.Count) + if (vectorCount > 0) { - (Vector128 lower, Vector128 upper) = Vector128.Widen(Vector128.LoadUnsafe(ref mapRow, (nuint)column)); - Vector256 indices = Vector256.Create(lower, upper); - TOperator.Predict(palette256, indices).StoreUnsafe(ref destinationRow, (nuint)column); + // Avoid creating the 256-bit table when the remainder belongs entirely to narrower paths. + Vector256 palette256 = Vector256.Create(palette128, palette128); + + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) + { + (Vector128 lower, Vector128 upper) = Vector128.Widen(Vector128.LoadUnsafe(ref mapRow, (nuint)column)); + Vector256 indices = Vector256.Create(lower, upper); + TOperator.Predict(palette256, indices).StoreUnsafe(ref destinationRow, (nuint)column); + } } } if (Vector128.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector128.Count; - for (; column <= oneVectorFromEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { ulong packedIndices = Unsafe.ReadUnaligned(ref Unsafe.Add(ref mapRow, column)); Vector128 indices = Vector128.WidenLower(Vector128.CreateScalarUnsafe(packedIndices).AsByte()); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PredictionDecoder.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PredictionDecoder.cs index 5ed8ed227c..8c68bc172f 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PredictionDecoder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PredictionDecoder.cs @@ -1689,11 +1689,11 @@ internal sealed class Av1PredictionDecoder // Unsigned 16-bit lanes therefore preserve every normative strength without widening to 32-bit vectors. if (Vector128.IsHardwareAccelerated) { - int eightSamplesFromEnd = outputCount - Vector128.Count; + int vectorEnd = (int)(Numerics.Vector128Count(outputCount) * (nuint)Vector128.Count); switch (strength) { case 1: - for (; processed <= eightSamplesFromEnd; processed += Vector128.Count) + for (; processed < vectorEnd; processed += Vector128.Count) { Vector128 source0 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16(); Vector128 source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 2)).AsUInt16(); @@ -1703,7 +1703,7 @@ internal sealed class Av1PredictionDecoder break; case 2: - for (; processed <= eightSamplesFromEnd; processed += Vector128.Count) + for (; processed < vectorEnd; processed += Vector128.Count) { Vector128 source0 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16(); Vector128 source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 2)).AsUInt16(); @@ -1713,7 +1713,7 @@ internal sealed class Av1PredictionDecoder break; default: - for (; processed <= eightSamplesFromEnd; processed += Vector128.Count) + for (; processed < vectorEnd; processed += Vector128.Count) { Vector128 source0 = Vector128.LoadUnsafe(ref edge, (nuint)processed).AsUInt16(); Vector128 source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16(); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs index a5f25b665e..cbcc5fcb79 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs @@ -42,7 +42,8 @@ internal partial class Av1ChromaFromLumaContext if (Vector256.IsHardwareAccelerated) { - for (; column <= width - Vector256.Count; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { (Vector256 lower, Vector256 upper) = Vector256.Widen(Vector256.LoadUnsafe(ref inputRow, (nuint)column)); @@ -53,7 +54,8 @@ internal partial class Av1ChromaFromLumaContext if (Vector128.IsHardwareAccelerated) { - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { (Vector128 lower, Vector128 upper) = Vector128.Widen(Vector128.LoadUnsafe(ref inputRow, (nuint)column)); @@ -103,7 +105,8 @@ internal partial class Av1ChromaFromLumaContext if (Avx2.IsSupported) { - for (; column <= width - Vector256.Count; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 sum = Avx2.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref inputRow, (nuint)column), ones256); if (this.subY) @@ -117,7 +120,8 @@ internal partial class Av1ChromaFromLumaContext if (Vector128.IsHardwareAccelerated) { - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 sum = PairSum(Vector128.LoadUnsafe(ref inputRow, (nuint)column), ones128); if (this.subY) @@ -193,7 +197,8 @@ internal partial class Av1ChromaFromLumaContext if (Vector256.IsHardwareAccelerated) { - for (; column <= width - Vector256.Count; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { (Vector256.LoadUnsafe(ref inputRow, (nuint)column) << 3).StoreUnsafe(ref outputRow, (nuint)column); } @@ -201,7 +206,8 @@ internal partial class Av1ChromaFromLumaContext if (Vector128.IsHardwareAccelerated) { - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { (Vector128.LoadUnsafe(ref inputRow, (nuint)column) << 3).StoreUnsafe(ref outputRow, (nuint)column); } @@ -237,7 +243,8 @@ internal partial class Av1ChromaFromLumaContext if (Vector256.IsHardwareAccelerated) { - for (; column <= width - Vector256.Count; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 sum = Vector256_.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref inputRow, (nuint)column), ones256); if (this.subY) @@ -251,7 +258,8 @@ internal partial class Av1ChromaFromLumaContext if (Vector128.IsHardwareAccelerated) { - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 sum = Vector128_.MultiplyAddAdjacent(Vector128.LoadUnsafe(ref inputRow, (nuint)column), ones128); if (this.subY) @@ -326,7 +334,8 @@ internal partial class Av1ChromaFromLumaContext { int rowOffset = row * BufferLine; int column = 0; - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { (Vector128 lower, Vector128 upper) = Vector128.Widen(Vector128.LoadUnsafe(ref bufferBase, (nuint)(rowOffset + column))); @@ -377,7 +386,8 @@ internal partial class Av1ChromaFromLumaContext { int rowOffset = row * BufferLine; int column = 0; - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { (Vector128.LoadUnsafe(ref bufferBase, (nuint)(rowOffset + column)) - average).StoreUnsafe(ref bufferBase, (nuint)(rowOffset + column)); } diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaPredictor.Operator.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaPredictor.Operator.cs index 23dfca5027..5cfbec7361 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaPredictor.Operator.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaPredictor.Operator.cs @@ -216,8 +216,8 @@ internal static partial class Av1ChromaFromLumaPredictor if (Vector512.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector512.Count; - for (; column <= oneVectorFromEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 prediction = TOperator.Predict(Vector512.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue); Vector256 packed = Vector512.Narrow(prediction.AsUInt16(), Vector512.Zero).GetLower(); @@ -227,8 +227,8 @@ internal static partial class Av1ChromaFromLumaPredictor if (Vector256.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector256.Count; - for (; column <= oneVectorFromEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 prediction = TOperator.Predict(Vector256.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue); Vector128 packed = Vector256.Narrow(prediction.AsUInt16(), Vector256.Zero).GetLower(); @@ -238,8 +238,8 @@ internal static partial class Av1ChromaFromLumaPredictor if (Vector128.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector128.Count; - for (; column <= oneVectorFromEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 prediction = TOperator.Predict(Vector128.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue); Vector64 packed = Vector128.Narrow(prediction.AsUInt16(), Vector128.Zero).GetLower(); @@ -282,8 +282,8 @@ internal static partial class Av1ChromaFromLumaPredictor if (Vector512.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector512.Count; - for (; column <= oneVectorFromEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { TOperator.Predict(Vector512.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column); } @@ -291,8 +291,8 @@ internal static partial class Av1ChromaFromLumaPredictor if (Vector256.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector256.Count; - for (; column <= oneVectorFromEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { TOperator.Predict(Vector256.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column); } @@ -300,8 +300,8 @@ internal static partial class Av1ChromaFromLumaPredictor if (Vector128.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector128.Count; - for (; column <= oneVectorFromEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { TOperator.Predict(Vector128.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column); } diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundAveragePredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundAveragePredictor.cs index 9d7390b0b2..e0fd1f25c3 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundAveragePredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundAveragePredictor.cs @@ -60,8 +60,8 @@ internal static partial class Av1CompoundAveragePredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -71,8 +71,8 @@ internal static partial class Av1CompoundAveragePredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -82,8 +82,8 @@ internal static partial class Av1CompoundAveragePredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); @@ -145,8 +145,8 @@ internal static partial class Av1CompoundAveragePredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -156,8 +156,8 @@ internal static partial class Av1CompoundAveragePredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -167,8 +167,8 @@ internal static partial class Av1CompoundAveragePredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundDistanceWeightedPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundDistanceWeightedPredictor.cs index 7ea35e7e90..f8d0c1a461 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundDistanceWeightedPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundDistanceWeightedPredictor.cs @@ -62,8 +62,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -73,8 +73,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -84,8 +84,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); @@ -147,8 +147,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -158,8 +158,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -169,8 +169,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundInterPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundInterPredictor.cs index 2ba6fcf201..27baf014df 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundInterPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundInterPredictor.cs @@ -418,8 +418,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 samples = Vector512.LoadUnsafe(ref sourceRow, (nuint)column); TOperator.Copy(samples, roundBits, roundOffset, out Vector512 lower, out Vector512 upper); @@ -430,8 +430,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 samples = Vector256.LoadUnsafe(ref sourceRow, (nuint)column); TOperator.Copy(samples, roundBits, roundOffset, out Vector256 lower, out Vector256 upper); @@ -442,8 +442,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 samples = Vector128.LoadUnsafe(ref sourceRow, (nuint)column); TOperator.Copy(samples, roundBits, roundOffset, out Vector128 lower, out Vector128 upper); @@ -487,8 +487,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 samples = Vector512.LoadUnsafe(ref sourceRow, (nuint)column); TOperator.CopyHighBitDepth(samples, roundBits, roundOffset) @@ -498,8 +498,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 samples = Vector256.LoadUnsafe(ref sourceRow, (nuint)column); TOperator.CopyHighBitDepth(samples, roundBits, roundOffset) @@ -509,8 +509,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 samples = Vector128.LoadUnsafe(ref sourceRow, (nuint)column); TOperator.CopyHighBitDepth(samples, roundBits, roundOffset) @@ -562,8 +562,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Convolve( ref sourceRow, @@ -587,8 +587,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Convolve( ref sourceRow, @@ -612,8 +612,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Convolve( ref sourceRow, @@ -683,8 +683,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Convolve( ref sourceRow, @@ -703,8 +703,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Convolve( ref sourceRow, @@ -723,8 +723,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Convolve( ref sourceRow, @@ -797,8 +797,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Convolve( ref sourceRow, @@ -820,8 +820,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Convolve( ref sourceRow, @@ -843,8 +843,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Convolve( ref sourceRow, @@ -884,8 +884,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Convolve( ref scratchRow, @@ -903,8 +903,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Convolve( ref scratchRow, @@ -922,8 +922,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Convolve( ref scratchRow, @@ -1000,8 +1000,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Convolve( ref sourceRow, @@ -1020,8 +1020,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Convolve( ref sourceRow, @@ -1040,8 +1040,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Convolve( ref sourceRow, @@ -1079,8 +1079,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Convolve( ref scratchRow, @@ -1099,8 +1099,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Convolve( ref scratchRow, @@ -1119,8 +1119,8 @@ internal static partial class Av1CompoundInterPredictor if (useSimd && Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Convolve( ref scratchRow, diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateAveragePredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateAveragePredictor.cs index b1e0f95b77..99b31068c6 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateAveragePredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateAveragePredictor.cs @@ -68,8 +68,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512.Count)); @@ -82,8 +82,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256.Count)); @@ -96,8 +96,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 first1 = Vector128.LoadUnsafe( @@ -177,8 +177,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -189,8 +189,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -201,8 +201,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDifferenceWeightedMaskBuilder.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDifferenceWeightedMaskBuilder.cs index eaf6d72f1d..181b0e6a75 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDifferenceWeightedMaskBuilder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDifferenceWeightedMaskBuilder.cs @@ -74,8 +74,8 @@ internal static partial class Av1CompoundIntermediateDifferenceWeightedMaskBuild if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512.Count)); @@ -88,8 +88,8 @@ internal static partial class Av1CompoundIntermediateDifferenceWeightedMaskBuild if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256.Count)); @@ -102,8 +102,8 @@ internal static partial class Av1CompoundIntermediateDifferenceWeightedMaskBuild if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 first1 = Vector128.LoadUnsafe( diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDistanceWeightedPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDistanceWeightedPredictor.cs index 15779f3dcf..719306216e 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDistanceWeightedPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDistanceWeightedPredictor.cs @@ -74,8 +74,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512.Count)); @@ -95,8 +95,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256.Count)); @@ -116,8 +116,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 first1 = Vector128.LoadUnsafe( @@ -215,8 +215,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -233,8 +233,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -251,8 +251,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateMaskBlendPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateMaskBlendPredictor.cs index 967e11a8a9..e2331c863a 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateMaskBlendPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateMaskBlendPredictor.cs @@ -82,8 +82,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor { ref byte maskReference = ref MemoryMarshal.GetReference(mask); int maskRowOffset = row * maskStride; - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512.Count)); @@ -99,8 +99,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor { ref byte maskReference = ref MemoryMarshal.GetReference(mask); int maskRowOffset = row * maskStride; - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256.Count)); @@ -116,8 +116,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor { ref byte maskReference = ref MemoryMarshal.GetReference(mask); int maskRowOffset = row * maskStride; - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 first1 = Vector128.LoadUnsafe( @@ -221,8 +221,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor { ref byte maskReference = ref MemoryMarshal.GetReference(mask); int maskRowOffset = row * maskStride; - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 first1 = Vector512.LoadUnsafe( @@ -261,8 +261,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor { ref byte maskReference = ref MemoryMarshal.GetReference(mask); int maskRowOffset = row * maskStride; - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 first1 = Vector256.LoadUnsafe( @@ -301,8 +301,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor { ref byte maskReference = ref MemoryMarshal.GetReference(mask); int maskRowOffset = row * maskStride; - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 first1 = Vector128.LoadUnsafe( diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundMaskBlendPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundMaskBlendPredictor.cs index c798d74c37..59400482ab 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundMaskBlendPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundMaskBlendPredictor.cs @@ -64,8 +64,8 @@ internal static partial class Av1CompoundMaskBlendPredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -76,8 +76,8 @@ internal static partial class Av1CompoundMaskBlendPredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -88,8 +88,8 @@ internal static partial class Av1CompoundMaskBlendPredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); @@ -154,8 +154,8 @@ internal static partial class Av1CompoundMaskBlendPredictor if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -166,8 +166,8 @@ internal static partial class Av1CompoundMaskBlendPredictor if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -178,8 +178,8 @@ internal static partial class Av1CompoundMaskBlendPredictor if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1DifferenceWeightedMaskBuilder.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1DifferenceWeightedMaskBuilder.cs index 1ebc07cbf3..f4e0c9ff7b 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1DifferenceWeightedMaskBuilder.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1DifferenceWeightedMaskBuilder.cs @@ -68,8 +68,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); @@ -79,8 +79,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); @@ -90,8 +90,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); @@ -165,8 +165,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder // temporary buffers before the following vector blend consumes the complete plane block. if (Vector512.IsHardwareAccelerated) { - int vectorEnd = width - Vector512.Count; - for (; column <= vectorEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512 first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512.Count)); @@ -179,8 +179,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder if (Vector256.IsHardwareAccelerated) { - int vectorEnd = width - Vector256.Count; - for (; column <= vectorEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256 first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256.Count)); @@ -193,8 +193,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder if (Vector128.IsHardwareAccelerated) { - int vectorEnd = width - Vector128.Count; - for (; column <= vectorEnd; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128 first1 = Vector128.LoadUnsafe(ref firstReference, (nuint)(column + Vector128.Count)); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1ScaledInterPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1ScaledInterPredictor.cs index abc09f67ed..a0414c7853 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1ScaledInterPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1ScaledInterPredictor.cs @@ -411,8 +411,8 @@ internal static partial class Av1ScaledInterPredictor if (Vector512.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector512.Count; - for (; column <= oneVectorFromEnd; column += Vector512.Count) + nuint vectorCount = Numerics.Vector512Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector512.Count) { Vector512 result = FilterScaledHorizontalVector512( ref sourceRow, @@ -431,8 +431,8 @@ internal static partial class Av1ScaledInterPredictor if (Vector256.IsHardwareAccelerated) { - int oneVectorFromEnd = width - Vector256.Count; - for (; column <= oneVectorFromEnd; column += Vector256.Count) + nuint vectorCount = Numerics.Vector256Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector256.Count) { Vector256 result = FilterScaledHorizontalVector256( ref sourceRow, @@ -451,7 +451,8 @@ internal static partial class Av1ScaledInterPredictor if (Vector128.IsHardwareAccelerated) { - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - column); + for (; vectorCount > 0; vectorCount--, column += Vector128.Count) { Vector128 result = FilterScaledHorizontalVector128( ref sourceRow, @@ -505,70 +506,89 @@ internal static partial class Av1ScaledInterPredictor if (Vector512.IsHardwareAccelerated) { - Vector512 initial = Vector512.Create(verticalBias); - Vector512 offset = Vector512.Create(roundOffset); - int oneVectorFromEnd = width - (Vector512.Count * 2); - for (; column <= oneVectorFromEnd; column += Vector512.Count * 2) + nuint vectorCount = Numerics.Vector512Count(width - column) / 2; + + if (vectorCount > 0) { - NativeOperator.Convolve( - ref scratchRow, - scratchStride, - (nuint)column, - ref coefficientBase, - FilterCoefficientCount, - initial, - out Vector512 result0, - out Vector512 result1); - - result0 = RoundPowerOfTwo(result0, round1) - offset; - result1 = RoundPowerOfTwo(result1, round1) - offset; - TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); + // The scaled vertical kernel emits two register widths per iteration, so constants are needed only + // when the remaining row contains at least two complete vectors. + Vector512 initial = Vector512.Create(verticalBias); + Vector512 offset = Vector512.Create(roundOffset); + + for (; vectorCount > 0; vectorCount--, column += Vector512.Count * 2) + { + NativeOperator.Convolve( + ref scratchRow, + scratchStride, + (nuint)column, + ref coefficientBase, + FilterCoefficientCount, + initial, + out Vector512 result0, + out Vector512 result1); + + result0 = RoundPowerOfTwo(result0, round1) - offset; + result1 = RoundPowerOfTwo(result1, round1) - offset; + TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); + } } } if (Vector256.IsHardwareAccelerated) { - Vector256 initial = Vector256.Create(verticalBias); - Vector256 offset = Vector256.Create(roundOffset); - int oneVectorFromEnd = width - (Vector256.Count * 2); - for (; column <= oneVectorFromEnd; column += Vector256.Count * 2) + nuint vectorCount = Numerics.Vector256Count(width - column) / 2; + + if (vectorCount > 0) { - NativeOperator.Convolve( - ref scratchRow, - scratchStride, - (nuint)column, - ref coefficientBase, - FilterCoefficientCount, - initial, - out Vector256 result0, - out Vector256 result1); - - result0 = RoundPowerOfTwo(result0, round1) - offset; - result1 = RoundPowerOfTwo(result1, round1) - offset; - TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); + // A single-vector remainder belongs to the next narrower tier and does not materialize YMM constants. + Vector256 initial = Vector256.Create(verticalBias); + Vector256 offset = Vector256.Create(roundOffset); + + for (; vectorCount > 0; vectorCount--, column += Vector256.Count * 2) + { + NativeOperator.Convolve( + ref scratchRow, + scratchStride, + (nuint)column, + ref coefficientBase, + FilterCoefficientCount, + initial, + out Vector256 result0, + out Vector256 result1); + + result0 = RoundPowerOfTwo(result0, round1) - offset; + result1 = RoundPowerOfTwo(result1, round1) - offset; + TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); + } } } if (Vector128.IsHardwareAccelerated) { - Vector128 initial = Vector128.Create(verticalBias); - Vector128 offset = Vector128.Create(roundOffset); - int oneVectorFromEnd = width - (Vector128.Count * 2); - for (; column <= oneVectorFromEnd; column += Vector128.Count * 2) + nuint vectorCount = Numerics.Vector128Count(width - column) / 2; + + if (vectorCount > 0) { - NativeOperator.Convolve( - ref scratchRow, - scratchStride, - (nuint)column, - ref coefficientBase, - FilterCoefficientCount, - initial, - out Vector128 result0, - out Vector128 result1); - - result0 = RoundPowerOfTwo(result0, round1) - offset; - result1 = RoundPowerOfTwo(result1, round1) - offset; - TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); + // XMM constants are likewise skipped when fewer than eight output samples remain. + Vector128 initial = Vector128.Create(verticalBias); + Vector128 offset = Vector128.Create(roundOffset); + + for (; vectorCount > 0; vectorCount--, column += Vector128.Count * 2) + { + NativeOperator.Convolve( + ref scratchRow, + scratchStride, + (nuint)column, + ref coefficientBase, + FilterCoefficientCount, + initial, + out Vector128 result0, + out Vector128 result1); + + result0 = RoundPowerOfTwo(result0, round1) - offset; + result1 = RoundPowerOfTwo(result1, round1) - offset; + TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); + } } } diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.Dispatch.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.Dispatch.cs index edc5ac5fef..26c3346fb8 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.Dispatch.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.Dispatch.cs @@ -371,14 +371,14 @@ internal static partial class Av1TranslationalInterPredictor if (Vector512.IsHardwareAccelerated && Vector.Count == Vector512.Count && width >= Vector512.Count) { - int vectorEnd = width - Vector512.Count; + int vectorEnd = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); for (int row = 0; row < height; row++) { ref byte sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride); ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int column = 0; - for (; column <= vectorEnd; column += Vector512.Count) + for (; column < vectorEnd; column += Vector512.Count) { Vector512.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); } @@ -394,14 +394,14 @@ internal static partial class Av1TranslationalInterPredictor if (Vector256.IsHardwareAccelerated && width >= Vector256.Count) { - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); for (int row = 0; row < height; row++) { ref byte sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride); ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int column = 0; - for (; column <= vectorEnd; column += Vector256.Count) + for (; column < vectorEnd; column += Vector256.Count) { Vector256.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); } @@ -428,9 +428,9 @@ internal static partial class Av1TranslationalInterPredictor continue; } - int vectorEnd = width - Vector128.Count; + int vectorEnd = (int)(Numerics.Vector128Count(width) * (nuint)Vector128.Count); int column = 0; - for (; column <= vectorEnd; column += Vector128.Count) + for (; column < vectorEnd; column += Vector128.Count) { Vector128.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); } @@ -464,14 +464,14 @@ internal static partial class Av1TranslationalInterPredictor if (Vector512.IsHardwareAccelerated && Vector.Count == Vector512.Count && width >= Vector512.Count) { - int vectorEnd = width - Vector512.Count; + int vectorEnd = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); for (int row = 0; row < height; row++) { ref ushort sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride); ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int column = 0; - for (; column <= vectorEnd; column += Vector512.Count) + for (; column < vectorEnd; column += Vector512.Count) { Vector512.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); } @@ -487,14 +487,14 @@ internal static partial class Av1TranslationalInterPredictor if (Vector256.IsHardwareAccelerated && width >= Vector256.Count) { - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); for (int row = 0; row < height; row++) { ref ushort sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride); ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int column = 0; - for (; column <= vectorEnd; column += Vector256.Count) + for (; column < vectorEnd; column += Vector256.Count) { Vector256.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); } @@ -523,9 +523,9 @@ internal static partial class Av1TranslationalInterPredictor continue; } - int vectorEnd = width - Vector128.Count; + int vectorEnd = (int)(Numerics.Vector128Count(width) * (nuint)Vector128.Count); int column = 0; - for (; column <= vectorEnd; column += Vector128.Count) + for (; column < vectorEnd; column += Vector128.Count) { Vector128.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); } diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.OneDimension.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.OneDimension.cs index bfa0d8d224..e25e8bbbd8 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.OneDimension.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.OneDimension.cs @@ -62,8 +62,8 @@ internal static partial class Av1TranslationalInterPredictor continue; } - int vectorEnd = width - Vector128.Count; - for (; processedColumns <= vectorEnd; processedColumns += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - processedColumns); + for (; vectorCount > 0; vectorCount--, processedColumns += Vector128.Count) { Convolve( ref sourceRow, @@ -107,7 +107,7 @@ internal static partial class Av1TranslationalInterPredictor ref byte sourceBase = ref Unsafe.Add(ref MemoryMarshal.GetReference(source), sourceOrigin); ref byte destinationBase = ref MemoryMarshal.GetReference(destination); ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients); - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); for (int row = 0; row < height; row++) { @@ -115,7 +115,7 @@ internal static partial class Av1TranslationalInterPredictor ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector256.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector256.Count) { Convolve( ref sourceRow, @@ -159,7 +159,7 @@ internal static partial class Av1TranslationalInterPredictor ref byte sourceBase = ref Unsafe.Add(ref MemoryMarshal.GetReference(source), sourceOrigin); ref byte destinationBase = ref MemoryMarshal.GetReference(destination); ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients); - int vectorEnd = width - Vector512.Count; + int vectorEnd = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); for (int row = 0; row < height; row++) { @@ -167,7 +167,7 @@ internal static partial class Av1TranslationalInterPredictor ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector512.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector512.Count) { Convolve( ref sourceRow, @@ -232,8 +232,8 @@ internal static partial class Av1TranslationalInterPredictor continue; } - int vectorEnd = width - Vector128.Count; - for (; processedColumns <= vectorEnd; processedColumns += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - processedColumns); + for (; vectorCount > 0; vectorCount--, processedColumns += Vector128.Count) { Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector128 result0, out Vector128 result1); Round(ref result0, ref result1, firstRound, secondRound); @@ -268,7 +268,7 @@ internal static partial class Av1TranslationalInterPredictor ref ushort destinationBase = ref MemoryMarshal.GetReference(destination); ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients); int maximum = (1 << bitDepth) - 1; - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); for (int row = 0; row < height; row++) { @@ -277,7 +277,7 @@ internal static partial class Av1TranslationalInterPredictor ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector256.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector256.Count) { Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector256 result0, out Vector256 result1); Round(ref result0, ref result1, firstRound, secondRound); @@ -312,7 +312,7 @@ internal static partial class Av1TranslationalInterPredictor ref ushort destinationBase = ref MemoryMarshal.GetReference(destination); ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients); int maximum = (1 << bitDepth) - 1; - int vectorEnd = width - Vector512.Count; + int vectorEnd = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); for (int row = 0; row < height; row++) { @@ -321,7 +321,7 @@ internal static partial class Av1TranslationalInterPredictor ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector512.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector512.Count) { Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector512 result0, out Vector512 result1); Round(ref result0, ref result1, firstRound, secondRound); diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.Byte.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.Byte.cs index 9ff9105a30..f917c91791 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.Byte.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.Byte.cs @@ -68,8 +68,8 @@ internal static partial class Av1TranslationalInterPredictor continue; } - int vectorEnd = width - Vector128.Count; - for (; processedColumns <= vectorEnd; processedColumns += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - processedColumns); + for (; vectorCount > 0; vectorCount--, processedColumns += Vector128.Count) { Convolve( ref sourceRow, @@ -135,8 +135,8 @@ internal static partial class Av1TranslationalInterPredictor continue; } - int vectorEnd = width - Vector128.Count; - for (; processedColumns <= vectorEnd; processedColumns += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - processedColumns); + for (; vectorCount > 0; vectorCount--, processedColumns += Vector128.Count) { Convolve( ref scratchRow, @@ -199,7 +199,7 @@ internal static partial class Av1TranslationalInterPredictor int scratchStride = Math.Max(width, MinimumScratchStride); int intermediateHeight = height + verticalTapCount - 1; Vector256 horizontalInitial = initial + Vector256.Create(1 << (bitDepth + FilterBits - 1)); - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); for (int row = 0; row < intermediateHeight; row++) { @@ -207,7 +207,7 @@ internal static partial class Av1TranslationalInterPredictor ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector256.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector256.Count) { Convolve( ref sourceRow, @@ -243,7 +243,7 @@ internal static partial class Av1TranslationalInterPredictor ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector256.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector256.Count) { Convolve( ref scratchRow, @@ -306,7 +306,7 @@ internal static partial class Av1TranslationalInterPredictor int scratchStride = Math.Max(width, MinimumScratchStride); int intermediateHeight = height + verticalTapCount - 1; Vector512 horizontalInitial = initial + Vector512.Create(1 << (bitDepth + FilterBits - 1)); - int vectorEnd = width - Vector512.Count; + int vectorEnd = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); for (int row = 0; row < intermediateHeight; row++) { @@ -314,7 +314,7 @@ internal static partial class Av1TranslationalInterPredictor ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector512.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector512.Count) { Convolve( ref sourceRow, @@ -350,7 +350,7 @@ internal static partial class Av1TranslationalInterPredictor ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector512.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector512.Count) { Convolve( ref scratchRow, diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.UInt16.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.UInt16.cs index e136694a1b..c7cf8e0d3e 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.UInt16.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.UInt16.cs @@ -74,8 +74,8 @@ internal static partial class Av1TranslationalInterPredictor continue; } - int vectorEnd = width - Vector128.Count; - for (; processedColumns <= vectorEnd; processedColumns += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - processedColumns); + for (; vectorCount > 0; vectorCount--, processedColumns += Vector128.Count) { Convolve( ref sourceRow, @@ -138,8 +138,8 @@ internal static partial class Av1TranslationalInterPredictor continue; } - int vectorEnd = width - Vector128.Count; - for (; processedColumns <= vectorEnd; processedColumns += Vector128.Count) + nuint vectorCount = Numerics.Vector128Count(width - processedColumns); + for (; vectorCount > 0; vectorCount--, processedColumns += Vector128.Count) { Convolve( ref scratchRow, @@ -199,7 +199,7 @@ internal static partial class Av1TranslationalInterPredictor int scratchStride = Math.Max(width, MinimumScratchStride); int intermediateHeight = height + verticalTapCount - 1; Vector256 horizontalInitial = initial + Vector256.Create(1 << (bitDepth + FilterBits - 1)); - int vectorEnd = width - Vector256.Count; + int vectorEnd = (int)(Numerics.Vector256Count(width) * (nuint)Vector256.Count); for (int row = 0; row < intermediateHeight; row++) { @@ -211,7 +211,7 @@ internal static partial class Av1TranslationalInterPredictor ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector256.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector256.Count) { Convolve( ref sourceRow, @@ -253,7 +253,7 @@ internal static partial class Av1TranslationalInterPredictor ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector256.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector256.Count) { Convolve( ref scratchRow, @@ -313,7 +313,7 @@ internal static partial class Av1TranslationalInterPredictor int scratchStride = Math.Max(width, MinimumScratchStride); int intermediateHeight = height + verticalTapCount - 1; Vector512 horizontalInitial = initial + Vector512.Create(1 << (bitDepth + FilterBits - 1)); - int vectorEnd = width - Vector512.Count; + int vectorEnd = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); for (int row = 0; row < intermediateHeight; row++) { @@ -325,7 +325,7 @@ internal static partial class Av1TranslationalInterPredictor ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector512.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector512.Count) { Convolve( ref sourceRow, @@ -367,7 +367,7 @@ internal static partial class Av1TranslationalInterPredictor ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); int processedColumns = 0; - for (; processedColumns <= vectorEnd; processedColumns += Vector512.Count) + for (; processedColumns < vectorEnd; processedColumns += Vector512.Count) { Convolve( ref scratchRow, diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyBilinearPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyBilinearPredictor.cs index b4f4321c88..3d4555f211 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyBilinearPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyBilinearPredictor.cs @@ -110,7 +110,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor // cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail. if (Vector512.IsHardwareAccelerated) { - int vectorizedColumns = width - (width % Vector512.Count); + int vectorizedColumns = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); if (vectorizedColumns > 0) { for (int row = 0; row < height; row++) @@ -136,7 +136,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor if (Vector256.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector256.Count); + int vectorizedColumns = (int)(Numerics.Vector256Count(remainingColumns) * (nuint)Vector256.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -161,7 +161,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor if (Vector128.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector128.Count); + int vectorizedColumns = (int)(Numerics.Vector128Count(remainingColumns) * (nuint)Vector128.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -244,7 +244,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor // continuation as the byte path. if (Vector512.IsHardwareAccelerated) { - int vectorizedColumns = width - (width % Vector512.Count); + int vectorizedColumns = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); if (vectorizedColumns > 0) { for (int row = 0; row < height; row++) @@ -270,7 +270,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor if (Vector256.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector256.Count); + int vectorizedColumns = (int)(Numerics.Vector256Count(remainingColumns) * (nuint)Vector256.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -295,7 +295,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor if (Vector128.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector128.Count); + int vectorizedColumns = (int)(Numerics.Vector128Count(remainingColumns) * (nuint)Vector128.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyHorizontalPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyHorizontalPredictor.cs index 437f110747..597f3a8872 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyHorizontalPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyHorizontalPredictor.cs @@ -108,7 +108,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor // cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail. if (Vector512.IsHardwareAccelerated) { - int vectorizedColumns = width - (width % Vector512.Count); + int vectorizedColumns = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); if (vectorizedColumns > 0) { for (int row = 0; row < height; row++) @@ -132,7 +132,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor if (Vector256.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector256.Count); + int vectorizedColumns = (int)(Numerics.Vector256Count(remainingColumns) * (nuint)Vector256.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -155,7 +155,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor if (Vector128.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector128.Count); + int vectorizedColumns = (int)(Numerics.Vector128Count(remainingColumns) * (nuint)Vector128.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -232,7 +232,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor // continuation as the byte path. if (Vector512.IsHardwareAccelerated) { - int vectorizedColumns = width - (width % Vector512.Count); + int vectorizedColumns = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); if (vectorizedColumns > 0) { for (int row = 0; row < height; row++) @@ -256,7 +256,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor if (Vector256.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector256.Count); + int vectorizedColumns = (int)(Numerics.Vector256Count(remainingColumns) * (nuint)Vector256.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -279,7 +279,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor if (Vector128.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector128.Count); + int vectorizedColumns = (int)(Numerics.Vector128Count(remainingColumns) * (nuint)Vector128.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) diff --git a/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyVerticalPredictor.cs b/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyVerticalPredictor.cs index 565020766d..e3afaec567 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyVerticalPredictor.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyVerticalPredictor.cs @@ -108,7 +108,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor // cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail. if (Vector512.IsHardwareAccelerated) { - int vectorizedColumns = width - (width % Vector512.Count); + int vectorizedColumns = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); if (vectorizedColumns > 0) { for (int row = 0; row < height; row++) @@ -132,7 +132,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor if (Vector256.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector256.Count); + int vectorizedColumns = (int)(Numerics.Vector256Count(remainingColumns) * (nuint)Vector256.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -155,7 +155,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor if (Vector128.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector128.Count); + int vectorizedColumns = (int)(Numerics.Vector128Count(remainingColumns) * (nuint)Vector128.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -232,7 +232,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor // continuation as the byte path. if (Vector512.IsHardwareAccelerated) { - int vectorizedColumns = width - (width % Vector512.Count); + int vectorizedColumns = (int)(Numerics.Vector512Count(width) * (nuint)Vector512.Count); if (vectorizedColumns > 0) { for (int row = 0; row < height; row++) @@ -256,7 +256,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor if (Vector256.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector256.Count); + int vectorizedColumns = (int)(Numerics.Vector256Count(remainingColumns) * (nuint)Vector256.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) @@ -279,7 +279,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor if (Vector128.IsHardwareAccelerated) { int remainingColumns = width - processedColumns; - int vectorizedColumns = remainingColumns - (remainingColumns % Vector128.Count); + int vectorizedColumns = (int)(Numerics.Vector128Count(remainingColumns) * (nuint)Vector128.Count); int endColumn = processedColumns + vectorizedColumns; for (int row = 0; row < height; row++) diff --git a/src/ImageSharp/Formats/Heif/Av1/Transform/Av1ForwardTransformer.cs b/src/ImageSharp/Formats/Heif/Av1/Transform/Av1ForwardTransformer.cs index d5f01327ae..6a97178e2c 100644 --- a/src/ImageSharp/Formats/Heif/Av1/Transform/Av1ForwardTransformer.cs +++ b/src/ImageSharp/Formats/Heif/Av1/Transform/Av1ForwardTransformer.cs @@ -1046,7 +1046,8 @@ internal static partial class Av1ForwardTransformer if (Avx512BW.IsSupported) { - for (; column <= width - Vector512.Count; column += Vector512.Count) + nuint vector512Count = Numerics.Vector512Count(width - column); + for (; vector512Count > 0; vector512Count--, column += Vector512.Count) { (Vector512 lower, Vector512 upper) = Vector512.Widen(Vector512.LoadUnsafe(ref sourceRow, (nuint)column)); @@ -1057,7 +1058,8 @@ internal static partial class Av1ForwardTransformer if (Avx2.IsSupported) { - for (; column <= width - Vector256.Count; column += Vector256.Count) + nuint vector256Count = Numerics.Vector256Count(width - column); + for (; vector256Count > 0; vector256Count--, column += Vector256.Count) { (Vector256 lower, Vector256 upper) = Vector256.Widen(Vector256.LoadUnsafe(ref sourceRow, (nuint)column)); @@ -1066,7 +1068,8 @@ internal static partial class Av1ForwardTransformer } } - for (; column <= width - Vector128.Count; column += Vector128.Count) + nuint vector128Count = Numerics.Vector128Count(width - column); + for (; vector128Count > 0; vector128Count--, column += Vector128.Count) { (Vector128 lower, Vector128 upper) = Vector128.Widen(Vector128.LoadUnsafe(ref sourceRow, (nuint)column)); diff --git a/src/ImageSharp/Formats/Heif/HeifEncoder.cs b/src/ImageSharp/Formats/Heif/HeifEncoder.cs index c1b28173ee..8afae17803 100644 --- a/src/ImageSharp/Formats/Heif/HeifEncoder.cs +++ b/src/ImageSharp/Formats/Heif/HeifEncoder.cs @@ -31,8 +31,8 @@ public sealed class HeifEncoder : AnimatedImageEncoder /// /// Gets the lossy compression quality, or to use the compression method's default quality. - /// Valid values range from 0 for the lowest quality to 100 for the highest quality. Legacy JPEG image items - /// support values from 1 through 100. A value of 100 does not enable encoding. + /// Valid values range from 0 for the lowest quality to 100 for the highest quality. A value of 100 does not + /// enable encoding. /// /// The quality is outside the range 0 to 100. public int? Quality diff --git a/src/ImageSharp/Formats/Heif/HeifEncoderCore.cs b/src/ImageSharp/Formats/Heif/HeifEncoderCore.cs index 8221fcca7c..747a18f1c7 100644 --- a/src/ImageSharp/Formats/Heif/HeifEncoderCore.cs +++ b/src/ImageSharp/Formats/Heif/HeifEncoderCore.cs @@ -64,9 +64,6 @@ internal sealed class HeifEncoderCore this.WriteMetadataBox(items, links, stream); this.WriteMediaDataBox(compressedPixels, stream); stream.Flush(); - - HeifMetadata meta = image.Metadata.GetHeifMetadata(); - meta.CompressionMethod = this.encoder.CompressionMethod; } /// @@ -448,13 +445,6 @@ internal sealed class HeifEncoderCore throw new NotSupportedException("Legacy JPEG image items support only 8-bit component encoding."); } - if (this.encoder.Quality == 0) - { - // Zero is meaningful to the AV1 quality scale, but ImageSharp's JPEG encoder deliberately - // exposes the JPEG quality scale as 1 through 100. Reject the codec-specific mismatch at this boundary. - throw new NotSupportedException("Legacy JPEG image items support quality values in the range [1..100]."); - } - JpegColorType colorType = this.encoder.ChromaSubsampling switch { null or HeifChromaSubsampling.Yuv420 => JpegColorType.YCbCrRatio420, @@ -467,7 +457,9 @@ internal sealed class HeifEncoderCore ChunkedMemoryStream stream = new(this.configuration.MemoryAllocator); JpegEncoder encoder = new() { - Quality = this.encoder.Quality, + // The HEIF quality scale includes zero while the JPEG payload encoder starts at one. + // Map the lowest HEIF setting to the lowest representable JPEG setting. + Quality = this.encoder.Quality == 0 ? 1 : this.encoder.Quality, ColorType = colorType }; diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1BitStreamTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1BitStreamTests.cs index b1006faa66..be05e0b085 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1BitStreamTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1BitStreamTests.cs @@ -175,6 +175,7 @@ public class Av1BitStreamTests [InlineData(4, 0, 1, 2, 3)] [InlineData(5, 0, 1, 2, 3)] + [InlineData(5, 1, 2, 3, 4)] [InlineData(8, 0, 1, 2, 3)] [InlineData(8, 4, 5, 6, 7)] [InlineData(16, 15, 0, 5, 8)] diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs new file mode 100644 index 0000000000..fc8c51a5f5 --- /dev/null +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs @@ -0,0 +1,98 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using SixLabors.ImageSharp.Formats.Heif.Av1; +using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; +using SixLabors.ImageSharp.Memory; + +namespace SixLabors.ImageSharp.Tests.Formats.Heif.Av1; + +public class Av1EncoderFrameTests +{ + [Fact] + public void ExtendBordersReplicatesEveryPhysicalPlaneEdge() + { + const int visibleWidth = 5; + const int visibleHeight = 3; + const int codedWidth = 8; + const int codedHeight = 8; + const int lumaBorder = Av1EncoderFrame.LumaBorder; + const int chromaBorder = lumaBorder / 2; + MemoryAllocator allocator = Configuration.Default.MemoryAllocator; + + Size lumaBufferSize = Av1EncoderFrame.GetPlaneBufferSize(visibleWidth, visibleHeight, 0, 0); + Size chromaBufferSize = Av1EncoderFrame.GetPlaneBufferSize(visibleWidth, visibleHeight, 1, 1); + using Buffer2D luma = allocator.Allocate2D(lumaBufferSize.Width, lumaBufferSize.Height); + using Buffer2D chromaBlue = allocator.Allocate2D(chromaBufferSize.Width, chromaBufferSize.Height); + using Buffer2D chromaRed = allocator.Allocate2D(chromaBufferSize.Width, chromaBufferSize.Height); + Buffer2DRegion lumaRegion = luma.GetRegion(lumaBorder, lumaBorder, codedWidth, codedHeight); + Buffer2DRegion chromaBlueRegion = chromaBlue.GetRegion(chromaBorder, chromaBorder, codedWidth / 2, codedHeight / 2); + Buffer2DRegion chromaRedRegion = chromaRed.GetRegion(chromaBorder, chromaBorder, codedWidth / 2, codedHeight / 2); + + FillVisible(luma, lumaBorder, lumaBorder, visibleWidth, visibleHeight, 10); + FillVisible(chromaBlue, chromaBorder, chromaBorder, (visibleWidth + 1) / 2, (visibleHeight + 1) / 2, 80); + FillVisible(chromaRed, chromaBorder, chromaBorder, (visibleWidth + 1) / 2, (visibleHeight + 1) / 2, 120); + + Av1EncoderFrame frame = new( + lumaRegion, + chromaBlueRegion, + chromaRedRegion, + visibleWidth, + visibleHeight, + 8, + Av1ColorFormat.Yuv420, + 1, + 1); + + frame.ExtendBorders(); + + AssertReplicatedPlane(luma, lumaBorder, lumaBorder, visibleWidth, visibleHeight, 10); + AssertReplicatedPlane(chromaBlue, chromaBorder, chromaBorder, (visibleWidth + 1) / 2, (visibleHeight + 1) / 2, 80); + AssertReplicatedPlane(chromaRed, chromaBorder, chromaBorder, (visibleWidth + 1) / 2, (visibleHeight + 1) / 2, 120); + } + + [Theory] + [InlineData(5, 3, 0, 0, 160, 136)] + [InlineData(5, 3, 1, 0, 80, 136)] + [InlineData(5, 3, 1, 1, 80, 68)] + [InlineData(1921, 1081, 0, 0, 2080, 1216)] + [InlineData(1921, 1081, 1, 1, 1040, 608)] + public void GetPlaneBufferSizeMatchesLibaomLayout( + int width, + int height, + int subsamplingX, + int subsamplingY, + int expectedWidth, + int expectedHeight) + { + Size actual = Av1EncoderFrame.GetPlaneBufferSize(width, height, subsamplingX, subsamplingY); + + Assert.Equal(new Size(expectedWidth, expectedHeight), actual); + } + + private static void FillVisible(Buffer2D plane, int originX, int originY, int width, int height, int seed) + { + for (int y = 0; y < height; y++) + { + Span row = plane.DangerousGetRowSpan(originY + y); + for (int x = 0; x < width; x++) + { + row[originX + x] = (byte)(seed + (y * width) + x); + } + } + } + + private static void AssertReplicatedPlane(Buffer2D plane, int originX, int originY, int width, int height, int seed) + { + for (int y = 0; y < plane.Height; y++) + { + ReadOnlySpan row = plane.DangerousGetRowSpan(y); + int sourceY = Math.Clamp(y - originY, 0, height - 1); + for (int x = 0; x < row.Length; x++) + { + int sourceX = Math.Clamp(x - originX, 0, width - 1); + Assert.Equal((byte)(seed + (sourceY * width) + sourceX), row[x]); + } + } + } +} diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1ForwardQuantizerTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1ForwardQuantizerTests.cs new file mode 100644 index 0000000000..ae488bb0b8 --- /dev/null +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1ForwardQuantizerTests.cs @@ -0,0 +1,212 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using SixLabors.ImageSharp.Formats.Heif.Av1; +using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; +using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; +using SixLabors.ImageSharp.Tests.TestUtilities; + +namespace SixLabors.ImageSharp.Tests.Formats.Heif.Av1; + +/// +/// Verifies AV1 forward quantization against current libaom's fast no-matrix arithmetic. +/// +[Trait("Format", "Avif")] +public class Av1ForwardQuantizerTests +{ + /// + /// The hardware configurations covering every quantizer vector tier and the scalar fallback. + /// + private const HwIntrinsics QuantizerConfigurations = + HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic; + + /// + /// Verifies raster quantization and scan-order EOB selection at every SIMD tier. + /// + [Fact] + public void FastQuantizerMatchesLibaomReferenceAcrossHardwareWidths() + => FeatureTestRunner.RunWithHwIntrinsicsFeature(ValidateQuantizer, QuantizerConfigurations); + + /// + /// Verifies that repeated transform quantization uses only caller-owned buffers. + /// + [Fact] + public void QuantizerDoesNotAllocatePerTransform() + { + const int coefficientCount = 64; + int[] coefficients = new int[coefficientCount]; + int[] quantized = new int[coefficientCount]; + int[] dequantized = new int[coefficientCount]; + FillCoefficients(coefficients, 73); + + Av1ForwardQuantizer.QuantizeLossy( + coefficients, + quantized, + dequantized, + Av1TransformSize.Size8x8, + Av1TransformType.DctDct, + 73, + -1, + 3, + Av1BitDepth.TenBit); + + long before = GC.GetAllocatedBytesForCurrentThread(); + for (int iteration = 0; iteration < 32; iteration++) + { + Av1ForwardQuantizer.QuantizeLossy( + coefficients, + quantized, + dequantized, + Av1TransformSize.Size8x8, + Av1TransformType.DctDct, + 73, + -1, + 3, + Av1BitDepth.TenBit); + } + + Assert.Equal(0, GC.GetAllocatedBytesForCurrentThread() - before); + } + + /// + /// Exercises each transform-scale category, coded 64-point layout, quantizer range, and sample precision. + /// + private static void ValidateQuantizer() + { + ReadOnlySpan transformSizes = + [ + Av1TransformSize.Size4x4, + Av1TransformSize.Size8x8, + Av1TransformSize.Size16x16, + Av1TransformSize.Size32x32, + Av1TransformSize.Size64x16, + Av1TransformSize.Size64x64, + ]; + + ReadOnlySpan quantizerIndices = [1, 73, 173, 255]; + ReadOnlySpan bitDepths = [Av1BitDepth.EightBit, Av1BitDepth.TenBit, Av1BitDepth.TwelveBit]; + + foreach (Av1TransformSize transformSize in transformSizes) + { + int coefficientCount = transformSize.GetAdjusted().GetSize2d(); + int[] coefficients = new int[coefficientCount]; + int[] expectedQuantized = new int[coefficientCount]; + int[] expectedDequantized = new int[coefficientCount]; + int[] actualQuantized = new int[coefficientCount]; + int[] actualDequantized = new int[coefficientCount]; + + foreach (int qIndex in quantizerIndices) + { + FillCoefficients(coefficients, qIndex); + + foreach (Av1BitDepth bitDepth in bitDepths) + { + ushort expectedEndOfBlock = QuantizeReference( + coefficients, + expectedQuantized, + expectedDequantized, + transformSize, + Av1TransformType.DctDct, + qIndex, + -1, + 3, + bitDepth); + + ushort actualEndOfBlock = Av1ForwardQuantizer.QuantizeLossy( + coefficients, + actualQuantized, + actualDequantized, + transformSize, + Av1TransformType.DctDct, + qIndex, + -1, + 3, + bitDepth); + + Assert.Equal(expectedEndOfBlock, actualEndOfBlock); + Assert.Equal(expectedQuantized, actualQuantized); + Assert.Equal(expectedDequantized, actualDequantized); + } + } + } + } + + /// + /// Fills one transform with deterministic signed values spanning threshold, rounding, and clamp behavior. + /// + private static void FillCoefficients(Span coefficients, int seed) + { + for (int i = 0; i < coefficients.Length; i++) + { + coefficients[i] = (((i * 7919) + (seed * 313)) % 90001) - 45000; + } + + coefficients[0] = 0; + coefficients[1] = 1; + coefficients[2] = -1; + coefficients[3] = short.MaxValue; + coefficients[4] = -short.MaxValue; + } + + /// + /// Mirrors av1_quantize_fp_no_qmatrix from current libaom without sharing the production traversal. + /// + private static ushort QuantizeReference( + ReadOnlySpan coefficients, + Span quantizedCoefficients, + Span dequantizedCoefficients, + Av1TransformSize transformSize, + Av1TransformType transformType, + int qIndex, + int dcDeltaQ, + int acDeltaQ, + Av1BitDepth bitDepth) + { + quantizedCoefficients.Clear(); + dequantizedCoefficients.Clear(); + + int logScale = transformSize.GetScale(); + int dcDequantizer = Av1QuantizationLookup.GetDcQuant(qIndex, dcDeltaQ, bitDepth); + int acDequantizer = Av1QuantizationLookup.GetAcQuant(qIndex, acDeltaQ, bitDepth); + int dcQuantizer = (1 << 16) / dcDequantizer; + int acQuantizer = (1 << 16) / acDequantizer; + int dcRounding = RoundPowerOfTwo((64 * dcDequantizer) >> 7, logScale); + int acRounding = RoundPowerOfTwo((64 * acDequantizer) >> 7, logScale); + ReadOnlySpan scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan; + ushort endOfBlock = 0; + + for (int scanIndex = 0; scanIndex < scan.Length; scanIndex++) + { + int coefficientIndex = scan[scanIndex]; + int coefficient = coefficients[coefficientIndex]; + int coefficientSign = coefficient >> 31; + long magnitude = ((long)coefficient ^ coefficientSign) - coefficientSign; + int dequantizer = coefficientIndex == 0 ? dcDequantizer : acDequantizer; + int quantizer = coefficientIndex == 0 ? dcQuantizer : acQuantizer; + int rounding = coefficientIndex == 0 ? dcRounding : acRounding; + int quantizedMagnitude = 0; + + if ((magnitude << (1 + logScale)) >= dequantizer) + { + magnitude = Math.Clamp(magnitude + rounding, short.MinValue, short.MaxValue); + quantizedMagnitude = (int)((magnitude * quantizer) >> (16 - logScale)); + } + + if (quantizedMagnitude != 0) + { + quantizedCoefficients[coefficientIndex] = (quantizedMagnitude ^ coefficientSign) - coefficientSign; + int dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; + dequantizedCoefficients[coefficientIndex] = (dequantizedMagnitude ^ coefficientSign) - coefficientSign; + endOfBlock = (ushort)(scanIndex + 1); + } + } + + return endOfBlock; + } + + /// + /// Applies libaom's positive round-power-of-two operation. + /// + private static int RoundPowerOfTwo(int value, int shift) + => shift == 0 ? value : (value + (1 << (shift - 1))) >> shift; +} diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TileDecoderStub.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TileDecoderStub.cs index 174188a8c6..b96216b2ef 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TileDecoderStub.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TileDecoderStub.cs @@ -17,6 +17,6 @@ internal class Av1TileDecoderStub : IAv1TileReader, IAv1TileWriter { } - public Span WriteTile(int tileNum) + public ReadOnlySpan GetTileData(int tileNum) => this.tileDatas[tileNum]; } diff --git a/tests/ImageSharp.Tests/Formats/Heif/Av1/ObuFrameHeaderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/Av1/ObuFrameHeaderTests.cs index 2a6d23ab78..9437fec2af 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/Av1/ObuFrameHeaderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/Av1/ObuFrameHeaderTests.cs @@ -162,7 +162,7 @@ public class ObuFrameHeaderTests // Assign 2 Span encodedBuffer = encoded.ToArray(); IAv1TileReader tileDecoder2 = new Av1TileDecoderStub(); - Av1BitStreamReader reader2 = new(span); + Av1BitStreamReader reader2 = new(encodedBuffer); ObuReader obuReader2 = new(); // Act 2 @@ -704,6 +704,59 @@ public class ObuFrameHeaderTests Assert.Equal(bitStream.Length * 8, reader.BitPosition); } + /// + /// Verifies non-uniform tile boundaries use the next stored boundary and retain the clipped final mode-info edge. + /// + [Fact] + public void WriteNonUniformTileBoundariesRoundTrip() + { + ObuSequenceHeader sequenceHeader = GetDefaultSequenceHeader(); + ObuFrameHeader frameHeader = GetKeyFrameHeader(); + ObuTileGroupHeader tileInfo = frameHeader.TilesInfo; + tileInfo.HasUniformTileSpacing = false; + tileInfo.TileColumnCount = 2; + tileInfo.TileRowCount = 1; + tileInfo.TileSizeBytes = 1; + tileInfo.TileColumnStartModeInfo[0] = 0; + tileInfo.TileColumnStartModeInfo[1] = 64; + tileInfo.TileColumnStartModeInfo[2] = frameHeader.ModeInfoColumnCount; + tileInfo.TileRowStartModeInfo[0] = 0; + tileInfo.TileRowStartModeInfo[1] = frameHeader.ModeInfoRowCount; + + Av1TileDecoderStub sourceTiles = new(); + sourceTiles.ReadTile([0x80], 0); + sourceTiles.ReadTile([0x80], 1); + + using MemoryStream stream = new(); + ObuWriter writer = new(); + writer.WriteAll(Configuration.Default, stream, sequenceHeader, frameHeader, sourceTiles); + byte[] bitStream = stream.ToArray(); + Assert.Equal([0x00, 0x80, 0x80], bitStream[^3..]); + + Av1BitStreamReader reader = new(bitStream); + ObuReader obuReader = new(); + Av1TileDecoderStub decodedTiles = new(); + + obuReader.ReadAll(ref reader, bitStream.Length, () => decodedTiles); + + ObuTileGroupHeader actual = obuReader.FrameHeader.TilesInfo; + Assert.False(actual.HasUniformTileSpacing); + Assert.Equal(2, actual.TileColumnCount); + Assert.Equal(1, actual.TileRowCount); + Assert.Equal(1, actual.TileSizeBytes); + Assert.Equal(0, actual.TileColumnStartModeInfo[0]); + Assert.Equal(64, actual.TileColumnStartModeInfo[1]); + Assert.Equal(frameHeader.ModeInfoColumnCount, actual.TileColumnStartModeInfo[2]); + Assert.Equal(0, actual.TileRowStartModeInfo[0]); + Assert.Equal(frameHeader.ModeInfoRowCount, actual.TileRowStartModeInfo[1]); + + ReadOnlySpan expectedTileData = [0x80]; + + Assert.True(decodedTiles.GetTileData(0).SequenceEqual(expectedTileData)); + Assert.True(decodedTiles.GetTileData(1).SequenceEqual(expectedTileData)); + Assert.Equal(bitStream.Length * 8, reader.BitPosition); + } + /// /// Encodes one non-reduced sequence header with a single selected conformance failure. /// diff --git a/tests/ImageSharp.Tests/Formats/Heif/HeifEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Heif/HeifEncoderTests.cs index 04b4adb954..4d9e20c453 100644 --- a/tests/ImageSharp.Tests/Formats/Heif/HeifEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Heif/HeifEncoderTests.cs @@ -64,13 +64,19 @@ public class HeifEncoderTests } [Fact] - public void LegacyJpegRejectsZeroQuality() + public void LegacyJpegAcceptsZeroQuality() { using Image image = new(1, 1); + image[0, 0] = new Rgba32(10, 20, 30); using MemoryStream stream = new(); HeifEncoder encoder = new() { Quality = 0 }; - Assert.Throws(() => image.Save(stream, encoder)); + image.Save(stream, encoder); + + Assert.NotEqual(0, stream.Length); + stream.Position = 0; + using Image decoded = Image.Load(stream); + Assert.Equal(image.Size, decoded.Size); } [Fact] @@ -81,6 +87,7 @@ public class HeifEncoderTests HeifEncoder encoder = new() { Lossless = true }; Assert.Throws(() => image.Save(stream, encoder)); + Assert.Equal(0, stream.Length); } [Theory] @@ -93,6 +100,34 @@ public class HeifEncoderTests HeifEncoder encoder = new() { BitDepth = bitDepth }; Assert.Throws(() => image.Save(stream, encoder)); + Assert.Equal(0, stream.Length); + } + + [Fact] + public void Av1RejectsEncodingBeforeWritingOutput() + { + using Image image = new(1, 1); + using MemoryStream stream = new(); + HeifEncoder encoder = new() { CompressionMethod = HeifCompressionMethod.Av1 }; + + Assert.Throws(() => image.Save(stream, encoder)); + Assert.Equal(0, stream.Length); + } + + [Fact] + public void LegacyJpegEncodingDoesNotMutateSourceHeifMetadata() + { + using Image image = new(1, 1); + image[0, 0] = new Rgba32(10, 20, 30, 255); + HeifMetadata metadata = image.Metadata.GetHeifMetadata(); + metadata.CompressionMethod = HeifCompressionMethod.Av1; + using MemoryStream stream = new(); + HeifEncoder encoder = new(); + + image.Save(stream, encoder); + + Assert.Same(metadata, image.Metadata.GetHeifMetadata()); + Assert.Equal(HeifCompressionMethod.Av1, metadata.CompressionMethod); } [Theory]