Browse Source

Advance HEIF AV1 encoder foundation

pull/2633/head
James Jackson-South 1 month ago
parent
commit
4e6cbe75fe
  1. 50
      HEIF_IMPLEMENTATION_PLAN.md
  2. 40
      src/ImageSharp/Common/Helpers/Numerics.cs
  3. 9
      src/ImageSharp/Formats/Heif/Av1/Av1BitStreamWriter.cs
  4. 4
      src/ImageSharp/Formats/Heif/Av1/Color/Av1PresentationSampleBuffer.cs
  5. 35
      src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs
  6. 6
      src/ImageSharp/Formats/Heif/Av1/IAv1TileWriter.cs
  7. 36
      src/ImageSharp/Formats/Heif/Av1/OpenBitstreamUnit/ObuWriter.cs
  8. 334
      src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs
  9. 3
      src/ImageSharp/Formats/Heif/Av1/Pipeline/Cdef/Av1CdefFilter.cs
  10. 16
      src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainNoise.cs
  11. 12
      src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainOverlap.cs
  12. 32
      src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopRestoration/Av1SelfGuidedFilter.Operations.cs
  13. 205
      src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.Operator.cs
  14. 209
      src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.cs
  15. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DcIntraPredictor.Operator.cs
  16. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DirectionalIntraPredictor.Operations.cs
  17. 12
      src/ImageSharp/Formats/Heif/Av1/Prediction/Av1NonDirectionalIntraPredictor.Operator.cs
  18. 76
      src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PalettePredictor.Operator.cs
  19. 8
      src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PredictionDecoder.cs
  20. 30
      src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs
  21. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaPredictor.Operator.cs
  22. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundAveragePredictor.cs
  23. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundDistanceWeightedPredictor.cs
  24. 96
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundInterPredictor.cs
  25. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateAveragePredictor.cs
  26. 12
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDifferenceWeightedMaskBuilder.cs
  27. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDistanceWeightedPredictor.cs
  28. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateMaskBlendPredictor.cs
  29. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundMaskBlendPredictor.cs
  30. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1DifferenceWeightedMaskBuilder.cs
  31. 132
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1ScaledInterPredictor.cs
  32. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.Dispatch.cs
  33. 24
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.OneDimension.cs
  34. 20
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.Byte.cs
  35. 20
      src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.UInt16.cs
  36. 12
      src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyBilinearPredictor.cs
  37. 12
      src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyHorizontalPredictor.cs
  38. 12
      src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyVerticalPredictor.cs
  39. 9
      src/ImageSharp/Formats/Heif/Av1/Transform/Av1ForwardTransformer.cs
  40. 4
      src/ImageSharp/Formats/Heif/HeifEncoder.cs
  41. 14
      src/ImageSharp/Formats/Heif/HeifEncoderCore.cs
  42. 1
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1BitStreamTests.cs
  43. 98
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs
  44. 212
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1ForwardQuantizerTests.cs
  45. 2
      tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TileDecoderStub.cs
  46. 55
      tests/ImageSharp.Tests/Formats/Heif/Av1/ObuFrameHeaderTests.cs
  47. 39
      tests/ImageSharp.Tests/Formats/Heif/HeifEncoderTests.cs

50
HEIF_IMPLEMENTATION_PLAN.md

@ -754,6 +754,21 @@ Final decoder stream, presentation, and public-registration evidence on 2026-09-
`git diff --check` passes, and `.gitattributes` is unchanged. Every VSTest invocation returned `git diff --check` passes, and `.gitattributes` is unchanged. Every VSTest invocation returned
normally with no surviving test host and no Windows application-error dialog. normally with no surviving test host and no Windows application-error dialog.
SIMD traversal consistency evidence on 2026-09-02:
- [x] The shared `Numerics` vector-count helpers now cover same-lane spans and all fixed hardware widths.
AV1 decoder and current encoder hot paths use those helpers for complete-vector traversal instead of
repeating local modulo or last-vector calculations. Reverse-source indexing and algorithm-specific
partial-output groups remain explicit because they are not vector-count calculations.
- [x] Forward quantization, palette prediction, and scaled inter prediction construct width-specific SIMD
constants only when at least one vector batch will execute. Narrower dispatch tiers consume only the
remainder left by wider tiers before the scalar tail.
- [x] The net11.0 Release production assembly builds with zero warnings and zero errors. The test project
builds with zero errors while retaining the existing repository warning set. Roslynk reports zero
compiler errors, `git diff --check` passes, and `.gitattributes` is unchanged.
Foreground VSTest passes 63 of 63 focused quantizer, forward-transform, CDEF, restoration, palette,
intra, inter, film-grain, and super-resolution cases.
Decoder exit gate: Decoder exit gate:
- [x] Every supported native format and AV1 tool has exact current-main libaom production-path evidence. - [x] Every supported native format and AV1 tool has exact current-main libaom production-path evidence.
@ -768,28 +783,49 @@ Writer primitives are not an encoder. The public encoder remains incomplete unti
### 5. Define and enforce the encoder contract ### 5. Define and enforce the encoder contract
- [x] Use official libaom `main` at `a40ed1ea9e4ecc3df58a5bccb76623f2c94ae727` as the encoder syntax, probability-model, transform, quantization, filtering, and bitstream reference.
- [x] Use the existing PNG, TIFF, and JPEG encoders as the ImageSharp architecture reference: generic `Image<TPixel>` input, encoder options taking precedence over converted format metadata and codec defaults, allocator-owned temporary storage, and deterministic disposal.
- [x] Treat source pixel type, source alpha representation, and decoded source bit depth as conversion inputs, never as output-eligibility checks. Do not pre-scan pixels before encoding.
- [x] Resolve output configuration once from explicit encoder options, converted `HeifMetadata`, and AV1 defaults in that order. Sanitize only combinations that cannot describe a legal requested output, and never write resolved values back to source metadata.
- [ ] Finalize observable options for quality, effort, lossless mode, bit depth, chroma subsampling, alpha quality, metadata, and bounded sequences. - [ ] Finalize observable options for quality, effort, lossless mode, bit depth, chroma subsampling, alpha quality, metadata, and bounded sequences.
- [ ] Preserve high-bit-depth source precision through 16-bit RGB and native 10/12-bit component planes. - [ ] Preserve high-bit-depth source precision through 16-bit RGB and native 10/12-bit component planes.
- [ ] Reject unsupported combinations at the public boundary before writing output. - [ ] Reject only genuinely unsupported output combinations at the public boundary before writing output.
- [ ] Register only capabilities that the completed encoder proves. - [ ] Register only capabilities that the completed encoder proves.
Encoder data-flow contract:
1. Resolve immutable frame and sequence output settings before allocating codec state.
2. Convert each generic `ImageFrame<TPixel>` once through `PixelOperations<TPixel>` and the SIMD-first HEIF planar converter into native 8, 10, or 12-bit planes. Alpha is encoded as an auxiliary image when requested by the resolved output contract; it is not discarded through a source scan.
3. Reuse allocator-owned plane, row, block, transform, quantization, entropy, and reconstruction workspaces for the complete frame. No active path may allocate per row, block, transform, scanline, or SIMD tail.
4. Analyze and encode tiles directly from those planes, retaining reconstructed reference frames only for the bounded sequence lifetime.
5. Stream OBUs and container extents through allocator-backed chunked storage. Every ownership transfer is explicit, every owner is disposed exactly once, and no `ToArray` or file-sized copy crosses a layer boundary.
6. Iterate image frames using ImageSharp frame metadata and format-connecting metadata. Root-frame-only behavior is permitted only for an explicitly static output contract.
Encoder verification contract:
- Exercise source pixel formats independently from requested AV1 bit depth, chroma subsampling, alpha, and lossless/lossy mode.
- Run every SIMD operator through FeatureTestRunner at Vector512, Vector256, Vector128, and scalar tiers against an independent scalar oracle shaped from the same libaom revision.
- Cover discontiguous allocator buffers, constrained memory groups, cancellation, non-seekable output, multiple extents, auxiliary alpha, and bounded sequences.
- Validate produced AV1 payloads with current-main libaom and compare native planes before using ImageSharp self-decode as supplemental container coverage.
### 6. Build the complete AV1 frame encoder ### 6. Build the complete AV1 frame encoder
- [~] SIMD-first RGB-to-native-plane conversion exists locally. - [~] SIMD-first RGB-to-native-plane conversion exists locally.
- [~] Forward transform families and transform workspace exist locally. - [~] Forward transform families and transform workspace exist locally.
- [~] Symbol writer, coefficient writer, and tile writer fragments exist locally. - [~] Symbol writer, coefficient writer, and tile writer fragments exist locally.
- [ ] Connect a frame-owned encoder lifecycle using ImageSharp allocators and pools. - [~] A non-owning encoder-frame view now separates visible conversion regions from coded regions and performs complete left, top, right, bottom, and corner extension across each bordered plane. Current libaom uses 8-sample-aligned coded dimensions, a 32-sample-aligned luma stride with chroma stride derived from it, and a 64-pixel luma border for non-resized all-intra encoding. Allocator-backed luma and 4:2:0 chroma extension passed direct net11 VSTest; the containing encode operation still needs to connect matching plane rents with ordinary `using` lifetimes.
- [ ] Write compliant temporal delimiter, sequence header, frame header, tile group, metadata, and padding OBUs as required. - [~] Temporal delimiter, sequence header, frame header, and combined-frame tile-group writing exist locally. The remaining required metadata, padding, and encoder-wide syntax paths are not complete.
- [ ] Implement superblock and partition analysis for every permitted block size and partition. - [ ] Implement superblock and partition analysis for every permitted block size and partition.
- [ ] Implement intra mode search, chroma mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions. - [ ] Implement intra mode search, chroma mode search, palette, filter intra, chroma-from-luma, and intra-block copy decisions.
- [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools. - [ ] Implement inter mode search for bounded sequences, including reference selection and the decoder-supported inter tools.
- [ ] Implement transform-size/type search, forward transform, quantization, coefficient optimization, and lossless behavior. - [~] Current-libaom `av1_quantize_fp_no_qmatrix` arithmetic is implemented as a closed generic forward-quantizer family with Vector512, Vector256, Vector128, and scalar paths, raster-order output, coded 64-point coefficient limits, and scan-order EOB selection. Transform search, coefficient optimization, and lossless behavior remain.
- [ ] Implement real rate-distortion selection and make quality and effort change work, size, and output quality. - [ ] Implement real rate-distortion selection and make quality and effort change work, size, and output quality.
- [ ] Implement tile-local entropy coding and CDF update behavior. - [ ] Implement tile-local entropy coding and CDF update behavior.
- [ ] Implement legal deblocking, CDEF, restoration, super-resolution, and film-grain signaling decisions. - [ ] Implement legal deblocking, CDEF, restoration, super-resolution, and film-grain signaling decisions.
- [ ] Remove per-transform and per-block managed allocations from active encoder paths. - [~] The coefficient symbol encoder now reuses tile-lifetime level and context workspaces instead of allocating per transform. Every remaining encoder fragment must be audited before it becomes active.
- [ ] Use descending SIMD dispatch: Vector512, Vector256, Vector128, then scalar. - [~] The planar conversion, forward transform, and forward quantizer use descending SIMD dispatch: Vector512, Vector256, Vector128, then scalar. Apply the same rule to every later hot-path family.
- [ ] Verify every SIMD operator with FeatureTestRunner and an independent scalar oracle shaped from the same current-main libaom behavior. - [~] Forward-quantizer FeatureTestRunner and zero-allocation tests compare every hardware tier with an independent scan-order scalar oracle shaped from current-main libaom. Both passed direct net11 VSTest in Release.
- [~] The combined-frame writer now completes the byte-counted uncompressed frame header before starting the optional multi-tile tile-group flag, matching current libaom's separate frame-header and tile-group writers. A non-uniform two-tile round trip verifies the explicit boundaries, both tile payloads, and complete stream consumption through direct net11 VSTest in Release.
### 7. Write complete AVIF output ### 7. Write complete AVIF output

40
src/ImageSharp/Common/Helpers/Numerics.cs

@ -1024,6 +1024,46 @@ internal static class Numerics
where TVector : struct where TVector : struct
=> (uint)span.Length / (uint)Vector512<TVector>.Count; => (uint)span.Length / (uint)Vector512<TVector>.Count;
/// <summary>
/// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type.
/// </summary>
/// <typeparam name="TVector">The type of the span elements and vector lanes.</typeparam>
/// <param name="span">The given span.</param>
/// <returns>Count of vectors that safely fit into the span.</returns>
public static nuint Vector128Count<TVector>(this ReadOnlySpan<TVector> span)
where TVector : struct
=> (uint)span.Length / (uint)Vector128<TVector>.Count;
/// <summary>
/// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type.
/// </summary>
/// <typeparam name="TVector">The type of the span elements and vector lanes.</typeparam>
/// <param name="span">The given span.</param>
/// <returns>Count of vectors that safely fit into the span.</returns>
public static nuint Vector256Count<TVector>(this ReadOnlySpan<TVector> span)
where TVector : struct
=> (uint)span.Length / (uint)Vector256<TVector>.Count;
/// <summary>
/// Gets the count of vectors that safely fit into a span whose element type matches the vector lane type.
/// </summary>
/// <typeparam name="TVector">The type of the span elements and vector lanes.</typeparam>
/// <param name="span">The given span.</param>
/// <returns>Count of vectors that safely fit into the span.</returns>
public static nuint Vector512Count<TVector>(this ReadOnlySpan<TVector> span)
where TVector : struct
=> (uint)span.Length / (uint)Vector512<TVector>.Count;
/// <summary>
/// Gets the count of vectors that safely fit into the given length.
/// </summary>
/// <typeparam name="TVector">The type of the vector.</typeparam>
/// <param name="length">The given length.</param>
/// <returns>Count of vectors that safely fit into the length.</returns>
public static nuint Vector128Count<TVector>(int length)
where TVector : struct
=> (uint)length / (uint)Vector128<TVector>.Count;
/// <summary> /// <summary>
/// Gets the count of vectors that safely fit into length. /// Gets the count of vectors that safely fit into length.
/// </summary> /// </summary>

9
src/ImageSharp/Formats/Heif/Av1/Av1BitStreamWriter.cs

@ -187,10 +187,11 @@ internal ref struct Av1BitStreamWriter
} }
else else
{ {
uint extraBit = ((value + m) >> 1) - value; // libaom partitions the upper values into a shorter prefix followed by the low bit of the offset from m.
uint k = (value + m - extraBit) >> 1; uint offset = value - m;
uint k = m + (offset >> 1);
this.WriteLiteral(k, w - 1); this.WriteLiteral(k, w - 1);
this.WriteLiteral(extraBit, 1); this.WriteLiteral(offset & 1, 1);
} }
} }
@ -236,7 +237,7 @@ internal ref struct Av1BitStreamWriter
DebugGuard.IsTrue(Av1Math.Modulus8(this.BitPosition) == 0, "Writing of Tile Data only allowed on byte alignment"); DebugGuard.IsTrue(Av1Math.Modulus8(this.BitPosition) == 0, "Writing of Tile Data only allowed on byte alignment");
int wordPosition = this.BitPosition >> 3; int wordPosition = this.BitPosition >> 3;
if (this.span.Length <= wordPosition + tileData.Length) if (this.span.Length < wordPosition + tileData.Length)
{ {
this.memory.GetSpan(wordPosition + tileData.Length); this.memory.GetSpan(wordPosition + tileData.Length);
this.span = this.memory.GetEntireSpan(); this.span = this.memory.GetEntireSpan();

4
src/ImageSharp/Formats/Heif/Av1/Color/Av1PresentationSampleBuffer.cs

@ -475,7 +475,9 @@ internal sealed class Av1PresentationSampleBuffer<TSample, TBuffer> : IDisposabl
ref ushort bottomSourceBase = ref MemoryMarshal.GetReference(bottomSource); ref ushort bottomSourceBase = ref MemoryMarshal.GetReference(bottomSource);
ref ushort topDestinationBase = ref MemoryMarshal.GetReference(topDestination); ref ushort topDestinationBase = ref MemoryMarshal.GetReference(topDestination);
ref ushort bottomDestinationBase = ref MemoryMarshal.GetReference(bottomDestination); ref ushort bottomDestinationBase = ref MemoryMarshal.GetReference(bottomDestination);
for (; x + Vector128<ushort>.Count <= lastSource; x += Vector128<ushort>.Count) nuint vectorCount = topSource[..lastSource].Vector128Count<ushort>();
for (; vectorCount > 0; vectorCount--, x += Vector128<ushort>.Count)
{ {
Vector128<ushort> top0 = Vector128.LoadUnsafe(ref topSourceBase, (nuint)x); Vector128<ushort> top0 = Vector128.LoadUnsafe(ref topSourceBase, (nuint)x);
Vector128<ushort> top1 = Vector128.LoadUnsafe(ref topSourceBase, (nuint)(x + 1)); Vector128<ushort> top1 = Vector128.LoadUnsafe(ref topSourceBase, (nuint)(x + 1));

35
src/ImageSharp/Formats/Heif/Av1/Entropy/Av1SymbolEncoder.cs

@ -16,6 +16,11 @@ namespace SixLabors.ImageSharp.Formats.Heif.Av1.Entropy;
/// </summary> /// </summary>
internal class Av1SymbolEncoder : IDisposable internal class Av1SymbolEncoder : IDisposable
{ {
/// <summary>
/// The largest coefficient-context plane required after AV1 removes the uncoded half of 64-point transforms.
/// </summary>
private const int MaximumCoefficientContextCount = (Av1Constants.MaxTransformSize / 2) * (Av1Constants.MaxTransformSize / 2);
/// <summary> /// <summary>
/// The tile-adaptive intra-block-copy distribution. /// The tile-adaptive intra-block-copy distribution.
/// </summary> /// </summary>
@ -132,9 +137,14 @@ internal class Av1SymbolEncoder : IDisposable
private bool isDisposed; private bool isDisposed;
/// <summary> /// <summary>
/// The configuration providing output and coefficient-context memory. /// The reusable padded coefficient levels used to derive entropy contexts.
/// </summary>
private readonly Av1LevelBuffer levels;
/// <summary>
/// The reusable raster-order coefficient contexts for one transform.
/// </summary> /// </summary>
private readonly Configuration configuration; private readonly IMemoryOwner<sbyte> coefficientContexts;
/// <summary> /// <summary>
/// The range writer producing the current tile payload. /// The range writer producing the current tile payload.
@ -178,7 +188,8 @@ internal class Av1SymbolEncoder : IDisposable
this.coefficientsBaseEndOfBlock = Av1DefaultDistributions.GetBaseEndOfBlock(qIndex); this.coefficientsBaseEndOfBlock = Av1DefaultDistributions.GetBaseEndOfBlock(qIndex);
this.dcSign = Av1DefaultDistributions.GetDcSign(qIndex); this.dcSign = Av1DefaultDistributions.GetDcSign(qIndex);
this.endOfBlockExtra = Av1DefaultDistributions.GetEndOfBlockExtra(qIndex); this.endOfBlockExtra = Av1DefaultDistributions.GetEndOfBlockExtra(qIndex);
this.configuration = configuration; this.levels = new(configuration);
this.coefficientContexts = configuration.MemoryAllocator.Allocate<sbyte>(MaximumCoefficientContextCount);
this.writer = new(configuration, initialSize, updateCdf); this.writer = new(configuration, initialSize, updateCdf);
this.baseQIndex = qIndex; this.baseQIndex = qIndex;
} }
@ -275,9 +286,11 @@ internal class Av1SymbolEncoder : IDisposable
ref Av1SymbolWriter w = ref this.writer; ref Av1SymbolWriter w = ref this.writer;
// AV1 omits high-frequency coefficients beyond 32 samples on every 64-point transform dimension. // AV1 omits high-frequency coefficients beyond 32 samples on every 64-point transform dimension. The tile
using Av1LevelBuffer levels = new(this.configuration, new Size(width, height)); // owns maximum-sized workspaces so repeated transform coding changes only their active views.
Span<sbyte> coefficientContexts = new sbyte[width * height]; this.levels.Reset(new Size(width, height));
Span<sbyte> coefficientContexts = this.coefficientContexts.Memory.Span[..(width * height)];
coefficientContexts.Clear();
Guard.MustBeLessThan((int)transformSizeContext, (int)Av1TransformSize.AllSizes, nameof(transformSizeContext)); Guard.MustBeLessThan((int)transformSizeContext, (int)Av1TransformSize.AllSizes, nameof(transformSizeContext));
@ -288,7 +301,7 @@ internal class Av1SymbolEncoder : IDisposable
return 0; return 0;
} }
levels.Initialize(coefficientBuffer); this.levels.Initialize(coefficientBuffer);
if (componentType == Av1ComponentType.Luminance) if (componentType == Av1ComponentType.Luminance)
{ {
this.WriteTransformType(transformType, transformSize, useReducedTransformSet, this.baseQIndex, filterIntraMode, intraDirection); this.WriteTransformType(transformType, transformSize, useReducedTransformSet, this.baseQIndex, filterIntraMode, intraDirection);
@ -296,14 +309,14 @@ internal class Av1SymbolEncoder : IDisposable
this.WriteEndOfBlockPosition(endOfBlock, componentType, transformClass, transformSize, transformSizeContext); this.WriteEndOfBlockPosition(endOfBlock, componentType, transformClass, transformSize, transformSizeContext);
Av1SymbolContextHelper.GetNzMapContexts(levels, scan, endOfBlock, transformSize, transformClass, coefficientContexts); Av1SymbolContextHelper.GetNzMapContexts(this.levels, scan, endOfBlock, transformSize, transformClass, coefficientContexts);
int limitedTransformSizeContext = Math.Min((int)transformSizeContext, (int)Av1TransformSize.Size32x32); int limitedTransformSizeContext = Math.Min((int)transformSizeContext, (int)Av1TransformSize.Size32x32);
for (c = endOfBlock - 1; c >= 0; --c) for (c = endOfBlock - 1; c >= 0; --c)
{ {
short pos = scan[c]; short pos = scan[c];
int v = coefficientBuffer[pos]; int v = coefficientBuffer[pos];
short coeffContext = coefficientContexts[pos]; short coeffContext = coefficientContexts[pos];
Point position = levels.GetPosition(pos); Point position = this.levels.GetPosition(pos);
int level = Math.Abs(v); int level = Math.Abs(v);
if (c == endOfBlock - 1) if (c == endOfBlock - 1)
@ -319,7 +332,7 @@ internal class Av1SymbolEncoder : IDisposable
{ {
// Base-range symbols extend levels above the two base levels in fixed-size chunks. // Base-range symbols extend levels above the two base levels in fixed-size chunks.
int baseRange = level - 1 - Av1Constants.BaseLevelsCount; int baseRange = level - 1 - Av1Constants.BaseLevelsCount;
int baseRangeContext = Av1SymbolContextHelper.GetBaseRangeContext(levels, position, transformClass); int baseRangeContext = Av1SymbolContextHelper.GetBaseRangeContext(this.levels, position, transformClass);
for (int idx = 0; idx < Av1Constants.CoefficientBaseRange; idx += Av1Constants.BaseRangeSizeMinus1) for (int idx = 0; idx < Av1Constants.CoefficientBaseRange; idx += Av1Constants.BaseRangeSizeMinus1)
{ {
int k = Math.Min(baseRange - idx, Av1Constants.BaseRangeSizeMinus1); int k = Math.Min(baseRange - idx, Av1Constants.BaseRangeSizeMinus1);
@ -429,6 +442,8 @@ internal class Av1SymbolEncoder : IDisposable
{ {
if (!this.isDisposed) if (!this.isDisposed)
{ {
this.coefficientContexts.Dispose();
this.levels.Dispose();
this.writer.Dispose(); this.writer.Dispose();
this.isDisposed = true; this.isDisposed = true;
} }

6
src/ImageSharp/Formats/Heif/Av1/IAv1TileWriter.cs

@ -9,11 +9,11 @@ namespace SixLabors.ImageSharp.Formats.Heif.Av1;
internal interface IAv1TileWriter internal interface IAv1TileWriter
{ {
/// <summary> /// <summary>
/// Write the information for a single tile. /// Gets the encoded bytes for a single tile.
/// </summary> /// </summary>
/// <param name="tileNum">The index of the tile that is to be read.</param> /// <param name="tileNum">The index of the encoded tile.</param>
/// <returns> /// <returns>
/// The bytes of encoded data in the bitstream dedicated to this tile. /// The bytes of encoded data in the bitstream dedicated to this tile.
/// </returns> /// </returns>
Span<byte> WriteTile(int tileNum); ReadOnlySpan<byte> GetTileData(int tileNum);
} }

36
src/ImageSharp/Formats/Heif/Av1/OpenBitstreamUnit/ObuWriter.cs

@ -29,7 +29,7 @@ internal class ObuWriter
// The allocation expands when necessary; this initial size avoids repeated growth for // The allocation expands when necessary; this initial size avoids repeated growth for
// the small headers and tiles produced by the current still-image encoder. // the small headers and tiles produced by the current still-image encoder.
int initialBufferSize = 2000; int initialBufferSize = 2000;
AutoExpandingMemory<byte> buffer = new(configuration, initialBufferSize); using AutoExpandingMemory<byte> buffer = new(configuration, initialBufferSize);
Av1BitStreamWriter writer = new(buffer); Av1BitStreamWriter writer = new(buffer);
WriteObuHeaderAndSize(stream, ObuType.TemporalDelimiter, []); WriteObuHeaderAndSize(stream, ObuType.TemporalDelimiter, []);
@ -173,9 +173,6 @@ internal class ObuWriter
{ {
// AV1 fixes this RGB identity-matrix combination to full-range 4:4:4 and omits // AV1 fixes this RGB identity-matrix combination to full-range 4:4:4 and omits
// the range and subsampling fields used by YUV configurations. // the range and subsampling fields used by YUV configurations.
colorConfig.ColorRange = true;
colorConfig.SubSamplingX = false;
colorConfig.SubSamplingY = false;
} }
else else
{ {
@ -335,13 +332,18 @@ internal class ObuWriter
else else
{ {
int startSuperBlock = 0; int startSuperBlock = 0;
int i = 0; for (int i = 0; i < tileInfo.TileColumnCount; i++)
for (; startSuperBlock < superblockColumnCount; i++)
{ {
uint widthInSuperBlocks = (uint)((tileInfo.TileColumnStartModeInfo[i] >> superblockShift) - startSuperBlock); int endSuperBlock = i == tileInfo.TileColumnCount - 1
? superblockColumnCount
: tileInfo.TileColumnStartModeInfo[i + 1] >> superblockShift;
// The stored terminal boundary is clipped to the visible mode-info width. libaom retains the exact
// superblock endpoint, so the final tile uses the derived frame-wide superblock count instead.
uint widthInSuperBlocks = (uint)(endSuperBlock - startSuperBlock);
uint maxWidth = (uint)Math.Min(superblockColumnCount - startSuperBlock, tileInfo.MaxTileWidthSuperblock); uint maxWidth = (uint)Math.Min(superblockColumnCount - startSuperBlock, tileInfo.MaxTileWidthSuperblock);
writer.WriteNonSymmetric(widthInSuperBlocks - 1, maxWidth); writer.WriteNonSymmetric(widthInSuperBlocks - 1, maxWidth);
startSuperBlock += (int)widthInSuperBlocks; startSuperBlock = endSuperBlock;
} }
if (startSuperBlock != superblockColumnCount) if (startSuperBlock != superblockColumnCount)
@ -350,12 +352,17 @@ internal class ObuWriter
} }
startSuperBlock = 0; startSuperBlock = 0;
for (i = 0; startSuperBlock < superblockRowCount; i++) for (int i = 0; i < tileInfo.TileRowCount; i++)
{ {
uint heightInSuperBlocks = (uint)((tileInfo.TileRowStartModeInfo[i] >> superblockShift) - startSuperBlock); int endSuperBlock = i == tileInfo.TileRowCount - 1
? superblockRowCount
: tileInfo.TileRowStartModeInfo[i + 1] >> superblockShift;
// As with columns, the final visible mode-info boundary may end inside its containing superblock.
uint heightInSuperBlocks = (uint)(endSuperBlock - startSuperBlock);
uint maxHeight = (uint)Math.Min(superblockRowCount - startSuperBlock, tileInfo.MaxTileHeightSuperblock); uint maxHeight = (uint)Math.Min(superblockRowCount - startSuperBlock, tileInfo.MaxTileHeightSuperblock);
writer.WriteNonSymmetric(heightInSuperBlocks - 1, maxHeight); writer.WriteNonSymmetric(heightInSuperBlocks - 1, maxHeight);
startSuperBlock += (int)heightInSuperBlocks; startSuperBlock = endSuperBlock;
} }
if (startSuperBlock != superblockRowCount) if (startSuperBlock != superblockRowCount)
@ -521,6 +528,11 @@ internal class ObuWriter
private static void WriteTileGroup(ref Av1BitStreamWriter writer, ObuTileGroupHeader tileInfo, IAv1TileWriter tileWriter) private static void WriteTileGroup(ref Av1BitStreamWriter writer, ObuTileGroupHeader tileInfo, IAv1TileWriter tileWriter)
{ {
int tileCount = tileInfo.TileColumnCount * tileInfo.TileRowCount; int tileCount = tileInfo.TileColumnCount * tileInfo.TileRowCount;
// libaom starts the tile-group header at the next byte after the uncompressed frame header. This
// boundary is required before the optional flag because the flag belongs to tile_group_obu syntax.
AlignToByteBoundary(ref writer);
if (tileCount > 1) if (tileCount > 1)
{ {
// A combined OBU_FRAME has implicit complete-frame tile bounds. The reference decoder still // A combined OBU_FRAME has implicit complete-frame tile bounds. The reference decoder still
@ -544,7 +556,7 @@ internal class ObuWriter
int tileCount = tileInfo.TileColumnCount * tileInfo.TileRowCount; int tileCount = tileInfo.TileColumnCount * tileInfo.TileRowCount;
for (int tileNum = 0; tileNum < tileCount; tileNum++) for (int tileNum = 0; tileNum < tileCount; tileNum++)
{ {
Span<byte> tileData = tileWriter.WriteTile(tileNum); ReadOnlySpan<byte> tileData = tileWriter.GetTileData(tileNum);
if (tileNum != tileCount - 1 && tileCount > 1) if (tileNum != tileCount - 1 && tileCount > 1)
{ {
writer.WriteLittleEndian((uint)tileData.Length - 1U, tileInfo.TileSizeBytes); writer.WriteLittleEndian((uint)tileData.Length - 1U, tileInfo.TileSizeBytes);

334
src/ImageSharp/Formats/Heif/Av1/Pipeline/Av1EncoderFrame.cs

@ -0,0 +1,334 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using SixLabors.ImageSharp.Formats.Heif.Components;
using SixLabors.ImageSharp.Memory;
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
/// <summary>
/// Provides non-owning visible and coded views over operation-scoped AV1 component planes.
/// </summary>
/// <typeparam name="TSample">The native unsigned sample storage type.</typeparam>
internal readonly struct Av1EncoderFrame<TSample>
where TSample : unmanaged
{
/// <summary>
/// The base-two alignment exponent applied to coded frame dimensions.
/// </summary>
private const int CodedDimensionAlignmentLog2 = 3;
/// <summary>
/// The base-two alignment exponent applied to the physical luma row stride.
/// </summary>
private const int LumaStrideAlignmentLog2 = 5;
/// <summary>
/// The physical luma border required by non-resized all-intra encoding.
/// </summary>
public const int LumaBorder = 64;
/// <summary>
/// Initializes a new instance of the <see cref="Av1EncoderFrame{TSample}"/> struct for a monochrome frame.
/// </summary>
/// <param name="luma">The coded luma region inside the bordered plane owned by the encode operation.</param>
/// <param name="width">The visible luma width.</param>
/// <param name="height">The visible luma height.</param>
/// <param name="bitDepth">The native component precision.</param>
public Av1EncoderFrame(Buffer2DRegion<TSample> luma, int width, int height, int bitDepth)
: this(luma, default, default, width, height, bitDepth, Av1ColorFormat.Yuv400, 0, 0)
{
}
/// <summary>
/// Initializes a new instance of the <see cref="Av1EncoderFrame{TSample}"/> struct for a color frame.
/// </summary>
/// <param name="luma">The coded luma region inside the bordered plane owned by the encode operation.</param>
/// <param name="chromaBlue">The coded blue-difference region inside its bordered plane.</param>
/// <param name="chromaRed">The coded red-difference region inside its bordered plane.</param>
/// <param name="width">The visible luma width.</param>
/// <param name="height">The visible luma height.</param>
/// <param name="bitDepth">The native component precision.</param>
/// <param name="colorFormat">The native luma and chroma sampling layout.</param>
/// <param name="chromaPositionX">The horizontal chroma position in half-luma-sample units.</param>
/// <param name="chromaPositionY">The vertical chroma position in half-luma-sample units.</param>
public Av1EncoderFrame(
Buffer2DRegion<TSample> luma,
Buffer2DRegion<TSample> chromaBlue,
Buffer2DRegion<TSample> chromaRed,
int width,
int height,
int bitDepth,
Av1ColorFormat colorFormat,
int chromaPositionX,
int chromaPositionY)
{
this.Width = width;
this.Height = height;
this.LumaBitDepth = bitDepth;
this.ChromaBitDepth = bitDepth;
this.IsMonochrome = colorFormat == Av1ColorFormat.Yuv400;
this.ChromaSubsamplingX = colorFormat is Av1ColorFormat.Yuv420 or Av1ColorFormat.Yuv422 ? 1 : 0;
this.ChromaSubsamplingY = colorFormat == Av1ColorFormat.Yuv420 ? 1 : 0;
this.ChromaPositionX = chromaPositionX;
this.ChromaPositionY = chromaPositionY;
this.CodedWidth = luma.Width;
this.CodedHeight = luma.Height;
int visibleChromaWidth = (width + this.ChromaSubsamplingX) >> this.ChromaSubsamplingX;
int visibleChromaHeight = (height + this.ChromaSubsamplingY) >> this.ChromaSubsamplingY;
Buffer2DRegion<TSample> visibleChromaBlue = this.IsMonochrome
? default
: chromaBlue.GetSubRegion(0, 0, visibleChromaWidth, visibleChromaHeight);
Buffer2DRegion<TSample> visibleChromaRed = this.IsMonochrome
? default
: chromaRed.GetSubRegion(0, 0, visibleChromaWidth, visibleChromaHeight);
this.View = new PlanarView(
luma.GetSubRegion(0, 0, width, height),
visibleChromaBlue,
visibleChromaRed,
width,
height,
bitDepth,
colorFormat,
chromaPositionX,
chromaPositionY);
this.CodedView = new PlanarView(
luma,
chromaBlue,
chromaRed,
this.CodedWidth,
this.CodedHeight,
bitDepth,
colorFormat,
chromaPositionX,
chromaPositionY);
}
/// <summary>
/// Gets the writable component-plane view used by closed generic conversion and coding operations.
/// </summary>
public PlanarView View { get; }
/// <summary>
/// Gets the writable coded component planes used by block coding and reconstruction.
/// </summary>
public PlanarView CodedView { get; }
/// <summary>
/// Gets the visible luma width.
/// </summary>
public int Width { get; }
/// <summary>
/// Gets the visible luma height.
/// </summary>
public int Height { get; }
/// <summary>
/// Gets the luma width rounded up to the fixed coding-block boundary.
/// </summary>
public int CodedWidth { get; }
/// <summary>
/// Gets the luma height rounded up to the fixed coding-block boundary.
/// </summary>
public int CodedHeight { get; }
/// <summary>
/// Gets the native luma sample precision.
/// </summary>
public int LumaBitDepth { get; }
/// <summary>
/// Gets the native chroma sample precision.
/// </summary>
public int ChromaBitDepth { get; }
/// <summary>
/// Gets a value indicating whether the frame contains only luma samples.
/// </summary>
public bool IsMonochrome { get; }
/// <summary>
/// Gets the horizontal chroma subsampling shift.
/// </summary>
public int ChromaSubsamplingX { get; }
/// <summary>
/// Gets the vertical chroma subsampling shift.
/// </summary>
public int ChromaSubsamplingY { get; }
/// <summary>
/// Gets the horizontal chroma position in half-luma-sample units.
/// </summary>
public int ChromaPositionX { get; }
/// <summary>
/// Gets the vertical chroma position in half-luma-sample units.
/// </summary>
public int ChromaPositionY { get; }
/// <summary>
/// Calculates the physical dimensions required for an all-intra component plane.
/// </summary>
/// <param name="width">The visible luma width.</param>
/// <param name="height">The visible luma height.</param>
/// <param name="subsamplingX">The plane's horizontal subsampling shift.</param>
/// <param name="subsamplingY">The plane's vertical subsampling shift.</param>
/// <returns>The physical plane dimensions, including its complete border and row padding.</returns>
public static Size GetPlaneBufferSize(int width, int height, int subsamplingX, int subsamplingY)
{
int codedWidth = Av1Math.AlignPowerOf2(width, CodedDimensionAlignmentLog2);
int codedHeight = Av1Math.AlignPowerOf2(height, CodedDimensionAlignmentLog2);
// libaom aligns the complete luma row before deriving a subsampled plane's stride.
// Aligning chroma independently would produce a different physical layout for narrow or odd-sized frames.
int lumaStride = Av1Math.AlignPowerOf2(codedWidth + (2 * LumaBorder), LumaStrideAlignmentLog2);
int planeStride = lumaStride >> subsamplingX;
int planeBorderHeight = LumaBorder >> subsamplingY;
return new Size(planeStride, (codedHeight >> subsamplingY) + (2 * planeBorderHeight));
}
/// <summary>
/// Extends the visible edge samples through the coded padding.
/// </summary>
public void ExtendBorders()
=> this.CodedView.ExtendBorders(this.Width, this.Height);
/// <summary>
/// Replicates the visible edge samples through a plane's complete physical border.
/// </summary>
private static void ExtendPlane(Buffer2DRegion<TSample> plane, int visibleWidth, int visibleHeight)
{
Buffer2D<TSample> buffer = plane.Buffer;
Rectangle bounds = plane.Bounds;
for (int y = 0; y < visibleHeight; y++)
{
Span<TSample> row = buffer.DangerousGetRowSpan(bounds.Y + y);
// libaom fills both physical borders and the right-hand coded alignment from the nearest visible sample.
row[..bounds.X].Fill(row[bounds.X]);
row[(bounds.X + visibleWidth)..].Fill(row[bounds.X + visibleWidth - 1]);
}
// Horizontal extension runs first so copying the first and last visible rows also initializes both corners.
ReadOnlySpan<TSample> firstVisibleRow = buffer.DangerousGetRowSpan(bounds.Y);
for (int y = 0; y < bounds.Y; y++)
{
firstVisibleRow.CopyTo(buffer.DangerousGetRowSpan(y));
}
ReadOnlySpan<TSample> finalVisibleRow = buffer.DangerousGetRowSpan(bounds.Y + visibleHeight - 1);
for (int y = bounds.Y + visibleHeight; y < buffer.Height; y++)
{
finalVisibleRow.CopyTo(buffer.DangerousGetRowSpan(y));
}
}
/// <summary>
/// Provides a non-owning component-plane view for generic hot-path operations.
/// </summary>
internal readonly struct PlanarView : IHeifPlanarSampleBuffer<TSample>
{
/// <summary>
/// The writable luma plane.
/// </summary>
private readonly Buffer2DRegion<TSample> luma;
/// <summary>
/// The writable blue-difference plane, or the default region for monochrome frames.
/// </summary>
private readonly Buffer2DRegion<TSample> chromaBlue;
/// <summary>
/// The writable red-difference plane, or the default region for monochrome frames.
/// </summary>
private readonly Buffer2DRegion<TSample> chromaRed;
/// <summary>
/// Initializes a new instance of the <see cref="PlanarView"/> struct.
/// </summary>
public PlanarView(
Buffer2DRegion<TSample> luma,
Buffer2DRegion<TSample> chromaBlue,
Buffer2DRegion<TSample> chromaRed,
int width,
int height,
int bitDepth,
Av1ColorFormat colorFormat,
int chromaPositionX,
int chromaPositionY)
{
this.luma = luma;
this.chromaBlue = chromaBlue;
this.chromaRed = chromaRed;
this.Width = width;
this.Height = height;
this.LumaBitDepth = bitDepth;
this.ChromaBitDepth = bitDepth;
this.IsMonochrome = colorFormat == Av1ColorFormat.Yuv400;
this.ChromaSubsamplingX = colorFormat is Av1ColorFormat.Yuv420 or Av1ColorFormat.Yuv422 ? 1 : 0;
this.ChromaSubsamplingY = colorFormat == Av1ColorFormat.Yuv420 ? 1 : 0;
this.ChromaPositionX = chromaPositionX;
this.ChromaPositionY = chromaPositionY;
}
/// <inheritdoc/>
public int Width { get; }
/// <inheritdoc/>
public int Height { get; }
/// <inheritdoc/>
public int LumaBitDepth { get; }
/// <inheritdoc/>
public int ChromaBitDepth { get; }
/// <inheritdoc/>
public bool IsMonochrome { get; }
/// <inheritdoc/>
public int ChromaSubsamplingX { get; }
/// <inheritdoc/>
public int ChromaSubsamplingY { get; }
/// <inheritdoc/>
public int ChromaPositionX { get; }
/// <inheritdoc/>
public int ChromaPositionY { get; }
/// <summary>
/// Replicates the visible component edges through the coded padding.
/// </summary>
/// <param name="visibleWidth">The visible luma width.</param>
/// <param name="visibleHeight">The visible luma height.</param>
public void ExtendBorders(int visibleWidth, int visibleHeight)
{
ExtendPlane(this.luma, visibleWidth, visibleHeight);
if (!this.IsMonochrome)
{
int visibleChromaWidth = (visibleWidth + this.ChromaSubsamplingX) >> this.ChromaSubsamplingX;
int visibleChromaHeight = (visibleHeight + this.ChromaSubsamplingY) >> this.ChromaSubsamplingY;
ExtendPlane(this.chromaBlue, visibleChromaWidth, visibleChromaHeight);
ExtendPlane(this.chromaRed, visibleChromaWidth, visibleChromaHeight);
}
}
/// <inheritdoc/>
public Span<TSample> GetLumaRowSpan(int row) => this.luma.DangerousGetRowSpan(row);
/// <inheritdoc/>
public Span<TSample> GetChromaBlueRowSpan(int row) => this.chromaBlue.DangerousGetRowSpan(row);
/// <inheritdoc/>
public Span<TSample> GetChromaRedRowSpan(int row) => this.chromaRed.DangerousGetRowSpan(row);
}
}

3
src/ImageSharp/Formats/Heif/Av1/Pipeline/Cdef/Av1CdefFilter.cs

@ -60,7 +60,8 @@ internal static partial class Av1CdefFilter
int firstDestinationRow = destinationOffset + (row * destinationStride); int firstDestinationRow = destinationOffset + (row * destinationStride);
int secondDestinationRow = firstDestinationRow + destinationStride; int secondDestinationRow = firstDestinationRow + destinationStride;
int column = 0; int column = 0;
for (; column <= width - Vector128<byte>.Count; column += Vector128<byte>.Count) nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<byte> first = Vector128.LoadUnsafe(ref sourceBase, (nuint)(firstSourceRow + column)); Vector128<byte> first = Vector128.LoadUnsafe(ref sourceBase, (nuint)(firstSourceRow + column));
Vector128<byte> second = Vector128.LoadUnsafe(ref sourceBase, (nuint)(secondSourceRow + column)); Vector128<byte> second = Vector128.LoadUnsafe(ref sourceBase, (nuint)(secondSourceRow + column));

16
src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainNoise.cs

@ -264,11 +264,11 @@ internal static class Av1FilmGrainNoise
int sampleRowOffset = row * sampleStride; int sampleRowOffset = row * sampleStride;
int grainRowOffset = row * grainStride; int grainRowOffset = row * grainStride;
int column = 0; int column = 0;
int vectorEnd = width - Vector256<int>.Count; int vectorEnd = (int)(Numerics.Vector256Count<int>(width) * (nuint)Vector256<int>.Count);
// Samples, grain, and scale indices share the same lane coordinate. No permutation is required between // Samples, grain, and scale indices share the same lane coordinate. No permutation is required between
// the lookup, fixed-point multiply, clipping, and native-sample store. // the lookup, fixed-point multiply, clipping, and native-sample store.
for (; column <= vectorEnd; column += Vector256<int>.Count) for (; column < vectorEnd; column += Vector256<int>.Count)
{ {
ref TSample destination = ref Unsafe.Add(ref sampleBase, sampleRowOffset + column); ref TSample destination = ref Unsafe.Add(ref sampleBase, sampleRowOffset + column);
Vector256<int> source = Av1FilmGrainSampleOperations<TSample>.Load8(ref destination); Vector256<int> source = Av1FilmGrainSampleOperations<TSample>.Load8(ref destination);
@ -319,11 +319,11 @@ internal static class Av1FilmGrainNoise
int sampleRowOffset = row * sampleStride; int sampleRowOffset = row * sampleStride;
int grainRowOffset = row * grainStride; int grainRowOffset = row * grainStride;
int column = 0; int column = 0;
int vectorEnd = width - Vector128<int>.Count; int vectorEnd = (int)(Numerics.Vector128Count<int>(width) * (nuint)Vector128<int>.Count);
// Four scalar table reads assemble the scale vector; the rest of the normative grain equation remains // Four scalar table reads assemble the scale vector; the rest of the normative grain equation remains
// lane-wise, including interpolation for 10- and 12-bit coordinates. // lane-wise, including interpolation for 10- and 12-bit coordinates.
for (; column <= vectorEnd; column += Vector128<int>.Count) for (; column < vectorEnd; column += Vector128<int>.Count)
{ {
ref TSample destination = ref Unsafe.Add(ref sampleBase, sampleRowOffset + column); ref TSample destination = ref Unsafe.Add(ref sampleBase, sampleRowOffset + column);
Vector128<int> source = Av1FilmGrainSampleOperations<TSample>.Load4(ref destination); Vector128<int> source = Av1FilmGrainSampleOperations<TSample>.Load4(ref destination);
@ -570,11 +570,11 @@ internal static class Av1FilmGrainNoise
int chromaRowOffset = row * chromaStride; int chromaRowOffset = row * chromaStride;
int grainRowOffset = row * grainStride; int grainRowOffset = row * grainStride;
int column = 0; int column = 0;
int vectorEnd = width - Vector256<int>.Count; int vectorEnd = (int)(Numerics.Vector256Count<int>(width) * (nuint)Vector256<int>.Count);
// Each lane represents one chroma coordinate and its corresponding reconstructed-luma coordinate. The Q6 // Each lane represents one chroma coordinate and its corresponding reconstructed-luma coordinate. The Q6
// luma/chroma blend is clamped to a legal sample code before it becomes a scaling-table index. // luma/chroma blend is clamped to a legal sample code before it becomes a scaling-table index.
for (; column <= vectorEnd; column += Vector256<int>.Count) for (; column < vectorEnd; column += Vector256<int>.Count)
{ {
ref TSample lumaSource = ref Unsafe.Add(ref lumaRow, column << subsamplingX); ref TSample lumaSource = ref Unsafe.Add(ref lumaRow, column << subsamplingX);
Vector256<int> averageLuma = Av1FilmGrainSampleOperations<TSample>.LoadChromaLuma8(ref lumaSource, subsamplingX); Vector256<int> averageLuma = Av1FilmGrainSampleOperations<TSample>.LoadChromaLuma8(ref lumaSource, subsamplingX);
@ -701,11 +701,11 @@ internal static class Av1FilmGrainNoise
int chromaRowOffset = row * chromaStride; int chromaRowOffset = row * chromaStride;
int grainRowOffset = row * grainStride; int grainRowOffset = row * grainStride;
int column = 0; int column = 0;
int vectorEnd = width - Vector128<int>.Count; int vectorEnd = (int)(Numerics.Vector128Count<int>(width) * (nuint)Vector128<int>.Count);
// The four-lane path preserves the same coordinate alignment and Q6 scaling-index arithmetic. Only the // The four-lane path preserves the same coordinate alignment and Q6 scaling-index arithmetic. Only the
// table read changes from a hardware gather to four scalar reads assembled into a vector. // table read changes from a hardware gather to four scalar reads assembled into a vector.
for (; column <= vectorEnd; column += Vector128<int>.Count) for (; column < vectorEnd; column += Vector128<int>.Count)
{ {
ref TSample lumaSource = ref Unsafe.Add(ref lumaRow, column << subsamplingX); ref TSample lumaSource = ref Unsafe.Add(ref lumaRow, column << subsamplingX);
Vector128<int> averageLuma = Av1FilmGrainSampleOperations<TSample>.LoadChromaLuma4(ref lumaSource, subsamplingX); Vector128<int> averageLuma = Av1FilmGrainSampleOperations<TSample>.LoadChromaLuma4(ref lumaSource, subsamplingX);

12
src/ImageSharp/Formats/Heif/Av1/Pipeline/FilmGrain/Av1FilmGrainOverlap.cs

@ -210,8 +210,8 @@ internal static class Av1FilmGrainOverlap
Vector512<int> rounding = Vector512.Create(16); Vector512<int> rounding = Vector512.Create(16);
Vector512<int> minima = Vector512.Create(minimum); Vector512<int> minima = Vector512.Create(minimum);
Vector512<int> maxima = Vector512.Create(maximum); Vector512<int> maxima = Vector512.Create(maximum);
int vectorEnd = width - Vector512<int>.Count; nuint vectorCount = Numerics.Vector512Count<int>(width - column);
for (; column <= vectorEnd; column += Vector512<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<int>.Count)
{ {
Vector512<int> leftValues = Vector512.LoadUnsafe(ref leftBase, (nuint)column); Vector512<int> leftValues = Vector512.LoadUnsafe(ref leftBase, (nuint)column);
Vector512<int> rightValues = Vector512.LoadUnsafe(ref rightBase, (nuint)column); Vector512<int> rightValues = Vector512.LoadUnsafe(ref rightBase, (nuint)column);
@ -257,8 +257,8 @@ internal static class Av1FilmGrainOverlap
Vector256<int> rounding = Vector256.Create(16); Vector256<int> rounding = Vector256.Create(16);
Vector256<int> minima = Vector256.Create(minimum); Vector256<int> minima = Vector256.Create(minimum);
Vector256<int> maxima = Vector256.Create(maximum); Vector256<int> maxima = Vector256.Create(maximum);
int vectorEnd = width - Vector256<int>.Count; nuint vectorCount = Numerics.Vector256Count<int>(width - column);
for (; column <= vectorEnd; column += Vector256<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<int>.Count)
{ {
Vector256<int> leftValues = Vector256.LoadUnsafe(ref leftBase, (nuint)column); Vector256<int> leftValues = Vector256.LoadUnsafe(ref leftBase, (nuint)column);
Vector256<int> rightValues = Vector256.LoadUnsafe(ref rightBase, (nuint)column); Vector256<int> rightValues = Vector256.LoadUnsafe(ref rightBase, (nuint)column);
@ -304,8 +304,8 @@ internal static class Av1FilmGrainOverlap
Vector128<int> rounding = Vector128.Create(16); Vector128<int> rounding = Vector128.Create(16);
Vector128<int> minima = Vector128.Create(minimum); Vector128<int> minima = Vector128.Create(minimum);
Vector128<int> maxima = Vector128.Create(maximum); Vector128<int> maxima = Vector128.Create(maximum);
int vectorEnd = width - Vector128<int>.Count; nuint vectorCount = Numerics.Vector128Count<int>(width - column);
for (; column <= vectorEnd; column += Vector128<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<int>.Count)
{ {
Vector128<int> leftValues = Vector128.LoadUnsafe(ref leftBase, (nuint)column); Vector128<int> leftValues = Vector128.LoadUnsafe(ref leftBase, (nuint)column);
Vector128<int> rightValues = Vector128.LoadUnsafe(ref rightBase, (nuint)column); Vector128<int> rightValues = Vector128.LoadUnsafe(ref rightBase, (nuint)column);

32
src/ImageSharp/Formats/Heif/Av1/Pipeline/LoopRestoration/Av1SelfGuidedFilter.Operations.cs

@ -252,8 +252,8 @@ internal static partial class Av1SelfGuidedFilter
Vector256<int> squareCarry = Vector256<int>.Zero; Vector256<int> squareCarry = Vector256<int>.Zero;
Vector256<int> sumCarry = Vector256<int>.Zero; Vector256<int> sumCarry = Vector256<int>.Zero;
int column = 0; int column = 0;
int vectorEnd = width - Vector256<int>.Count; nuint vectorCount = Numerics.Vector256Count<int>(width - column);
for (; column <= vectorEnd; column += Vector256<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<int>.Count)
{ {
// Eight packed 16-bit samples become eight 32-bit lanes. The prefix scans mirror // Eight packed 16-bit samples become eight 32-bit lanes. The prefix scans mirror
// the reference decoder's scan_32, and the replicated carry joins consecutive vector batches. // the reference decoder's scan_32, and the replicated carry joins consecutive vector batches.
@ -323,8 +323,8 @@ internal static partial class Av1SelfGuidedFilter
Vector128<int> squareCarry = Vector128<int>.Zero; Vector128<int> squareCarry = Vector128<int>.Zero;
Vector128<int> sumCarry = Vector128<int>.Zero; Vector128<int> sumCarry = Vector128<int>.Zero;
int column = 0; int column = 0;
int vectorEnd = width - Vector128<int>.Count; nuint vectorCount = Numerics.Vector128Count<int>(width - column);
for (; column <= vectorEnd; column += Vector128<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<int>.Count)
{ {
// Loading through Vector64 avoids reading beyond the four samples owned by this // Loading through Vector64 avoids reading beyond the four samples owned by this
// batch. Widening is normalized by the runtime for both x86 and Arm64 targets. // batch. Widening is normalized by the runtime for both x86 and Arm64 targets.
@ -827,11 +827,11 @@ internal static partial class Av1SelfGuidedFilter
int roundingBits = SelfGuidedBits + ((row & 1) == 0 ? 5 : 4) - RestorationBits; int roundingBits = SelfGuidedBits + ((row & 1) == 0 ? 5 : 4) - RestorationBits;
Vector256<int> rounding = Vector256.Create(1 << (roundingBits - 1)); Vector256<int> rounding = Vector256.Create(1 << (roundingBits - 1));
int column = 0; int column = 0;
int vectorEnd = width - Vector256<int>.Count; int vectorEnd = (int)(Numerics.Vector256Count<int>(width) * (nuint)Vector256<int>.Count);
// Filtered signals use Q4 precision. Projection applies the signaled Q7 weights to their difference from // Filtered signals use Q4 precision. Projection applies the signaled Q7 weights to their difference from
// the unfiltered Q4 sample, then performs the combined Q11 rounding shift once before clipping. // the unfiltered Q4 sample, then performs the combined Q11 rounding shift once before clipping.
for (; column <= vectorEnd; column += Vector256<int>.Count) for (; column < vectorEnd; column += Vector256<int>.Count)
{ {
Vector256<int> factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, row, vector); Vector256<int> factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, row, vector);
Vector256<int> means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, row, vector); Vector256<int> means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, row, vector);
@ -886,11 +886,11 @@ internal static partial class Av1SelfGuidedFilter
int roundingBits = SelfGuidedBits + ((row & 1) == 0 ? 5 : 4) - RestorationBits; int roundingBits = SelfGuidedBits + ((row & 1) == 0 ? 5 : 4) - RestorationBits;
Vector128<int> rounding = Vector128.Create(1 << (roundingBits - 1)); Vector128<int> rounding = Vector128.Create(1 << (roundingBits - 1));
int column = 0; int column = 0;
int vectorEnd = width - Vector128<int>.Count; int vectorEnd = (int)(Numerics.Vector128Count<int>(width) * (nuint)Vector128<int>.Count);
// The 128-bit path uses the same Q4/Q7 projection equation. The four-sample load and store are deliberately // The 128-bit path uses the same Q4/Q7 projection equation. The four-sample load and store are deliberately
// 64 bits wide so a tightly strided destination row never requires writable padding. // 64 bits wide so a tightly strided destination row never requires writable padding.
for (; column <= vectorEnd; column += Vector128<int>.Count) for (; column < vectorEnd; column += Vector128<int>.Count)
{ {
Vector128<int> factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, row, vector); Vector128<int> factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, row, vector);
Vector128<int> means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, row, vector); Vector128<int> means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, row, vector);
@ -946,8 +946,8 @@ internal static partial class Av1SelfGuidedFilter
int filteredRowOffset = row * width; int filteredRowOffset = row * width;
int coefficientRowOffset = bufferOrigin + (row * bufferStride); int coefficientRowOffset = bufferOrigin + (row * bufferStride);
int column = 0; int column = 0;
int vectorEnd = width - Vector256<int>.Count; nuint vectorCount = Numerics.Vector256Count<int>(width - column);
for (; column <= vectorEnd; column += Vector256<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<int>.Count)
{ {
Vector256<int> factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, vector); Vector256<int> factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, vector);
Vector256<int> means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, vector); Vector256<int> means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, vector);
@ -1002,8 +1002,8 @@ internal static partial class Av1SelfGuidedFilter
int filteredRowOffset = row * width; int filteredRowOffset = row * width;
int coefficientRowOffset = bufferOrigin + (row * bufferStride); int coefficientRowOffset = bufferOrigin + (row * bufferStride);
int column = 0; int column = 0;
int vectorEnd = width - Vector128<int>.Count; nuint vectorCount = Numerics.Vector128Count<int>(width - column);
for (; column <= vectorEnd; column += Vector128<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<int>.Count)
{ {
Vector128<int> factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, vector); Vector128<int> factors = CrossSum(blendFactors, coefficientRowOffset + column, bufferStride, vector);
Vector128<int> means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, vector); Vector128<int> means = CrossSum(localMeans, coefficientRowOffset + column, bufferStride, vector);
@ -1255,8 +1255,8 @@ internal static partial class Av1SelfGuidedFilter
int destinationRowOffset = row * destinationStride; int destinationRowOffset = row * destinationStride;
int filteredRowOffset = row * width; int filteredRowOffset = row * width;
int column = 0; int column = 0;
int vectorEnd = width - Vector256<int>.Count; nuint vectorCount = Numerics.Vector256Count<int>(width - column);
for (; column <= vectorEnd; column += Vector256<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<int>.Count)
{ {
Vector128<ushort> packed = Vector128.LoadUnsafe(ref sourceBase, (nuint)(sourceRowOffset + column)); Vector128<ushort> packed = Vector128.LoadUnsafe(ref sourceBase, (nuint)(sourceRowOffset + column));
Vector256<int> samples = Vector256.WidenLower(Vector256.Create(packed, Vector128<ushort>.Zero)).AsInt32(); Vector256<int> samples = Vector256.WidenLower(Vector256.Create(packed, Vector128<ushort>.Zero)).AsInt32();
@ -1348,8 +1348,8 @@ internal static partial class Av1SelfGuidedFilter
int destinationRowOffset = row * destinationStride; int destinationRowOffset = row * destinationStride;
int filteredRowOffset = row * width; int filteredRowOffset = row * width;
int column = 0; int column = 0;
int vectorEnd = width - Vector128<int>.Count; nuint vectorCount = Numerics.Vector128Count<int>(width - column);
for (; column <= vectorEnd; column += Vector128<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<int>.Count)
{ {
ref ushort sourceReference = ref Unsafe.Add(ref sourceBase, sourceRowOffset + column); ref ushort sourceReference = ref Unsafe.Add(ref sourceBase, sourceRowOffset + column);
Vector64<ushort> packed = Unsafe.As<ushort, Vector64<ushort>>(ref sourceReference); Vector64<ushort> packed = Unsafe.As<ushort, Vector64<ushort>>(ref sourceReference);

205
src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.Operator.cs

@ -0,0 +1,205 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using System.Runtime.Intrinsics;
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers;
/// <content>
/// Defines the arithmetic operators used by <see cref="Av1ForwardQuantizer"/>.
/// </content>
internal static partial class Av1ForwardQuantizer
{
/// <summary>
/// Defines one AV1 forward-quantization arithmetic contract across hardware widths.
/// </summary>
internal interface IForwardQuantizationOperator
{
/// <summary>
/// Quantizes four raster-order transform coefficients.
/// </summary>
/// <param name="coefficients">The signed transform coefficients.</param>
/// <param name="rounding">The positive rounding constant after transform-size scaling.</param>
/// <param name="quantizer">The Q16 reciprocal quantizer.</param>
/// <param name="dequantizer">The Q3 reconstruction quantizer.</param>
/// <param name="logScale">The transform-size quantization scale.</param>
/// <param name="dequantizedCoefficients">The signed reconstruction coefficients.</param>
/// <returns>The signed entropy-coding coefficients.</returns>
public static abstract Vector128<int> Quantize(
Vector128<int> coefficients,
Vector128<int> rounding,
Vector128<int> quantizer,
Vector128<int> dequantizer,
int logScale,
out Vector128<int> dequantizedCoefficients);
/// <summary>
/// Quantizes eight raster-order transform coefficients.
/// </summary>
/// <param name="coefficients">The signed transform coefficients.</param>
/// <param name="rounding">The positive rounding constant after transform-size scaling.</param>
/// <param name="quantizer">The Q16 reciprocal quantizer.</param>
/// <param name="dequantizer">The Q3 reconstruction quantizer.</param>
/// <param name="logScale">The transform-size quantization scale.</param>
/// <param name="dequantizedCoefficients">The signed reconstruction coefficients.</param>
/// <returns>The signed entropy-coding coefficients.</returns>
public static abstract Vector256<int> Quantize(
Vector256<int> coefficients,
Vector256<int> rounding,
Vector256<int> quantizer,
Vector256<int> dequantizer,
int logScale,
out Vector256<int> dequantizedCoefficients);
/// <summary>
/// Quantizes sixteen raster-order transform coefficients.
/// </summary>
/// <param name="coefficients">The signed transform coefficients.</param>
/// <param name="rounding">The positive rounding constant after transform-size scaling.</param>
/// <param name="quantizer">The Q16 reciprocal quantizer.</param>
/// <param name="dequantizer">The Q3 reconstruction quantizer.</param>
/// <param name="logScale">The transform-size quantization scale.</param>
/// <param name="dequantizedCoefficients">The signed reconstruction coefficients.</param>
/// <returns>The signed entropy-coding coefficients.</returns>
public static abstract Vector512<int> Quantize(
Vector512<int> coefficients,
Vector512<int> rounding,
Vector512<int> quantizer,
Vector512<int> dequantizer,
int logScale,
out Vector512<int> dequantizedCoefficients);
/// <summary>
/// Quantizes one transform coefficient.
/// </summary>
/// <param name="coefficient">The signed transform coefficient.</param>
/// <param name="rounding">The positive rounding constant after transform-size scaling.</param>
/// <param name="quantizer">The Q16 reciprocal quantizer.</param>
/// <param name="dequantizer">The Q3 reconstruction quantizer.</param>
/// <param name="logScale">The transform-size quantization scale.</param>
/// <param name="dequantizedCoefficient">The signed reconstruction coefficient.</param>
/// <returns>The signed entropy-coding coefficient.</returns>
public static abstract int Quantize(
int coefficient,
int rounding,
int quantizer,
int dequantizer,
int logScale,
out int dequantizedCoefficient);
}
/// <summary>
/// Implements libaom's fast no-matrix quantizer for lossy transform blocks.
/// </summary>
/// <remarks>
/// Every SIMD overload preserves the scalar operation order: magnitude threshold, saturating round, Q16 reciprocal
/// multiply, transform-size shift, dequantization, and sign restoration. Each lane owns one raster-order coefficient.
/// </remarks>
internal readonly struct FastQuantizationOperator : IForwardQuantizationOperator
{
/// <inheritdoc/>
public static Vector128<int> Quantize(
Vector128<int> coefficients,
Vector128<int> rounding,
Vector128<int> quantizer,
Vector128<int> dequantizer,
int logScale,
out Vector128<int> dequantizedCoefficients)
{
Vector128<int> coefficientSign = Vector128.ShiftRightArithmetic(coefficients, 31);
Vector128<int> magnitude = Vector128.Abs(coefficients);
// libaom retains equality at the dequantizer threshold. Reversing the comparison and complementing its mask
// expresses scaledMagnitude >= dequantizer with the vector operations available for every supported ISA.
Vector128<int> thresholdMask = ~Vector128.GreaterThan(dequantizer, magnitude << (1 + logScale));
// Clamp the rounded magnitude to 32,767 so the following Q16 product remains within a signed 32-bit lane.
Vector128<int> rounded = Vector128.Min(magnitude + rounding, Vector128.Create((int)short.MaxValue));
Vector128<int> quantizedMagnitude = ((rounded * quantizer) >> (16 - logScale)) & thresholdMask;
// XOR followed by subtraction restores the original sign without a lane-wise branch.
Vector128<int> quantized = (quantizedMagnitude ^ coefficientSign) - coefficientSign;
Vector128<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale;
dequantizedCoefficients = (dequantizedMagnitude ^ coefficientSign) - coefficientSign;
return quantized;
}
/// <inheritdoc/>
public static Vector256<int> Quantize(
Vector256<int> coefficients,
Vector256<int> rounding,
Vector256<int> quantizer,
Vector256<int> dequantizer,
int logScale,
out Vector256<int> dequantizedCoefficients)
{
Vector256<int> coefficientSign = Vector256.ShiftRightArithmetic(coefficients, 31);
Vector256<int> magnitude = Vector256.Abs(coefficients);
// Preserve threshold equality by complementing dequantizer > scaledMagnitude.
Vector256<int> thresholdMask = ~Vector256.GreaterThan(dequantizer, magnitude << (1 + logScale));
// Clamp the rounded magnitude to 32,767 so the following Q16 product remains within a signed 32-bit lane.
Vector256<int> rounded = Vector256.Min(magnitude + rounding, Vector256.Create((int)short.MaxValue));
Vector256<int> quantizedMagnitude = ((rounded * quantizer) >> (16 - logScale)) & thresholdMask;
// Apply the input sign to both coded and reconstructed magnitudes without branching.
Vector256<int> quantized = (quantizedMagnitude ^ coefficientSign) - coefficientSign;
Vector256<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale;
dequantizedCoefficients = (dequantizedMagnitude ^ coefficientSign) - coefficientSign;
return quantized;
}
/// <inheritdoc/>
public static Vector512<int> Quantize(
Vector512<int> coefficients,
Vector512<int> rounding,
Vector512<int> quantizer,
Vector512<int> dequantizer,
int logScale,
out Vector512<int> dequantizedCoefficients)
{
Vector512<int> coefficientSign = Vector512.ShiftRightArithmetic(coefficients, 31);
Vector512<int> magnitude = Vector512.Abs(coefficients);
// Preserve threshold equality by complementing dequantizer > scaledMagnitude.
Vector512<int> thresholdMask = ~Vector512.GreaterThan(dequantizer, magnitude << (1 + logScale));
// Clamp the rounded magnitude to 32,767 so the following Q16 product remains within a signed 32-bit lane.
Vector512<int> rounded = Vector512.Min(magnitude + rounding, Vector512.Create((int)short.MaxValue));
Vector512<int> quantizedMagnitude = ((rounded * quantizer) >> (16 - logScale)) & thresholdMask;
// Apply the input sign to both coded and reconstructed magnitudes without branching.
Vector512<int> quantized = (quantizedMagnitude ^ coefficientSign) - coefficientSign;
Vector512<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale;
dequantizedCoefficients = (dequantizedMagnitude ^ coefficientSign) - coefficientSign;
return quantized;
}
/// <inheritdoc/>
public static int Quantize(
int coefficient,
int rounding,
int quantizer,
int dequantizer,
int logScale,
out int dequantizedCoefficient)
{
int coefficientSign = coefficient >> 31;
int magnitude = (coefficient ^ coefficientSign) - coefficientSign;
int quantizedMagnitude = 0;
// The scalar tail keeps the same threshold, clamp, and fixed-point operation order as every vector lane.
if (((long)magnitude << (1 + logScale)) >= dequantizer)
{
int rounded = Math.Min(magnitude + rounding, short.MaxValue);
quantizedMagnitude = (rounded * quantizer) >> (16 - logScale);
}
int quantized = (quantizedMagnitude ^ coefficientSign) - coefficientSign;
int dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale;
dequantizedCoefficient = (dequantizedMagnitude ^ coefficientSign) - coefficientSign;
return quantized;
}
}
}

209
src/ImageSharp/Formats/Heif/Av1/Pipeline/Quantizers/Av1ForwardQuantizer.cs

@ -0,0 +1,209 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using System.Runtime.CompilerServices;
using System.Runtime.InteropServices;
using System.Runtime.Intrinsics;
using SixLabors.ImageSharp.Formats.Heif.Av1.Transform;
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers;
/// <summary>
/// Applies AV1 forward quantization to raster-order transform coefficients.
/// </summary>
internal static partial class Av1ForwardQuantizer
{
/// <summary>
/// Quantizes one lossy transform block with libaom's fast no-matrix arithmetic.
/// </summary>
/// <param name="coefficients">The raster-order forward-transform coefficients.</param>
/// <param name="quantizedCoefficients">The raster-order entropy-coding coefficients.</param>
/// <param name="dequantizedCoefficients">The raster-order reconstruction coefficients.</param>
/// <param name="transformSize">The transform-block dimensions.</param>
/// <param name="transformType">The compound transform type.</param>
/// <param name="qIndex">The segment quantizer index.</param>
/// <param name="dcDeltaQ">The plane DC quantizer adjustment.</param>
/// <param name="acDeltaQ">The plane AC quantizer adjustment.</param>
/// <param name="bitDepth">The coded sample bit depth.</param>
/// <returns>The one-based end position in coefficient scan order.</returns>
public static ushort QuantizeLossy(
ReadOnlySpan<int> coefficients,
Span<int> quantizedCoefficients,
Span<int> dequantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
int qIndex,
int dcDeltaQ,
int acDeltaQ,
Av1BitDepth bitDepth)
=> QuantizeLossy<FastQuantizationOperator>(
coefficients,
quantizedCoefficients,
dequantizedCoefficients,
transformSize,
transformType,
qIndex,
dcDeltaQ,
acDeltaQ,
bitDepth);
/// <summary>
/// Applies a closed generic quantization operator across the widest available hardware widths.
/// </summary>
internal static ushort QuantizeLossy<TOperator>(
ReadOnlySpan<int> coefficients,
Span<int> quantizedCoefficients,
Span<int> dequantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
int qIndex,
int dcDeltaQ,
int acDeltaQ,
Av1BitDepth bitDepth)
where TOperator : struct, IForwardQuantizationOperator
{
int coefficientCount = transformSize.GetAdjusted().GetSize2d();
int logScale = transformSize.GetScale();
int dcDequantizer = Av1QuantizationLookup.GetDcQuant(qIndex, dcDeltaQ, bitDepth);
int acDequantizer = Av1QuantizationLookup.GetAcQuant(qIndex, acDeltaQ, bitDepth);
int dcQuantizer = (1 << 16) / dcDequantizer;
int acQuantizer = (1 << 16) / acDequantizer;
int dcRounding = RoundPowerOfTwo((64 * dcDequantizer) >> 7, logScale);
int acRounding = RoundPowerOfTwo((64 * acDequantizer) >> 7, logScale);
ref int sourceBase = ref MemoryMarshal.GetReference(coefficients);
ref int quantizedBase = ref MemoryMarshal.GetReference(quantizedCoefficients);
ref int dequantizedBase = ref MemoryMarshal.GetReference(dequantizedCoefficients);
// Raster coefficient zero is the only DC coefficient, so it is encoded once with the plane's DC constants
// before the AC-only SIMD traversal begins.
Unsafe.Add(ref quantizedBase, 0) = TOperator.Quantize(
Unsafe.Add(ref sourceBase, 0),
dcRounding,
dcQuantizer,
dcDequantizer,
logScale,
out Unsafe.Add(ref dequantizedBase, 0));
int i = 1;
// Raster traversal keeps loads and stores contiguous. Each narrower tier resumes at the shared offset left by
// the previous tier, retaining vector execution for the widest possible remainder without overlapping lanes.
if (Vector512.IsHardwareAccelerated)
{
nuint vectorCount = coefficients[i..coefficientCount].Vector512Count<int>();
if (vectorCount > 0)
{
// Width-specific constants are created only when at least one complete vector remains.
Vector512<int> rounding = Vector512.Create(acRounding);
Vector512<int> quantizer = Vector512.Create(acQuantizer);
Vector512<int> dequantizer = Vector512.Create(acDequantizer);
for (; vectorCount > 0; vectorCount--, i += Vector512<int>.Count)
{
Vector512<int> source = Unsafe.As<int, Vector512<int>>(ref Unsafe.Add(ref sourceBase, i));
Vector512<int> quantized = TOperator.Quantize(
source,
rounding,
quantizer,
dequantizer,
logScale,
out Vector512<int> dequantized);
Unsafe.As<int, Vector512<int>>(ref Unsafe.Add(ref quantizedBase, i)) = quantized;
Unsafe.As<int, Vector512<int>>(ref Unsafe.Add(ref dequantizedBase, i)) = dequantized;
}
}
}
if (Vector256.IsHardwareAccelerated)
{
nuint vectorCount = coefficients[i..coefficientCount].Vector256Count<int>();
if (vectorCount > 0)
{
// The shared offset exposes only the remainder left by wider lanes, so no coefficient is revisited.
Vector256<int> rounding = Vector256.Create(acRounding);
Vector256<int> quantizer = Vector256.Create(acQuantizer);
Vector256<int> dequantizer = Vector256.Create(acDequantizer);
for (; vectorCount > 0; vectorCount--, i += Vector256<int>.Count)
{
Vector256<int> source = Unsafe.As<int, Vector256<int>>(ref Unsafe.Add(ref sourceBase, i));
Vector256<int> quantized = TOperator.Quantize(
source,
rounding,
quantizer,
dequantizer,
logScale,
out Vector256<int> dequantized);
Unsafe.As<int, Vector256<int>>(ref Unsafe.Add(ref quantizedBase, i)) = quantized;
Unsafe.As<int, Vector256<int>>(ref Unsafe.Add(ref dequantizedBase, i)) = dequantized;
}
}
}
if (Vector128.IsHardwareAccelerated)
{
nuint vectorCount = coefficients[i..coefficientCount].Vector128Count<int>();
if (vectorCount > 0)
{
// The final SIMD tier consumes complete four-lane groups and leaves fewer than four coefficients.
Vector128<int> rounding = Vector128.Create(acRounding);
Vector128<int> quantizer = Vector128.Create(acQuantizer);
Vector128<int> dequantizer = Vector128.Create(acDequantizer);
for (; vectorCount > 0; vectorCount--, i += Vector128<int>.Count)
{
Vector128<int> source = Unsafe.As<int, Vector128<int>>(ref Unsafe.Add(ref sourceBase, i));
Vector128<int> quantized = TOperator.Quantize(
source,
rounding,
quantizer,
dequantizer,
logScale,
out Vector128<int> dequantized);
Unsafe.As<int, Vector128<int>>(ref Unsafe.Add(ref quantizedBase, i)) = quantized;
Unsafe.As<int, Vector128<int>>(ref Unsafe.Add(ref dequantizedBase, i)) = dequantized;
}
}
}
// On SIMD-capable systems this loop receives only the final zero-to-three AC coefficients.
for (; i < coefficientCount; i++)
{
Unsafe.Add(ref quantizedBase, i) = TOperator.Quantize(
Unsafe.Add(ref sourceBase, i),
acRounding,
acQuantizer,
acDequantizer,
logScale,
out Unsafe.Add(ref dequantizedBase, i));
}
ReadOnlySpan<short> scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan;
// Quantized coefficients remain in raster order for reconstruction and entropy coding. A reverse scan finds
// the final nonzero position without another buffer, and normally exits on its first iteration at high quality.
for (int scanIndex = coefficientCount - 1; scanIndex >= 0; scanIndex--)
{
if (Unsafe.Add(ref quantizedBase, scan[scanIndex]) != 0)
{
return (ushort)(scanIndex + 1);
}
}
return 0;
}
/// <summary>
/// Applies libaom's positive round-power-of-two operation to one quantizer constant.
/// </summary>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
private static int RoundPowerOfTwo(int value, int shift)
=> shift == 0 ? value : (value + (1 << (shift - 1))) >> shift;
}

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DcIntraPredictor.Operator.cs

@ -316,8 +316,8 @@ internal static class Av1DcIntraPredictor
// length without a separate dispatch tree and leaves only an incomplete final vector to scalar code. // length without a separate dispatch tree and leaves only an incomplete final vector to scalar code.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = samples.Length - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(samples.Length - index);
for (; index <= oneVectorFromEnd; index += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, index += Vector512<byte>.Count)
{ {
sum += TOperator.Sum(Vector512.LoadUnsafe(ref samplesBase, (nuint)index)); sum += TOperator.Sum(Vector512.LoadUnsafe(ref samplesBase, (nuint)index));
} }
@ -325,8 +325,8 @@ internal static class Av1DcIntraPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = samples.Length - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(samples.Length - index);
for (; index <= oneVectorFromEnd; index += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, index += Vector256<byte>.Count)
{ {
sum += TOperator.Sum(Vector256.LoadUnsafe(ref samplesBase, (nuint)index)); sum += TOperator.Sum(Vector256.LoadUnsafe(ref samplesBase, (nuint)index));
} }
@ -334,8 +334,8 @@ internal static class Av1DcIntraPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = samples.Length - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(samples.Length - index);
for (; index <= oneVectorFromEnd; index += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, index += Vector128<byte>.Count)
{ {
sum += TOperator.Sum(Vector128.LoadUnsafe(ref samplesBase, (nuint)index)); sum += TOperator.Sum(Vector128.LoadUnsafe(ref samplesBase, (nuint)index));
} }
@ -362,8 +362,8 @@ internal static class Av1DcIntraPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = samples.Length - Vector512<short>.Count; nuint vectorCount = Numerics.Vector512Count<short>(samples.Length - index);
for (; index <= oneVectorFromEnd; index += Vector512<short>.Count) for (; vectorCount > 0; vectorCount--, index += Vector512<short>.Count)
{ {
sum += TOperator.Sum(Vector512.LoadUnsafe(ref samplesBase, (nuint)index)); sum += TOperator.Sum(Vector512.LoadUnsafe(ref samplesBase, (nuint)index));
} }
@ -371,8 +371,8 @@ internal static class Av1DcIntraPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = samples.Length - Vector256<short>.Count; nuint vectorCount = Numerics.Vector256Count<short>(samples.Length - index);
for (; index <= oneVectorFromEnd; index += Vector256<short>.Count) for (; vectorCount > 0; vectorCount--, index += Vector256<short>.Count)
{ {
sum += TOperator.Sum(Vector256.LoadUnsafe(ref samplesBase, (nuint)index)); sum += TOperator.Sum(Vector256.LoadUnsafe(ref samplesBase, (nuint)index));
} }
@ -380,8 +380,8 @@ internal static class Av1DcIntraPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = samples.Length - Vector128<short>.Count; nuint vectorCount = Numerics.Vector128Count<short>(samples.Length - index);
for (; index <= oneVectorFromEnd; index += Vector128<short>.Count) for (; vectorCount > 0; vectorCount--, index += Vector128<short>.Count)
{ {
sum += TOperator.Sum(Vector128.LoadUnsafe(ref samplesBase, (nuint)index)); sum += TOperator.Sum(Vector128.LoadUnsafe(ref samplesBase, (nuint)index));
} }

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Av1DirectionalIntraPredictor.Operations.cs

@ -204,8 +204,8 @@ internal static partial class Av1DirectionalIntraPredictor
// left by the wider path, so the row is written once without requiring padded destination storage. // left by the wider path, so the row is written once without requiring padded destination storage.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = validCount - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(validCount - index);
for (; index <= oneVectorFromEnd; index += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, index += Vector512<byte>.Count)
{ {
Vector512<byte> left = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector512<byte> left = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index));
Vector512<byte> right = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); Vector512<byte> right = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1));
@ -215,8 +215,8 @@ internal static partial class Av1DirectionalIntraPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = validCount - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(validCount - index);
for (; index <= oneVectorFromEnd; index += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, index += Vector256<byte>.Count)
{ {
Vector256<byte> left = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector256<byte> left = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index));
Vector256<byte> right = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); Vector256<byte> right = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1));
@ -226,8 +226,8 @@ internal static partial class Av1DirectionalIntraPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = validCount - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(validCount - index);
for (; index <= oneVectorFromEnd; index += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, index += Vector128<byte>.Count)
{ {
Vector128<byte> left = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector128<byte> left = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index));
Vector128<byte> right = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); Vector128<byte> right = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1));
@ -312,8 +312,8 @@ internal static partial class Av1DirectionalIntraPredictor
// weighted sum. The largest supported 12-bit sample therefore cannot overflow an intermediate lane. // weighted sum. The largest supported 12-bit sample therefore cannot overflow an intermediate lane.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = validCount - Vector512<short>.Count; nuint vectorCount = Numerics.Vector512Count<short>(validCount - index);
for (; index <= oneVectorFromEnd; index += Vector512<short>.Count) for (; vectorCount > 0; vectorCount--, index += Vector512<short>.Count)
{ {
Vector512<short> left = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector512<short> left = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index));
Vector512<short> right = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); Vector512<short> right = Vector512.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1));
@ -323,8 +323,8 @@ internal static partial class Av1DirectionalIntraPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = validCount - Vector256<short>.Count; nuint vectorCount = Numerics.Vector256Count<short>(validCount - index);
for (; index <= oneVectorFromEnd; index += Vector256<short>.Count) for (; vectorCount > 0; vectorCount--, index += Vector256<short>.Count)
{ {
Vector256<short> left = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector256<short> left = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index));
Vector256<short> right = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); Vector256<short> right = Vector256.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1));
@ -334,8 +334,8 @@ internal static partial class Av1DirectionalIntraPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = validCount - Vector128<short>.Count; nuint vectorCount = Numerics.Vector128Count<short>(validCount - index);
for (; index <= oneVectorFromEnd; index += Vector128<short>.Count) for (; vectorCount > 0; vectorCount--, index += Vector128<short>.Count)
{ {
Vector128<short> left = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index)); Vector128<short> left = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index));
Vector128<short> right = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1)); Vector128<short> right = Vector128.LoadUnsafe(ref referenceBase, (nuint)(basis + index + 1));

12
src/ImageSharp/Formats/Heif/Av1/Prediction/Av1NonDirectionalIntraPredictor.Operator.cs

@ -226,7 +226,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase
// narrower paths consume any complete vectors left before the scalar tail handles the final columns. // narrower paths consume any complete vectors left before the scalar tail handles the final columns.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorizedColumns = width - (width % Vector512<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector512Count<byte>(width) * (nuint)Vector512<byte>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
Vector512<byte> topLeftVector = usesTopLeft ? Vector512.Create(topLeft) : default; Vector512<byte> topLeftVector = usesTopLeft ? Vector512.Create(topLeft) : default;
@ -255,7 +255,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector256<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector256Count<byte>(remainingColumns) * (nuint)Vector256<byte>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
Vector256<byte> topLeftVector = usesTopLeft ? Vector256.Create(topLeft) : default; Vector256<byte> topLeftVector = usesTopLeft ? Vector256.Create(topLeft) : default;
@ -285,7 +285,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector128<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector128Count<byte>(remainingColumns) * (nuint)Vector128<byte>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
Vector128<byte> topLeftVector = usesTopLeft ? Vector128.Create(topLeft) : default; Vector128<byte> topLeftVector = usesTopLeft ? Vector128.Create(topLeft) : default;
@ -355,7 +355,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase
// column, so the same width-progressive traversal is valid without inter-lane packing or saturation. // column, so the same width-progressive traversal is valid without inter-lane packing or saturation.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorizedColumns = width - (width % Vector512<short>.Count); int vectorizedColumns = (int)(Numerics.Vector512Count<short>(width) * (nuint)Vector512<short>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
Vector512<short> topLeftVector = usesTopLeft ? Vector512.Create(topLeft) : default; Vector512<short> topLeftVector = usesTopLeft ? Vector512.Create(topLeft) : default;
@ -384,7 +384,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector256<short>.Count); int vectorizedColumns = (int)(Numerics.Vector256Count<short>(remainingColumns) * (nuint)Vector256<short>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
Vector256<short> topLeftVector = usesTopLeft ? Vector256.Create(topLeft) : default; Vector256<short> topLeftVector = usesTopLeft ? Vector256.Create(topLeft) : default;
@ -414,7 +414,7 @@ internal abstract partial class Av1NonDirectionalIntraPredictorBase
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector128<short>.Count); int vectorizedColumns = (int)(Numerics.Vector128Count<short>(remainingColumns) * (nuint)Vector128<short>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
Vector128<short> topLeftVector = usesTopLeft ? Vector128.Create(topLeft) : default; Vector128<short> topLeftVector = usesTopLeft ? Vector128.Create(topLeft) : default;

76
src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PalettePredictor.Operator.cs

@ -212,33 +212,43 @@ internal static class Av1PalettePredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
Vector256<byte> palette256 = Vector256.Create(palette128, palette128); nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
Vector512<byte> palette512 = Vector512.Create(palette256, palette256);
int oneVectorFromEnd = width - Vector512<byte>.Count;
for (; column <= oneVectorFromEnd; column += Vector512<byte>.Count) if (vectorCount > 0)
{ {
Vector512<byte> indices = Vector512.LoadUnsafe(ref mapRow, (nuint)column); // Replicate the lookup table only when the row has a complete 64-lane batch.
TOperator.Predict(palette512, indices).StoreUnsafe(ref destinationRow, (nuint)column); Vector256<byte> palette256 = Vector256.Create(palette128, palette128);
Vector512<byte> palette512 = Vector512.Create(palette256, palette256);
for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{
Vector512<byte> indices = Vector512.LoadUnsafe(ref mapRow, (nuint)column);
TOperator.Predict(palette512, indices).StoreUnsafe(ref destinationRow, (nuint)column);
}
} }
} }
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
Vector256<byte> palette256 = Vector256.Create(palette128, palette128); nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
int oneVectorFromEnd = width - Vector256<byte>.Count;
for (; column <= oneVectorFromEnd; column += Vector256<byte>.Count) if (vectorCount > 0)
{ {
Vector256<byte> indices = Vector256.LoadUnsafe(ref mapRow, (nuint)column); // The narrower table is likewise materialized only for a complete 32-lane remainder.
TOperator.Predict(palette256, indices).StoreUnsafe(ref destinationRow, (nuint)column); Vector256<byte> palette256 = Vector256.Create(palette128, palette128);
for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{
Vector256<byte> indices = Vector256.LoadUnsafe(ref mapRow, (nuint)column);
TOperator.Predict(palette256, indices).StoreUnsafe(ref destinationRow, (nuint)column);
}
} }
} }
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= oneVectorFromEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<byte> indices = Vector128.LoadUnsafe(ref mapRow, (nuint)column); Vector128<byte> indices = Vector128.LoadUnsafe(ref mapRow, (nuint)column);
TOperator.Predict(palette128, indices).StoreUnsafe(ref destinationRow, (nuint)column); TOperator.Predict(palette128, indices).StoreUnsafe(ref destinationRow, (nuint)column);
@ -296,35 +306,45 @@ internal static class Av1PalettePredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
Vector256<byte> palette256 = Vector256.Create(palette128, palette128); nuint vectorCount = Numerics.Vector512Count<short>(width - column);
Vector512<byte> palette512 = Vector512.Create(palette256, palette256);
int oneVectorFromEnd = width - Vector512<short>.Count;
for (; column <= oneVectorFromEnd; column += Vector512<short>.Count) if (vectorCount > 0)
{ {
(Vector256<ushort> lower, Vector256<ushort> upper) = Vector256.Widen(Vector256.LoadUnsafe(ref mapRow, (nuint)column)); // High-bit-depth output has half as many lanes, so gate table replication with that lane count.
Vector512<ushort> indices = Vector512.Create(lower, upper); Vector256<byte> palette256 = Vector256.Create(palette128, palette128);
TOperator.Predict(palette512, indices).StoreUnsafe(ref destinationRow, (nuint)column); Vector512<byte> palette512 = Vector512.Create(palette256, palette256);
for (; vectorCount > 0; vectorCount--, column += Vector512<short>.Count)
{
(Vector256<ushort> lower, Vector256<ushort> upper) = Vector256.Widen(Vector256.LoadUnsafe(ref mapRow, (nuint)column));
Vector512<ushort> indices = Vector512.Create(lower, upper);
TOperator.Predict(palette512, indices).StoreUnsafe(ref destinationRow, (nuint)column);
}
} }
} }
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
Vector256<byte> palette256 = Vector256.Create(palette128, palette128); nuint vectorCount = Numerics.Vector256Count<short>(width - column);
int oneVectorFromEnd = width - Vector256<short>.Count;
for (; column <= oneVectorFromEnd; column += Vector256<short>.Count) if (vectorCount > 0)
{ {
(Vector128<ushort> lower, Vector128<ushort> upper) = Vector128.Widen(Vector128.LoadUnsafe(ref mapRow, (nuint)column)); // Avoid creating the 256-bit table when the remainder belongs entirely to narrower paths.
Vector256<ushort> indices = Vector256.Create(lower, upper); Vector256<byte> palette256 = Vector256.Create(palette128, palette128);
TOperator.Predict(palette256, indices).StoreUnsafe(ref destinationRow, (nuint)column);
for (; vectorCount > 0; vectorCount--, column += Vector256<short>.Count)
{
(Vector128<ushort> lower, Vector128<ushort> upper) = Vector128.Widen(Vector128.LoadUnsafe(ref mapRow, (nuint)column));
Vector256<ushort> indices = Vector256.Create(lower, upper);
TOperator.Predict(palette256, indices).StoreUnsafe(ref destinationRow, (nuint)column);
}
} }
} }
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector128<short>.Count; nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; column <= oneVectorFromEnd; column += Vector128<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
ulong packedIndices = Unsafe.ReadUnaligned<ulong>(ref Unsafe.Add(ref mapRow, column)); ulong packedIndices = Unsafe.ReadUnaligned<ulong>(ref Unsafe.Add(ref mapRow, column));
Vector128<ushort> indices = Vector128.WidenLower(Vector128.CreateScalarUnsafe(packedIndices).AsByte()); Vector128<ushort> indices = Vector128.WidenLower(Vector128.CreateScalarUnsafe(packedIndices).AsByte());

8
src/ImageSharp/Formats/Heif/Av1/Prediction/Av1PredictionDecoder.cs

@ -1689,11 +1689,11 @@ internal sealed class Av1PredictionDecoder
// Unsigned 16-bit lanes therefore preserve every normative strength without widening to 32-bit vectors. // Unsigned 16-bit lanes therefore preserve every normative strength without widening to 32-bit vectors.
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int eightSamplesFromEnd = outputCount - Vector128<short>.Count; int vectorEnd = (int)(Numerics.Vector128Count<short>(outputCount) * (nuint)Vector128<short>.Count);
switch (strength) switch (strength)
{ {
case 1: case 1:
for (; processed <= eightSamplesFromEnd; processed += Vector128<short>.Count) for (; processed < vectorEnd; processed += Vector128<short>.Count)
{ {
Vector128<ushort> source0 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16(); Vector128<ushort> source0 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16();
Vector128<ushort> source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 2)).AsUInt16(); Vector128<ushort> source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 2)).AsUInt16();
@ -1703,7 +1703,7 @@ internal sealed class Av1PredictionDecoder
break; break;
case 2: case 2:
for (; processed <= eightSamplesFromEnd; processed += Vector128<short>.Count) for (; processed < vectorEnd; processed += Vector128<short>.Count)
{ {
Vector128<ushort> source0 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16(); Vector128<ushort> source0 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16();
Vector128<ushort> source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 2)).AsUInt16(); Vector128<ushort> source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 2)).AsUInt16();
@ -1713,7 +1713,7 @@ internal sealed class Av1PredictionDecoder
break; break;
default: default:
for (; processed <= eightSamplesFromEnd; processed += Vector128<short>.Count) for (; processed < vectorEnd; processed += Vector128<short>.Count)
{ {
Vector128<ushort> source0 = Vector128.LoadUnsafe(ref edge, (nuint)processed).AsUInt16(); Vector128<ushort> source0 = Vector128.LoadUnsafe(ref edge, (nuint)processed).AsUInt16();
Vector128<ushort> source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16(); Vector128<ushort> source1 = Vector128.LoadUnsafe(ref edge, (nuint)(processed + 1)).AsUInt16();

30
src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaContext.Operations.cs

@ -42,7 +42,8 @@ internal partial class Av1ChromaFromLumaContext
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
for (; column <= width - Vector256<byte>.Count; column += Vector256<byte>.Count) nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
(Vector256<ushort> lower, Vector256<ushort> upper) = Vector256.Widen(Vector256.LoadUnsafe(ref inputRow, (nuint)column)); (Vector256<ushort> lower, Vector256<ushort> upper) = Vector256.Widen(Vector256.LoadUnsafe(ref inputRow, (nuint)column));
@ -53,7 +54,8 @@ internal partial class Av1ChromaFromLumaContext
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
for (; column <= width - Vector128<byte>.Count; column += Vector128<byte>.Count) nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
(Vector128<ushort> lower, Vector128<ushort> upper) = Vector128.Widen(Vector128.LoadUnsafe(ref inputRow, (nuint)column)); (Vector128<ushort> lower, Vector128<ushort> upper) = Vector128.Widen(Vector128.LoadUnsafe(ref inputRow, (nuint)column));
@ -103,7 +105,8 @@ internal partial class Av1ChromaFromLumaContext
if (Avx2.IsSupported) if (Avx2.IsSupported)
{ {
for (; column <= width - Vector256<byte>.Count; column += Vector256<byte>.Count) nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<short> sum = Avx2.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref inputRow, (nuint)column), ones256); Vector256<short> sum = Avx2.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref inputRow, (nuint)column), ones256);
if (this.subY) if (this.subY)
@ -117,7 +120,8 @@ internal partial class Av1ChromaFromLumaContext
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
for (; column <= width - Vector128<byte>.Count; column += Vector128<byte>.Count) nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<short> sum = PairSum(Vector128.LoadUnsafe(ref inputRow, (nuint)column), ones128); Vector128<short> sum = PairSum(Vector128.LoadUnsafe(ref inputRow, (nuint)column), ones128);
if (this.subY) if (this.subY)
@ -193,7 +197,8 @@ internal partial class Av1ChromaFromLumaContext
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
for (; column <= width - Vector256<short>.Count; column += Vector256<short>.Count) nuint vectorCount = Numerics.Vector256Count<short>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector256<short>.Count)
{ {
(Vector256.LoadUnsafe(ref inputRow, (nuint)column) << 3).StoreUnsafe(ref outputRow, (nuint)column); (Vector256.LoadUnsafe(ref inputRow, (nuint)column) << 3).StoreUnsafe(ref outputRow, (nuint)column);
} }
@ -201,7 +206,8 @@ internal partial class Av1ChromaFromLumaContext
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
for (; column <= width - Vector128<short>.Count; column += Vector128<short>.Count) nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
(Vector128.LoadUnsafe(ref inputRow, (nuint)column) << 3).StoreUnsafe(ref outputRow, (nuint)column); (Vector128.LoadUnsafe(ref inputRow, (nuint)column) << 3).StoreUnsafe(ref outputRow, (nuint)column);
} }
@ -237,7 +243,8 @@ internal partial class Av1ChromaFromLumaContext
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
for (; column <= width - Vector256<short>.Count; column += Vector256<short>.Count) nuint vectorCount = Numerics.Vector256Count<short>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector256<short>.Count)
{ {
Vector256<int> sum = Vector256_.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref inputRow, (nuint)column), ones256); Vector256<int> sum = Vector256_.MultiplyAddAdjacent(Vector256.LoadUnsafe(ref inputRow, (nuint)column), ones256);
if (this.subY) if (this.subY)
@ -251,7 +258,8 @@ internal partial class Av1ChromaFromLumaContext
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
for (; column <= width - Vector128<short>.Count; column += Vector128<short>.Count) nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
Vector128<int> sum = Vector128_.MultiplyAddAdjacent(Vector128.LoadUnsafe(ref inputRow, (nuint)column), ones128); Vector128<int> sum = Vector128_.MultiplyAddAdjacent(Vector128.LoadUnsafe(ref inputRow, (nuint)column), ones128);
if (this.subY) if (this.subY)
@ -326,7 +334,8 @@ internal partial class Av1ChromaFromLumaContext
{ {
int rowOffset = row * BufferLine; int rowOffset = row * BufferLine;
int column = 0; int column = 0;
for (; column <= width - Vector128<short>.Count; column += Vector128<short>.Count) nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
(Vector128<int> lower, Vector128<int> upper) = Vector128.Widen(Vector128.LoadUnsafe(ref bufferBase, (nuint)(rowOffset + column))); (Vector128<int> lower, Vector128<int> upper) = Vector128.Widen(Vector128.LoadUnsafe(ref bufferBase, (nuint)(rowOffset + column)));
@ -377,7 +386,8 @@ internal partial class Av1ChromaFromLumaContext
{ {
int rowOffset = row * BufferLine; int rowOffset = row * BufferLine;
int column = 0; int column = 0;
for (; column <= width - Vector128<short>.Count; column += Vector128<short>.Count) nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
(Vector128.LoadUnsafe(ref bufferBase, (nuint)(rowOffset + column)) - average).StoreUnsafe(ref bufferBase, (nuint)(rowOffset + column)); (Vector128.LoadUnsafe(ref bufferBase, (nuint)(rowOffset + column)) - average).StoreUnsafe(ref bufferBase, (nuint)(rowOffset + column));
} }

24
src/ImageSharp/Formats/Heif/Av1/Prediction/ChromaFromLuma/Av1ChromaFromLumaPredictor.Operator.cs

@ -216,8 +216,8 @@ internal static partial class Av1ChromaFromLumaPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector512<short>.Count; nuint vectorCount = Numerics.Vector512Count<short>(width - column);
for (; column <= oneVectorFromEnd; column += Vector512<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<short>.Count)
{ {
Vector512<short> prediction = TOperator.Predict(Vector512.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue); Vector512<short> prediction = TOperator.Predict(Vector512.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue);
Vector256<byte> packed = Vector512.Narrow(prediction.AsUInt16(), Vector512<ushort>.Zero).GetLower(); Vector256<byte> packed = Vector512.Narrow(prediction.AsUInt16(), Vector512<ushort>.Zero).GetLower();
@ -227,8 +227,8 @@ internal static partial class Av1ChromaFromLumaPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector256<short>.Count; nuint vectorCount = Numerics.Vector256Count<short>(width - column);
for (; column <= oneVectorFromEnd; column += Vector256<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<short>.Count)
{ {
Vector256<short> prediction = TOperator.Predict(Vector256.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue); Vector256<short> prediction = TOperator.Predict(Vector256.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue);
Vector128<byte> packed = Vector256.Narrow(prediction.AsUInt16(), Vector256<ushort>.Zero).GetLower(); Vector128<byte> packed = Vector256.Narrow(prediction.AsUInt16(), Vector256<ushort>.Zero).GetLower();
@ -238,8 +238,8 @@ internal static partial class Av1ChromaFromLumaPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector128<short>.Count; nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; column <= oneVectorFromEnd; column += Vector128<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
Vector128<short> prediction = TOperator.Predict(Vector128.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue); Vector128<short> prediction = TOperator.Predict(Vector128.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, byte.MaxValue);
Vector64<byte> packed = Vector128.Narrow(prediction.AsUInt16(), Vector128<ushort>.Zero).GetLower(); Vector64<byte> packed = Vector128.Narrow(prediction.AsUInt16(), Vector128<ushort>.Zero).GetLower();
@ -282,8 +282,8 @@ internal static partial class Av1ChromaFromLumaPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector512<short>.Count; nuint vectorCount = Numerics.Vector512Count<short>(width - column);
for (; column <= oneVectorFromEnd; column += Vector512<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<short>.Count)
{ {
TOperator.Predict(Vector512.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column); TOperator.Predict(Vector512.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column);
} }
@ -291,8 +291,8 @@ internal static partial class Av1ChromaFromLumaPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector256<short>.Count; nuint vectorCount = Numerics.Vector256Count<short>(width - column);
for (; column <= oneVectorFromEnd; column += Vector256<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<short>.Count)
{ {
TOperator.Predict(Vector256.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column); TOperator.Predict(Vector256.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column);
} }
@ -300,8 +300,8 @@ internal static partial class Av1ChromaFromLumaPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector128<short>.Count; nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; column <= oneVectorFromEnd; column += Vector128<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
TOperator.Predict(Vector128.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column); TOperator.Predict(Vector128.LoadUnsafe(ref lumaRow, (nuint)column), dc, alphaQ3, maximum).StoreUnsafe(ref destinationRow, (nuint)column);
} }

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundAveragePredictor.cs

@ -60,8 +60,8 @@ internal static partial class Av1CompoundAveragePredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<byte> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512<byte> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column);
Vector512<byte> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<byte> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -71,8 +71,8 @@ internal static partial class Av1CompoundAveragePredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<byte> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256<byte> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column);
Vector256<byte> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<byte> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -82,8 +82,8 @@ internal static partial class Av1CompoundAveragePredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<byte> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128<byte> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column);
Vector128<byte> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<byte> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);
@ -145,8 +145,8 @@ internal static partial class Av1CompoundAveragePredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<ushort>.Count; nuint vectorCount = Numerics.Vector512Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<ushort>.Count)
{ {
Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column);
Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -156,8 +156,8 @@ internal static partial class Av1CompoundAveragePredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<ushort>.Count; nuint vectorCount = Numerics.Vector256Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<ushort>.Count)
{ {
Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column);
Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -167,8 +167,8 @@ internal static partial class Av1CompoundAveragePredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<ushort>.Count)
{ {
Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column);
Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundDistanceWeightedPredictor.cs

@ -62,8 +62,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<byte> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512<byte> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column);
Vector512<byte> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<byte> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -73,8 +73,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<byte> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256<byte> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column);
Vector256<byte> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<byte> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -84,8 +84,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<byte> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128<byte> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column);
Vector128<byte> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<byte> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);
@ -147,8 +147,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<ushort>.Count; nuint vectorCount = Numerics.Vector512Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<ushort>.Count)
{ {
Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column);
Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -158,8 +158,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<ushort>.Count; nuint vectorCount = Numerics.Vector256Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<ushort>.Count)
{ {
Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column);
Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -169,8 +169,8 @@ internal static partial class Av1CompoundDistanceWeightedPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<ushort>.Count)
{ {
Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column);
Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);

96
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundInterPredictor.cs

@ -418,8 +418,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector512.IsHardwareAccelerated) if (useSimd && Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<byte> samples = Vector512.LoadUnsafe(ref sourceRow, (nuint)column); Vector512<byte> samples = Vector512.LoadUnsafe(ref sourceRow, (nuint)column);
TOperator.Copy(samples, roundBits, roundOffset, out Vector512<ushort> lower, out Vector512<ushort> upper); TOperator.Copy(samples, roundBits, roundOffset, out Vector512<ushort> lower, out Vector512<ushort> upper);
@ -430,8 +430,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector256.IsHardwareAccelerated) if (useSimd && Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<byte> samples = Vector256.LoadUnsafe(ref sourceRow, (nuint)column); Vector256<byte> samples = Vector256.LoadUnsafe(ref sourceRow, (nuint)column);
TOperator.Copy(samples, roundBits, roundOffset, out Vector256<ushort> lower, out Vector256<ushort> upper); TOperator.Copy(samples, roundBits, roundOffset, out Vector256<ushort> lower, out Vector256<ushort> upper);
@ -442,8 +442,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector128.IsHardwareAccelerated) if (useSimd && Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<byte> samples = Vector128.LoadUnsafe(ref sourceRow, (nuint)column); Vector128<byte> samples = Vector128.LoadUnsafe(ref sourceRow, (nuint)column);
TOperator.Copy(samples, roundBits, roundOffset, out Vector128<ushort> lower, out Vector128<ushort> upper); TOperator.Copy(samples, roundBits, roundOffset, out Vector128<ushort> lower, out Vector128<ushort> upper);
@ -487,8 +487,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector512.IsHardwareAccelerated) if (useSimd && Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<ushort>.Count; nuint vectorCount = Numerics.Vector512Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<ushort>.Count)
{ {
Vector512<ushort> samples = Vector512.LoadUnsafe(ref sourceRow, (nuint)column); Vector512<ushort> samples = Vector512.LoadUnsafe(ref sourceRow, (nuint)column);
TOperator.CopyHighBitDepth(samples, roundBits, roundOffset) TOperator.CopyHighBitDepth(samples, roundBits, roundOffset)
@ -498,8 +498,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector256.IsHardwareAccelerated) if (useSimd && Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<ushort>.Count; nuint vectorCount = Numerics.Vector256Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<ushort>.Count)
{ {
Vector256<ushort> samples = Vector256.LoadUnsafe(ref sourceRow, (nuint)column); Vector256<ushort> samples = Vector256.LoadUnsafe(ref sourceRow, (nuint)column);
TOperator.CopyHighBitDepth(samples, roundBits, roundOffset) TOperator.CopyHighBitDepth(samples, roundBits, roundOffset)
@ -509,8 +509,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector128.IsHardwareAccelerated) if (useSimd && Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<ushort>.Count)
{ {
Vector128<ushort> samples = Vector128.LoadUnsafe(ref sourceRow, (nuint)column); Vector128<ushort> samples = Vector128.LoadUnsafe(ref sourceRow, (nuint)column);
TOperator.CopyHighBitDepth(samples, roundBits, roundOffset) TOperator.CopyHighBitDepth(samples, roundBits, roundOffset)
@ -562,8 +562,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector512.IsHardwareAccelerated) if (useSimd && Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -587,8 +587,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector256.IsHardwareAccelerated) if (useSimd && Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -612,8 +612,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector128.IsHardwareAccelerated) if (useSimd && Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -683,8 +683,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector512.IsHardwareAccelerated) if (useSimd && Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<ushort>.Count; nuint vectorCount = Numerics.Vector512Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -703,8 +703,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector256.IsHardwareAccelerated) if (useSimd && Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<ushort>.Count; nuint vectorCount = Numerics.Vector256Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -723,8 +723,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector128.IsHardwareAccelerated) if (useSimd && Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -797,8 +797,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector512.IsHardwareAccelerated) if (useSimd && Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -820,8 +820,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector256.IsHardwareAccelerated) if (useSimd && Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -843,8 +843,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector128.IsHardwareAccelerated) if (useSimd && Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -884,8 +884,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector512.IsHardwareAccelerated) if (useSimd && Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<short>.Count; nuint vectorCount = Numerics.Vector512Count<short>(width - column);
for (; column <= vectorEnd; column += Vector512<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<short>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -903,8 +903,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector256.IsHardwareAccelerated) if (useSimd && Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<short>.Count; nuint vectorCount = Numerics.Vector256Count<short>(width - column);
for (; column <= vectorEnd; column += Vector256<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<short>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -922,8 +922,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector128.IsHardwareAccelerated) if (useSimd && Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<short>.Count; nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; column <= vectorEnd; column += Vector128<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -1000,8 +1000,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector512.IsHardwareAccelerated) if (useSimd && Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<ushort>.Count; nuint vectorCount = Numerics.Vector512Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -1020,8 +1020,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector256.IsHardwareAccelerated) if (useSimd && Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<ushort>.Count; nuint vectorCount = Numerics.Vector256Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -1040,8 +1040,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector128.IsHardwareAccelerated) if (useSimd && Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -1079,8 +1079,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector512.IsHardwareAccelerated) if (useSimd && Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<short>.Count; nuint vectorCount = Numerics.Vector512Count<short>(width - column);
for (; column <= vectorEnd; column += Vector512<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<short>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -1099,8 +1099,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector256.IsHardwareAccelerated) if (useSimd && Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<short>.Count; nuint vectorCount = Numerics.Vector256Count<short>(width - column);
for (; column <= vectorEnd; column += Vector256<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<short>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -1119,8 +1119,8 @@ internal static partial class Av1CompoundInterPredictor
if (useSimd && Vector128.IsHardwareAccelerated) if (useSimd && Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<short>.Count; nuint vectorCount = Numerics.Vector128Count<short>(width - column);
for (; column <= vectorEnd; column += Vector128<short>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<short>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateAveragePredictor.cs

@ -68,8 +68,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count)); Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count));
@ -82,8 +82,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count)); Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count));
@ -96,8 +96,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<ushort> first1 = Vector128.LoadUnsafe( Vector128<ushort> first1 = Vector128.LoadUnsafe(
@ -177,8 +177,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<ushort>.Count; nuint vectorCount = Numerics.Vector512Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<ushort>.Count)
{ {
Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -189,8 +189,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<ushort>.Count; nuint vectorCount = Numerics.Vector256Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<ushort>.Count)
{ {
Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -201,8 +201,8 @@ internal static partial class Av1CompoundIntermediateAveragePredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<ushort>.Count)
{ {
Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);

12
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDifferenceWeightedMaskBuilder.cs

@ -74,8 +74,8 @@ internal static partial class Av1CompoundIntermediateDifferenceWeightedMaskBuild
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count)); Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count));
@ -88,8 +88,8 @@ internal static partial class Av1CompoundIntermediateDifferenceWeightedMaskBuild
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count)); Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count));
@ -102,8 +102,8 @@ internal static partial class Av1CompoundIntermediateDifferenceWeightedMaskBuild
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<ushort> first1 = Vector128.LoadUnsafe( Vector128<ushort> first1 = Vector128.LoadUnsafe(

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateDistanceWeightedPredictor.cs

@ -74,8 +74,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count)); Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count));
@ -95,8 +95,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count)); Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count));
@ -116,8 +116,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<ushort> first1 = Vector128.LoadUnsafe( Vector128<ushort> first1 = Vector128.LoadUnsafe(
@ -215,8 +215,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<ushort>.Count; nuint vectorCount = Numerics.Vector512Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<ushort>.Count)
{ {
Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -233,8 +233,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<ushort>.Count; nuint vectorCount = Numerics.Vector256Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<ushort>.Count)
{ {
Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -251,8 +251,8 @@ internal static partial class Av1CompoundIntermediateDistanceWeightedPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<ushort>.Count)
{ {
Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundIntermediateMaskBlendPredictor.cs

@ -82,8 +82,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor
{ {
ref byte maskReference = ref MemoryMarshal.GetReference(mask); ref byte maskReference = ref MemoryMarshal.GetReference(mask);
int maskRowOffset = row * maskStride; int maskRowOffset = row * maskStride;
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count)); Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count));
@ -99,8 +99,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor
{ {
ref byte maskReference = ref MemoryMarshal.GetReference(mask); ref byte maskReference = ref MemoryMarshal.GetReference(mask);
int maskRowOffset = row * maskStride; int maskRowOffset = row * maskStride;
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count)); Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count));
@ -116,8 +116,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor
{ {
ref byte maskReference = ref MemoryMarshal.GetReference(mask); ref byte maskReference = ref MemoryMarshal.GetReference(mask);
int maskRowOffset = row * maskStride; int maskRowOffset = row * maskStride;
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<ushort> first1 = Vector128.LoadUnsafe( Vector128<ushort> first1 = Vector128.LoadUnsafe(
@ -221,8 +221,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor
{ {
ref byte maskReference = ref MemoryMarshal.GetReference(mask); ref byte maskReference = ref MemoryMarshal.GetReference(mask);
int maskRowOffset = row * maskStride; int maskRowOffset = row * maskStride;
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<ushort> first1 = Vector512.LoadUnsafe( Vector512<ushort> first1 = Vector512.LoadUnsafe(
@ -261,8 +261,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor
{ {
ref byte maskReference = ref MemoryMarshal.GetReference(mask); ref byte maskReference = ref MemoryMarshal.GetReference(mask);
int maskRowOffset = row * maskStride; int maskRowOffset = row * maskStride;
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<ushort> first1 = Vector256.LoadUnsafe( Vector256<ushort> first1 = Vector256.LoadUnsafe(
@ -301,8 +301,8 @@ internal static partial class Av1CompoundIntermediateMaskBlendPredictor
{ {
ref byte maskReference = ref MemoryMarshal.GetReference(mask); ref byte maskReference = ref MemoryMarshal.GetReference(mask);
int maskRowOffset = row * maskStride; int maskRowOffset = row * maskStride;
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<ushort> first1 = Vector128.LoadUnsafe( Vector128<ushort> first1 = Vector128.LoadUnsafe(

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1CompoundMaskBlendPredictor.cs

@ -64,8 +64,8 @@ internal static partial class Av1CompoundMaskBlendPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<byte> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512<byte> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column);
Vector512<byte> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<byte> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -76,8 +76,8 @@ internal static partial class Av1CompoundMaskBlendPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<byte> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256<byte> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column);
Vector256<byte> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<byte> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -88,8 +88,8 @@ internal static partial class Av1CompoundMaskBlendPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<byte> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128<byte> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column);
Vector128<byte> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<byte> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);
@ -154,8 +154,8 @@ internal static partial class Av1CompoundMaskBlendPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<ushort>.Count; nuint vectorCount = Numerics.Vector512Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<ushort>.Count)
{ {
Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column); Vector512<ushort> firstVector = Vector512.LoadUnsafe(ref destinationReference, (nuint)column);
Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<ushort> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -166,8 +166,8 @@ internal static partial class Av1CompoundMaskBlendPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<ushort>.Count; nuint vectorCount = Numerics.Vector256Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<ushort>.Count)
{ {
Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column); Vector256<ushort> firstVector = Vector256.LoadUnsafe(ref destinationReference, (nuint)column);
Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<ushort> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -178,8 +178,8 @@ internal static partial class Av1CompoundMaskBlendPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - column);
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<ushort>.Count)
{ {
Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column); Vector128<ushort> firstVector = Vector128.LoadUnsafe(ref destinationReference, (nuint)column);
Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<ushort> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1DifferenceWeightedMaskBuilder.cs

@ -68,8 +68,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<byte> firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<byte> firstVector = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<byte> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column); Vector512<byte> secondVector = Vector512.LoadUnsafe(ref secondReference, (nuint)column);
@ -79,8 +79,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<byte> firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<byte> firstVector = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<byte> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column); Vector256<byte> secondVector = Vector256.LoadUnsafe(ref secondReference, (nuint)column);
@ -90,8 +90,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<byte> firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<byte> firstVector = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<byte> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column); Vector128<byte> secondVector = Vector128.LoadUnsafe(ref secondReference, (nuint)column);
@ -165,8 +165,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder
// temporary buffers before the following vector blend consumes the complete plane block. // temporary buffers before the following vector blend consumes the complete plane block.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector512<byte>.Count; nuint vectorCount = Numerics.Vector512Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<byte>.Count)
{ {
Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column); Vector512<ushort> first0 = Vector512.LoadUnsafe(ref firstReference, (nuint)column);
Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count)); Vector512<ushort> first1 = Vector512.LoadUnsafe(ref firstReference, (nuint)(column + Vector512<ushort>.Count));
@ -179,8 +179,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector256<byte>.Count; nuint vectorCount = Numerics.Vector256Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<byte>.Count)
{ {
Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column); Vector256<ushort> first0 = Vector256.LoadUnsafe(ref firstReference, (nuint)column);
Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count)); Vector256<ushort> first1 = Vector256.LoadUnsafe(ref firstReference, (nuint)(column + Vector256<ushort>.Count));
@ -193,8 +193,8 @@ internal static partial class Av1DifferenceWeightedMaskBuilder
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - column);
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, column += Vector128<byte>.Count)
{ {
Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column); Vector128<ushort> first0 = Vector128.LoadUnsafe(ref firstReference, (nuint)column);
Vector128<ushort> first1 = Vector128.LoadUnsafe(ref firstReference, (nuint)(column + Vector128<ushort>.Count)); Vector128<ushort> first1 = Vector128.LoadUnsafe(ref firstReference, (nuint)(column + Vector128<ushort>.Count));

132
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1ScaledInterPredictor.cs

@ -411,8 +411,8 @@ internal static partial class Av1ScaledInterPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector512<int>.Count; nuint vectorCount = Numerics.Vector512Count<int>(width - column);
for (; column <= oneVectorFromEnd; column += Vector512<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector512<int>.Count)
{ {
Vector512<int> result = FilterScaledHorizontalVector512<TSource, THorizontal>( Vector512<int> result = FilterScaledHorizontalVector512<TSource, THorizontal>(
ref sourceRow, ref sourceRow,
@ -431,8 +431,8 @@ internal static partial class Av1ScaledInterPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int oneVectorFromEnd = width - Vector256<int>.Count; nuint vectorCount = Numerics.Vector256Count<int>(width - column);
for (; column <= oneVectorFromEnd; column += Vector256<int>.Count) for (; vectorCount > 0; vectorCount--, column += Vector256<int>.Count)
{ {
Vector256<int> result = FilterScaledHorizontalVector256<TSource, THorizontal>( Vector256<int> result = FilterScaledHorizontalVector256<TSource, THorizontal>(
ref sourceRow, ref sourceRow,
@ -451,7 +451,8 @@ internal static partial class Av1ScaledInterPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
for (; column <= width - Vector128<int>.Count; column += Vector128<int>.Count) nuint vectorCount = Numerics.Vector128Count<int>(width - column);
for (; vectorCount > 0; vectorCount--, column += Vector128<int>.Count)
{ {
Vector128<int> result = FilterScaledHorizontalVector128<TSource, THorizontal>( Vector128<int> result = FilterScaledHorizontalVector128<TSource, THorizontal>(
ref sourceRow, ref sourceRow,
@ -505,70 +506,89 @@ internal static partial class Av1ScaledInterPredictor
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
Vector512<int> initial = Vector512.Create(verticalBias); nuint vectorCount = Numerics.Vector512Count<int>(width - column) / 2;
Vector512<int> offset = Vector512.Create(roundOffset);
int oneVectorFromEnd = width - (Vector512<int>.Count * 2); if (vectorCount > 0)
for (; column <= oneVectorFromEnd; column += Vector512<int>.Count * 2)
{ {
NativeOperator.Convolve( // The scaled vertical kernel emits two register widths per iteration, so constants are needed only
ref scratchRow, // when the remaining row contains at least two complete vectors.
scratchStride, Vector512<int> initial = Vector512.Create(verticalBias);
(nuint)column, Vector512<int> offset = Vector512.Create(roundOffset);
ref coefficientBase,
FilterCoefficientCount, for (; vectorCount > 0; vectorCount--, column += Vector512<int>.Count * 2)
initial, {
out Vector512<int> result0, NativeOperator.Convolve(
out Vector512<int> result1); ref scratchRow,
scratchStride,
result0 = RoundPowerOfTwo(result0, round1) - offset; (nuint)column,
result1 = RoundPowerOfTwo(result1, round1) - offset; ref coefficientBase,
TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); FilterCoefficientCount,
initial,
out Vector512<int> result0,
out Vector512<int> result1);
result0 = RoundPowerOfTwo(result0, round1) - offset;
result1 = RoundPowerOfTwo(result1, round1) - offset;
TOperator.Store(ref destinationRow, column, result0, result1, bitDepth);
}
} }
} }
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
Vector256<int> initial = Vector256.Create(verticalBias); nuint vectorCount = Numerics.Vector256Count<int>(width - column) / 2;
Vector256<int> offset = Vector256.Create(roundOffset);
int oneVectorFromEnd = width - (Vector256<int>.Count * 2); if (vectorCount > 0)
for (; column <= oneVectorFromEnd; column += Vector256<int>.Count * 2)
{ {
NativeOperator.Convolve( // A single-vector remainder belongs to the next narrower tier and does not materialize YMM constants.
ref scratchRow, Vector256<int> initial = Vector256.Create(verticalBias);
scratchStride, Vector256<int> offset = Vector256.Create(roundOffset);
(nuint)column,
ref coefficientBase, for (; vectorCount > 0; vectorCount--, column += Vector256<int>.Count * 2)
FilterCoefficientCount, {
initial, NativeOperator.Convolve(
out Vector256<int> result0, ref scratchRow,
out Vector256<int> result1); scratchStride,
(nuint)column,
result0 = RoundPowerOfTwo(result0, round1) - offset; ref coefficientBase,
result1 = RoundPowerOfTwo(result1, round1) - offset; FilterCoefficientCount,
TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); initial,
out Vector256<int> result0,
out Vector256<int> result1);
result0 = RoundPowerOfTwo(result0, round1) - offset;
result1 = RoundPowerOfTwo(result1, round1) - offset;
TOperator.Store(ref destinationRow, column, result0, result1, bitDepth);
}
} }
} }
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
Vector128<int> initial = Vector128.Create(verticalBias); nuint vectorCount = Numerics.Vector128Count<int>(width - column) / 2;
Vector128<int> offset = Vector128.Create(roundOffset);
int oneVectorFromEnd = width - (Vector128<int>.Count * 2); if (vectorCount > 0)
for (; column <= oneVectorFromEnd; column += Vector128<int>.Count * 2)
{ {
NativeOperator.Convolve( // XMM constants are likewise skipped when fewer than eight output samples remain.
ref scratchRow, Vector128<int> initial = Vector128.Create(verticalBias);
scratchStride, Vector128<int> offset = Vector128.Create(roundOffset);
(nuint)column,
ref coefficientBase, for (; vectorCount > 0; vectorCount--, column += Vector128<int>.Count * 2)
FilterCoefficientCount, {
initial, NativeOperator.Convolve(
out Vector128<int> result0, ref scratchRow,
out Vector128<int> result1); scratchStride,
(nuint)column,
result0 = RoundPowerOfTwo(result0, round1) - offset; ref coefficientBase,
result1 = RoundPowerOfTwo(result1, round1) - offset; FilterCoefficientCount,
TOperator.Store(ref destinationRow, column, result0, result1, bitDepth); initial,
out Vector128<int> result0,
out Vector128<int> result1);
result0 = RoundPowerOfTwo(result0, round1) - offset;
result1 = RoundPowerOfTwo(result1, round1) - offset;
TOperator.Store(ref destinationRow, column, result0, result1, bitDepth);
}
} }
} }

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.Dispatch.cs

@ -371,14 +371,14 @@ internal static partial class Av1TranslationalInterPredictor
if (Vector512.IsHardwareAccelerated && Vector<int>.Count == Vector512<int>.Count && width >= Vector512<byte>.Count) if (Vector512.IsHardwareAccelerated && Vector<int>.Count == Vector512<int>.Count && width >= Vector512<byte>.Count)
{ {
int vectorEnd = width - Vector512<byte>.Count; int vectorEnd = (int)(Numerics.Vector512Count<byte>(width) * (nuint)Vector512<byte>.Count);
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
{ {
ref byte sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride); ref byte sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride);
ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int column = 0; int column = 0;
for (; column <= vectorEnd; column += Vector512<byte>.Count) for (; column < vectorEnd; column += Vector512<byte>.Count)
{ {
Vector512.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); Vector512.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column);
} }
@ -394,14 +394,14 @@ internal static partial class Av1TranslationalInterPredictor
if (Vector256.IsHardwareAccelerated && width >= Vector256<byte>.Count) if (Vector256.IsHardwareAccelerated && width >= Vector256<byte>.Count)
{ {
int vectorEnd = width - Vector256<byte>.Count; int vectorEnd = (int)(Numerics.Vector256Count<byte>(width) * (nuint)Vector256<byte>.Count);
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
{ {
ref byte sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride); ref byte sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride);
ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int column = 0; int column = 0;
for (; column <= vectorEnd; column += Vector256<byte>.Count) for (; column < vectorEnd; column += Vector256<byte>.Count)
{ {
Vector256.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); Vector256.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column);
} }
@ -428,9 +428,9 @@ internal static partial class Av1TranslationalInterPredictor
continue; continue;
} }
int vectorEnd = width - Vector128<byte>.Count; int vectorEnd = (int)(Numerics.Vector128Count<byte>(width) * (nuint)Vector128<byte>.Count);
int column = 0; int column = 0;
for (; column <= vectorEnd; column += Vector128<byte>.Count) for (; column < vectorEnd; column += Vector128<byte>.Count)
{ {
Vector128.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); Vector128.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column);
} }
@ -464,14 +464,14 @@ internal static partial class Av1TranslationalInterPredictor
if (Vector512.IsHardwareAccelerated && Vector<int>.Count == Vector512<int>.Count && width >= Vector512<ushort>.Count) if (Vector512.IsHardwareAccelerated && Vector<int>.Count == Vector512<int>.Count && width >= Vector512<ushort>.Count)
{ {
int vectorEnd = width - Vector512<ushort>.Count; int vectorEnd = (int)(Numerics.Vector512Count<ushort>(width) * (nuint)Vector512<ushort>.Count);
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
{ {
ref ushort sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride); ref ushort sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride);
ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int column = 0; int column = 0;
for (; column <= vectorEnd; column += Vector512<ushort>.Count) for (; column < vectorEnd; column += Vector512<ushort>.Count)
{ {
Vector512.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); Vector512.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column);
} }
@ -487,14 +487,14 @@ internal static partial class Av1TranslationalInterPredictor
if (Vector256.IsHardwareAccelerated && width >= Vector256<ushort>.Count) if (Vector256.IsHardwareAccelerated && width >= Vector256<ushort>.Count)
{ {
int vectorEnd = width - Vector256<ushort>.Count; int vectorEnd = (int)(Numerics.Vector256Count<ushort>(width) * (nuint)Vector256<ushort>.Count);
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
{ {
ref ushort sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride); ref ushort sourceRow = ref Unsafe.Add(ref sourceBase, row * sourceStride);
ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int column = 0; int column = 0;
for (; column <= vectorEnd; column += Vector256<ushort>.Count) for (; column < vectorEnd; column += Vector256<ushort>.Count)
{ {
Vector256.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); Vector256.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column);
} }
@ -523,9 +523,9 @@ internal static partial class Av1TranslationalInterPredictor
continue; continue;
} }
int vectorEnd = width - Vector128<ushort>.Count; int vectorEnd = (int)(Numerics.Vector128Count<ushort>(width) * (nuint)Vector128<ushort>.Count);
int column = 0; int column = 0;
for (; column <= vectorEnd; column += Vector128<ushort>.Count) for (; column < vectorEnd; column += Vector128<ushort>.Count)
{ {
Vector128.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column); Vector128.LoadUnsafe(ref sourceRow, (nuint)column).StoreUnsafe(ref destinationRow, (nuint)column);
} }

24
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.OneDimension.cs

@ -62,8 +62,8 @@ internal static partial class Av1TranslationalInterPredictor
continue; continue;
} }
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - processedColumns);
for (; processedColumns <= vectorEnd; processedColumns += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, processedColumns += Vector128<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -107,7 +107,7 @@ internal static partial class Av1TranslationalInterPredictor
ref byte sourceBase = ref Unsafe.Add(ref MemoryMarshal.GetReference(source), sourceOrigin); ref byte sourceBase = ref Unsafe.Add(ref MemoryMarshal.GetReference(source), sourceOrigin);
ref byte destinationBase = ref MemoryMarshal.GetReference(destination); ref byte destinationBase = ref MemoryMarshal.GetReference(destination);
ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients); ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients);
int vectorEnd = width - Vector256<byte>.Count; int vectorEnd = (int)(Numerics.Vector256Count<byte>(width) * (nuint)Vector256<byte>.Count);
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
{ {
@ -115,7 +115,7 @@ internal static partial class Av1TranslationalInterPredictor
ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector256<byte>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector256<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -159,7 +159,7 @@ internal static partial class Av1TranslationalInterPredictor
ref byte sourceBase = ref Unsafe.Add(ref MemoryMarshal.GetReference(source), sourceOrigin); ref byte sourceBase = ref Unsafe.Add(ref MemoryMarshal.GetReference(source), sourceOrigin);
ref byte destinationBase = ref MemoryMarshal.GetReference(destination); ref byte destinationBase = ref MemoryMarshal.GetReference(destination);
ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients); ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients);
int vectorEnd = width - Vector512<byte>.Count; int vectorEnd = (int)(Numerics.Vector512Count<byte>(width) * (nuint)Vector512<byte>.Count);
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
{ {
@ -167,7 +167,7 @@ internal static partial class Av1TranslationalInterPredictor
ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector512<byte>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector512<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -232,8 +232,8 @@ internal static partial class Av1TranslationalInterPredictor
continue; continue;
} }
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - processedColumns);
for (; processedColumns <= vectorEnd; processedColumns += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, processedColumns += Vector128<ushort>.Count)
{ {
Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector128<int> result0, out Vector128<int> result1); Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector128<int> result0, out Vector128<int> result1);
Round(ref result0, ref result1, firstRound, secondRound); Round(ref result0, ref result1, firstRound, secondRound);
@ -268,7 +268,7 @@ internal static partial class Av1TranslationalInterPredictor
ref ushort destinationBase = ref MemoryMarshal.GetReference(destination); ref ushort destinationBase = ref MemoryMarshal.GetReference(destination);
ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients); ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients);
int maximum = (1 << bitDepth) - 1; int maximum = (1 << bitDepth) - 1;
int vectorEnd = width - Vector256<ushort>.Count; int vectorEnd = (int)(Numerics.Vector256Count<ushort>(width) * (nuint)Vector256<ushort>.Count);
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
{ {
@ -277,7 +277,7 @@ internal static partial class Av1TranslationalInterPredictor
ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector256<ushort>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector256<ushort>.Count)
{ {
Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector256<int> result0, out Vector256<int> result1); Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector256<int> result0, out Vector256<int> result1);
Round(ref result0, ref result1, firstRound, secondRound); Round(ref result0, ref result1, firstRound, secondRound);
@ -312,7 +312,7 @@ internal static partial class Av1TranslationalInterPredictor
ref ushort destinationBase = ref MemoryMarshal.GetReference(destination); ref ushort destinationBase = ref MemoryMarshal.GetReference(destination);
ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients); ref short coefficientBase = ref MemoryMarshal.GetReference(coefficients);
int maximum = (1 << bitDepth) - 1; int maximum = (1 << bitDepth) - 1;
int vectorEnd = width - Vector512<ushort>.Count; int vectorEnd = (int)(Numerics.Vector512Count<ushort>(width) * (nuint)Vector512<ushort>.Count);
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
{ {
@ -321,7 +321,7 @@ internal static partial class Av1TranslationalInterPredictor
ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector512<ushort>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector512<ushort>.Count)
{ {
Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector512<int> result0, out Vector512<int> result1); Convolve(ref sourceRow, tapStride, (nuint)processedColumns, ref coefficientBase, tapCount, initial, out Vector512<int> result0, out Vector512<int> result1);
Round(ref result0, ref result1, firstRound, secondRound); Round(ref result0, ref result1, firstRound, secondRound);

20
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.Byte.cs

@ -68,8 +68,8 @@ internal static partial class Av1TranslationalInterPredictor
continue; continue;
} }
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - processedColumns);
for (; processedColumns <= vectorEnd; processedColumns += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, processedColumns += Vector128<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -135,8 +135,8 @@ internal static partial class Av1TranslationalInterPredictor
continue; continue;
} }
int vectorEnd = width - Vector128<byte>.Count; nuint vectorCount = Numerics.Vector128Count<byte>(width - processedColumns);
for (; processedColumns <= vectorEnd; processedColumns += Vector128<byte>.Count) for (; vectorCount > 0; vectorCount--, processedColumns += Vector128<byte>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -199,7 +199,7 @@ internal static partial class Av1TranslationalInterPredictor
int scratchStride = Math.Max(width, MinimumScratchStride); int scratchStride = Math.Max(width, MinimumScratchStride);
int intermediateHeight = height + verticalTapCount - 1; int intermediateHeight = height + verticalTapCount - 1;
Vector256<int> horizontalInitial = initial + Vector256.Create(1 << (bitDepth + FilterBits - 1)); Vector256<int> horizontalInitial = initial + Vector256.Create(1 << (bitDepth + FilterBits - 1));
int vectorEnd = width - Vector256<byte>.Count; int vectorEnd = (int)(Numerics.Vector256Count<byte>(width) * (nuint)Vector256<byte>.Count);
for (int row = 0; row < intermediateHeight; row++) for (int row = 0; row < intermediateHeight; row++)
{ {
@ -207,7 +207,7 @@ internal static partial class Av1TranslationalInterPredictor
ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride); ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector256<byte>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector256<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -243,7 +243,7 @@ internal static partial class Av1TranslationalInterPredictor
ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector256<byte>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector256<byte>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -306,7 +306,7 @@ internal static partial class Av1TranslationalInterPredictor
int scratchStride = Math.Max(width, MinimumScratchStride); int scratchStride = Math.Max(width, MinimumScratchStride);
int intermediateHeight = height + verticalTapCount - 1; int intermediateHeight = height + verticalTapCount - 1;
Vector512<int> horizontalInitial = initial + Vector512.Create(1 << (bitDepth + FilterBits - 1)); Vector512<int> horizontalInitial = initial + Vector512.Create(1 << (bitDepth + FilterBits - 1));
int vectorEnd = width - Vector512<byte>.Count; int vectorEnd = (int)(Numerics.Vector512Count<byte>(width) * (nuint)Vector512<byte>.Count);
for (int row = 0; row < intermediateHeight; row++) for (int row = 0; row < intermediateHeight; row++)
{ {
@ -314,7 +314,7 @@ internal static partial class Av1TranslationalInterPredictor
ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride); ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector512<byte>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector512<byte>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -350,7 +350,7 @@ internal static partial class Av1TranslationalInterPredictor
ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref byte destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector512<byte>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector512<byte>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,

20
src/ImageSharp/Formats/Heif/Av1/Prediction/Inter/Av1TranslationalInterPredictor.TwoDimensions.UInt16.cs

@ -74,8 +74,8 @@ internal static partial class Av1TranslationalInterPredictor
continue; continue;
} }
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - processedColumns);
for (; processedColumns <= vectorEnd; processedColumns += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, processedColumns += Vector128<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -138,8 +138,8 @@ internal static partial class Av1TranslationalInterPredictor
continue; continue;
} }
int vectorEnd = width - Vector128<ushort>.Count; nuint vectorCount = Numerics.Vector128Count<ushort>(width - processedColumns);
for (; processedColumns <= vectorEnd; processedColumns += Vector128<ushort>.Count) for (; vectorCount > 0; vectorCount--, processedColumns += Vector128<ushort>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -199,7 +199,7 @@ internal static partial class Av1TranslationalInterPredictor
int scratchStride = Math.Max(width, MinimumScratchStride); int scratchStride = Math.Max(width, MinimumScratchStride);
int intermediateHeight = height + verticalTapCount - 1; int intermediateHeight = height + verticalTapCount - 1;
Vector256<int> horizontalInitial = initial + Vector256.Create(1 << (bitDepth + FilterBits - 1)); Vector256<int> horizontalInitial = initial + Vector256.Create(1 << (bitDepth + FilterBits - 1));
int vectorEnd = width - Vector256<ushort>.Count; int vectorEnd = (int)(Numerics.Vector256Count<ushort>(width) * (nuint)Vector256<ushort>.Count);
for (int row = 0; row < intermediateHeight; row++) for (int row = 0; row < intermediateHeight; row++)
{ {
@ -211,7 +211,7 @@ internal static partial class Av1TranslationalInterPredictor
ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride); ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector256<ushort>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector256<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -253,7 +253,7 @@ internal static partial class Av1TranslationalInterPredictor
ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector256<ushort>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector256<ushort>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,
@ -313,7 +313,7 @@ internal static partial class Av1TranslationalInterPredictor
int scratchStride = Math.Max(width, MinimumScratchStride); int scratchStride = Math.Max(width, MinimumScratchStride);
int intermediateHeight = height + verticalTapCount - 1; int intermediateHeight = height + verticalTapCount - 1;
Vector512<int> horizontalInitial = initial + Vector512.Create(1 << (bitDepth + FilterBits - 1)); Vector512<int> horizontalInitial = initial + Vector512.Create(1 << (bitDepth + FilterBits - 1));
int vectorEnd = width - Vector512<ushort>.Count; int vectorEnd = (int)(Numerics.Vector512Count<ushort>(width) * (nuint)Vector512<ushort>.Count);
for (int row = 0; row < intermediateHeight; row++) for (int row = 0; row < intermediateHeight; row++)
{ {
@ -325,7 +325,7 @@ internal static partial class Av1TranslationalInterPredictor
ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride); ref short scratchRow = ref Unsafe.Add(ref scratchBase, row * scratchStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector512<ushort>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector512<ushort>.Count)
{ {
Convolve( Convolve(
ref sourceRow, ref sourceRow,
@ -367,7 +367,7 @@ internal static partial class Av1TranslationalInterPredictor
ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride); ref ushort destinationRow = ref Unsafe.Add(ref destinationBase, row * destinationStride);
int processedColumns = 0; int processedColumns = 0;
for (; processedColumns <= vectorEnd; processedColumns += Vector512<ushort>.Count) for (; processedColumns < vectorEnd; processedColumns += Vector512<ushort>.Count)
{ {
Convolve( Convolve(
ref scratchRow, ref scratchRow,

12
src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyBilinearPredictor.cs

@ -110,7 +110,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor
// cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail. // cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorizedColumns = width - (width % Vector512<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector512Count<byte>(width) * (nuint)Vector512<byte>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -136,7 +136,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector256<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector256Count<byte>(remainingColumns) * (nuint)Vector256<byte>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -161,7 +161,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector128<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector128Count<byte>(remainingColumns) * (nuint)Vector128<byte>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -244,7 +244,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor
// continuation as the byte path. // continuation as the byte path.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorizedColumns = width - (width % Vector512<short>.Count); int vectorizedColumns = (int)(Numerics.Vector512Count<short>(width) * (nuint)Vector512<short>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -270,7 +270,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector256<short>.Count); int vectorizedColumns = (int)(Numerics.Vector256Count<short>(remainingColumns) * (nuint)Vector256<short>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -295,7 +295,7 @@ internal static partial class Av1IntraBlockCopyBilinearPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector128<short>.Count); int vectorizedColumns = (int)(Numerics.Vector128Count<short>(remainingColumns) * (nuint)Vector128<short>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)

12
src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyHorizontalPredictor.cs

@ -108,7 +108,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor
// cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail. // cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorizedColumns = width - (width % Vector512<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector512Count<byte>(width) * (nuint)Vector512<byte>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -132,7 +132,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector256<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector256Count<byte>(remainingColumns) * (nuint)Vector256<byte>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -155,7 +155,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector128<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector128Count<byte>(remainingColumns) * (nuint)Vector128<byte>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -232,7 +232,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor
// continuation as the byte path. // continuation as the byte path.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorizedColumns = width - (width % Vector512<short>.Count); int vectorizedColumns = (int)(Numerics.Vector512Count<short>(width) * (nuint)Vector512<short>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -256,7 +256,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector256<short>.Count); int vectorizedColumns = (int)(Numerics.Vector256Count<short>(remainingColumns) * (nuint)Vector256<short>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -279,7 +279,7 @@ internal static partial class Av1IntraBlockCopyHorizontalPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector128<short>.Count); int vectorizedColumns = (int)(Numerics.Vector128Count<short>(remainingColumns) * (nuint)Vector128<short>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)

12
src/ImageSharp/Formats/Heif/Av1/Prediction/IntraBlockCopy/Av1IntraBlockCopyVerticalPredictor.cs

@ -108,7 +108,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor
// cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail. // cumulative narrower tiers preserve the same contract for future legal widths without over-reading a tail.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorizedColumns = width - (width % Vector512<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector512Count<byte>(width) * (nuint)Vector512<byte>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -132,7 +132,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector256<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector256Count<byte>(remainingColumns) * (nuint)Vector256<byte>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -155,7 +155,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector128<byte>.Count); int vectorizedColumns = (int)(Numerics.Vector128Count<byte>(remainingColumns) * (nuint)Vector128<byte>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -232,7 +232,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor
// continuation as the byte path. // continuation as the byte path.
if (Vector512.IsHardwareAccelerated) if (Vector512.IsHardwareAccelerated)
{ {
int vectorizedColumns = width - (width % Vector512<short>.Count); int vectorizedColumns = (int)(Numerics.Vector512Count<short>(width) * (nuint)Vector512<short>.Count);
if (vectorizedColumns > 0) if (vectorizedColumns > 0)
{ {
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -256,7 +256,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor
if (Vector256.IsHardwareAccelerated) if (Vector256.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector256<short>.Count); int vectorizedColumns = (int)(Numerics.Vector256Count<short>(remainingColumns) * (nuint)Vector256<short>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)
@ -279,7 +279,7 @@ internal static partial class Av1IntraBlockCopyVerticalPredictor
if (Vector128.IsHardwareAccelerated) if (Vector128.IsHardwareAccelerated)
{ {
int remainingColumns = width - processedColumns; int remainingColumns = width - processedColumns;
int vectorizedColumns = remainingColumns - (remainingColumns % Vector128<short>.Count); int vectorizedColumns = (int)(Numerics.Vector128Count<short>(remainingColumns) * (nuint)Vector128<short>.Count);
int endColumn = processedColumns + vectorizedColumns; int endColumn = processedColumns + vectorizedColumns;
for (int row = 0; row < height; row++) for (int row = 0; row < height; row++)

9
src/ImageSharp/Formats/Heif/Av1/Transform/Av1ForwardTransformer.cs

@ -1046,7 +1046,8 @@ internal static partial class Av1ForwardTransformer
if (Avx512BW.IsSupported) if (Avx512BW.IsSupported)
{ {
for (; column <= width - Vector512<short>.Count; column += Vector512<short>.Count) nuint vector512Count = Numerics.Vector512Count<short>(width - column);
for (; vector512Count > 0; vector512Count--, column += Vector512<short>.Count)
{ {
(Vector512<int> lower, Vector512<int> upper) = Vector512.Widen(Vector512.LoadUnsafe(ref sourceRow, (nuint)column)); (Vector512<int> lower, Vector512<int> upper) = Vector512.Widen(Vector512.LoadUnsafe(ref sourceRow, (nuint)column));
@ -1057,7 +1058,8 @@ internal static partial class Av1ForwardTransformer
if (Avx2.IsSupported) if (Avx2.IsSupported)
{ {
for (; column <= width - Vector256<short>.Count; column += Vector256<short>.Count) nuint vector256Count = Numerics.Vector256Count<short>(width - column);
for (; vector256Count > 0; vector256Count--, column += Vector256<short>.Count)
{ {
(Vector256<int> lower, Vector256<int> upper) = Vector256.Widen(Vector256.LoadUnsafe(ref sourceRow, (nuint)column)); (Vector256<int> lower, Vector256<int> upper) = Vector256.Widen(Vector256.LoadUnsafe(ref sourceRow, (nuint)column));
@ -1066,7 +1068,8 @@ internal static partial class Av1ForwardTransformer
} }
} }
for (; column <= width - Vector128<short>.Count; column += Vector128<short>.Count) nuint vector128Count = Numerics.Vector128Count<short>(width - column);
for (; vector128Count > 0; vector128Count--, column += Vector128<short>.Count)
{ {
(Vector128<int> lower, Vector128<int> upper) = Vector128.Widen(Vector128.LoadUnsafe(ref sourceRow, (nuint)column)); (Vector128<int> lower, Vector128<int> upper) = Vector128.Widen(Vector128.LoadUnsafe(ref sourceRow, (nuint)column));

4
src/ImageSharp/Formats/Heif/HeifEncoder.cs

@ -31,8 +31,8 @@ public sealed class HeifEncoder : AnimatedImageEncoder
/// <summary> /// <summary>
/// Gets the lossy compression quality, or <see langword="null"/> to use the compression method's default quality. /// Gets the lossy compression quality, or <see langword="null"/> to use the compression method's default quality.
/// Valid values range from 0 for the lowest quality to 100 for the highest quality. Legacy JPEG image items /// Valid values range from 0 for the lowest quality to 100 for the highest quality. A value of 100 does not
/// support values from 1 through 100. A value of 100 does not enable <see cref="Lossless"/> encoding. /// enable <see cref="Lossless"/> encoding.
/// </summary> /// </summary>
/// <exception cref="ArgumentException">The quality is outside the range 0 to 100.</exception> /// <exception cref="ArgumentException">The quality is outside the range 0 to 100.</exception>
public int? Quality public int? Quality

14
src/ImageSharp/Formats/Heif/HeifEncoderCore.cs

@ -64,9 +64,6 @@ internal sealed class HeifEncoderCore
this.WriteMetadataBox(items, links, stream); this.WriteMetadataBox(items, links, stream);
this.WriteMediaDataBox(compressedPixels, stream); this.WriteMediaDataBox(compressedPixels, stream);
stream.Flush(); stream.Flush();
HeifMetadata meta = image.Metadata.GetHeifMetadata();
meta.CompressionMethod = this.encoder.CompressionMethod;
} }
/// <summary> /// <summary>
@ -448,13 +445,6 @@ internal sealed class HeifEncoderCore
throw new NotSupportedException("Legacy JPEG image items support only 8-bit component encoding."); throw new NotSupportedException("Legacy JPEG image items support only 8-bit component encoding.");
} }
if (this.encoder.Quality == 0)
{
// Zero is meaningful to the AV1 quality scale, but ImageSharp's JPEG encoder deliberately
// exposes the JPEG quality scale as 1 through 100. Reject the codec-specific mismatch at this boundary.
throw new NotSupportedException("Legacy JPEG image items support quality values in the range [1..100].");
}
JpegColorType colorType = this.encoder.ChromaSubsampling switch JpegColorType colorType = this.encoder.ChromaSubsampling switch
{ {
null or HeifChromaSubsampling.Yuv420 => JpegColorType.YCbCrRatio420, null or HeifChromaSubsampling.Yuv420 => JpegColorType.YCbCrRatio420,
@ -467,7 +457,9 @@ internal sealed class HeifEncoderCore
ChunkedMemoryStream stream = new(this.configuration.MemoryAllocator); ChunkedMemoryStream stream = new(this.configuration.MemoryAllocator);
JpegEncoder encoder = new() JpegEncoder encoder = new()
{ {
Quality = this.encoder.Quality, // The HEIF quality scale includes zero while the JPEG payload encoder starts at one.
// Map the lowest HEIF setting to the lowest representable JPEG setting.
Quality = this.encoder.Quality == 0 ? 1 : this.encoder.Quality,
ColorType = colorType ColorType = colorType
}; };

1
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1BitStreamTests.cs

@ -175,6 +175,7 @@ public class Av1BitStreamTests
[InlineData(4, 0, 1, 2, 3)] [InlineData(4, 0, 1, 2, 3)]
[InlineData(5, 0, 1, 2, 3)] [InlineData(5, 0, 1, 2, 3)]
[InlineData(5, 1, 2, 3, 4)]
[InlineData(8, 0, 1, 2, 3)] [InlineData(8, 0, 1, 2, 3)]
[InlineData(8, 4, 5, 6, 7)] [InlineData(8, 4, 5, 6, 7)]
[InlineData(16, 15, 0, 5, 8)] [InlineData(16, 15, 0, 5, 8)]

98
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1EncoderFrameTests.cs

@ -0,0 +1,98 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using SixLabors.ImageSharp.Formats.Heif.Av1;
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
using SixLabors.ImageSharp.Memory;
namespace SixLabors.ImageSharp.Tests.Formats.Heif.Av1;
public class Av1EncoderFrameTests
{
[Fact]
public void ExtendBordersReplicatesEveryPhysicalPlaneEdge()
{
const int visibleWidth = 5;
const int visibleHeight = 3;
const int codedWidth = 8;
const int codedHeight = 8;
const int lumaBorder = Av1EncoderFrame<byte>.LumaBorder;
const int chromaBorder = lumaBorder / 2;
MemoryAllocator allocator = Configuration.Default.MemoryAllocator;
Size lumaBufferSize = Av1EncoderFrame<byte>.GetPlaneBufferSize(visibleWidth, visibleHeight, 0, 0);
Size chromaBufferSize = Av1EncoderFrame<byte>.GetPlaneBufferSize(visibleWidth, visibleHeight, 1, 1);
using Buffer2D<byte> luma = allocator.Allocate2D<byte>(lumaBufferSize.Width, lumaBufferSize.Height);
using Buffer2D<byte> chromaBlue = allocator.Allocate2D<byte>(chromaBufferSize.Width, chromaBufferSize.Height);
using Buffer2D<byte> chromaRed = allocator.Allocate2D<byte>(chromaBufferSize.Width, chromaBufferSize.Height);
Buffer2DRegion<byte> lumaRegion = luma.GetRegion(lumaBorder, lumaBorder, codedWidth, codedHeight);
Buffer2DRegion<byte> chromaBlueRegion = chromaBlue.GetRegion(chromaBorder, chromaBorder, codedWidth / 2, codedHeight / 2);
Buffer2DRegion<byte> chromaRedRegion = chromaRed.GetRegion(chromaBorder, chromaBorder, codedWidth / 2, codedHeight / 2);
FillVisible(luma, lumaBorder, lumaBorder, visibleWidth, visibleHeight, 10);
FillVisible(chromaBlue, chromaBorder, chromaBorder, (visibleWidth + 1) / 2, (visibleHeight + 1) / 2, 80);
FillVisible(chromaRed, chromaBorder, chromaBorder, (visibleWidth + 1) / 2, (visibleHeight + 1) / 2, 120);
Av1EncoderFrame<byte> frame = new(
lumaRegion,
chromaBlueRegion,
chromaRedRegion,
visibleWidth,
visibleHeight,
8,
Av1ColorFormat.Yuv420,
1,
1);
frame.ExtendBorders();
AssertReplicatedPlane(luma, lumaBorder, lumaBorder, visibleWidth, visibleHeight, 10);
AssertReplicatedPlane(chromaBlue, chromaBorder, chromaBorder, (visibleWidth + 1) / 2, (visibleHeight + 1) / 2, 80);
AssertReplicatedPlane(chromaRed, chromaBorder, chromaBorder, (visibleWidth + 1) / 2, (visibleHeight + 1) / 2, 120);
}
[Theory]
[InlineData(5, 3, 0, 0, 160, 136)]
[InlineData(5, 3, 1, 0, 80, 136)]
[InlineData(5, 3, 1, 1, 80, 68)]
[InlineData(1921, 1081, 0, 0, 2080, 1216)]
[InlineData(1921, 1081, 1, 1, 1040, 608)]
public void GetPlaneBufferSizeMatchesLibaomLayout(
int width,
int height,
int subsamplingX,
int subsamplingY,
int expectedWidth,
int expectedHeight)
{
Size actual = Av1EncoderFrame<byte>.GetPlaneBufferSize(width, height, subsamplingX, subsamplingY);
Assert.Equal(new Size(expectedWidth, expectedHeight), actual);
}
private static void FillVisible(Buffer2D<byte> plane, int originX, int originY, int width, int height, int seed)
{
for (int y = 0; y < height; y++)
{
Span<byte> row = plane.DangerousGetRowSpan(originY + y);
for (int x = 0; x < width; x++)
{
row[originX + x] = (byte)(seed + (y * width) + x);
}
}
}
private static void AssertReplicatedPlane(Buffer2D<byte> plane, int originX, int originY, int width, int height, int seed)
{
for (int y = 0; y < plane.Height; y++)
{
ReadOnlySpan<byte> row = plane.DangerousGetRowSpan(y);
int sourceY = Math.Clamp(y - originY, 0, height - 1);
for (int x = 0; x < row.Length; x++)
{
int sourceX = Math.Clamp(x - originX, 0, width - 1);
Assert.Equal((byte)(seed + (sourceY * width) + sourceX), row[x]);
}
}
}
}

212
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1ForwardQuantizerTests.cs

@ -0,0 +1,212 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using SixLabors.ImageSharp.Formats.Heif.Av1;
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers;
using SixLabors.ImageSharp.Formats.Heif.Av1.Transform;
using SixLabors.ImageSharp.Tests.TestUtilities;
namespace SixLabors.ImageSharp.Tests.Formats.Heif.Av1;
/// <summary>
/// Verifies AV1 forward quantization against current libaom's fast no-matrix arithmetic.
/// </summary>
[Trait("Format", "Avif")]
public class Av1ForwardQuantizerTests
{
/// <summary>
/// The hardware configurations covering every quantizer vector tier and the scalar fallback.
/// </summary>
private const HwIntrinsics QuantizerConfigurations =
HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic;
/// <summary>
/// Verifies raster quantization and scan-order EOB selection at every SIMD tier.
/// </summary>
[Fact]
public void FastQuantizerMatchesLibaomReferenceAcrossHardwareWidths()
=> FeatureTestRunner.RunWithHwIntrinsicsFeature(ValidateQuantizer, QuantizerConfigurations);
/// <summary>
/// Verifies that repeated transform quantization uses only caller-owned buffers.
/// </summary>
[Fact]
public void QuantizerDoesNotAllocatePerTransform()
{
const int coefficientCount = 64;
int[] coefficients = new int[coefficientCount];
int[] quantized = new int[coefficientCount];
int[] dequantized = new int[coefficientCount];
FillCoefficients(coefficients, 73);
Av1ForwardQuantizer.QuantizeLossy(
coefficients,
quantized,
dequantized,
Av1TransformSize.Size8x8,
Av1TransformType.DctDct,
73,
-1,
3,
Av1BitDepth.TenBit);
long before = GC.GetAllocatedBytesForCurrentThread();
for (int iteration = 0; iteration < 32; iteration++)
{
Av1ForwardQuantizer.QuantizeLossy(
coefficients,
quantized,
dequantized,
Av1TransformSize.Size8x8,
Av1TransformType.DctDct,
73,
-1,
3,
Av1BitDepth.TenBit);
}
Assert.Equal(0, GC.GetAllocatedBytesForCurrentThread() - before);
}
/// <summary>
/// Exercises each transform-scale category, coded 64-point layout, quantizer range, and sample precision.
/// </summary>
private static void ValidateQuantizer()
{
ReadOnlySpan<Av1TransformSize> transformSizes =
[
Av1TransformSize.Size4x4,
Av1TransformSize.Size8x8,
Av1TransformSize.Size16x16,
Av1TransformSize.Size32x32,
Av1TransformSize.Size64x16,
Av1TransformSize.Size64x64,
];
ReadOnlySpan<int> quantizerIndices = [1, 73, 173, 255];
ReadOnlySpan<Av1BitDepth> bitDepths = [Av1BitDepth.EightBit, Av1BitDepth.TenBit, Av1BitDepth.TwelveBit];
foreach (Av1TransformSize transformSize in transformSizes)
{
int coefficientCount = transformSize.GetAdjusted().GetSize2d();
int[] coefficients = new int[coefficientCount];
int[] expectedQuantized = new int[coefficientCount];
int[] expectedDequantized = new int[coefficientCount];
int[] actualQuantized = new int[coefficientCount];
int[] actualDequantized = new int[coefficientCount];
foreach (int qIndex in quantizerIndices)
{
FillCoefficients(coefficients, qIndex);
foreach (Av1BitDepth bitDepth in bitDepths)
{
ushort expectedEndOfBlock = QuantizeReference(
coefficients,
expectedQuantized,
expectedDequantized,
transformSize,
Av1TransformType.DctDct,
qIndex,
-1,
3,
bitDepth);
ushort actualEndOfBlock = Av1ForwardQuantizer.QuantizeLossy(
coefficients,
actualQuantized,
actualDequantized,
transformSize,
Av1TransformType.DctDct,
qIndex,
-1,
3,
bitDepth);
Assert.Equal(expectedEndOfBlock, actualEndOfBlock);
Assert.Equal(expectedQuantized, actualQuantized);
Assert.Equal(expectedDequantized, actualDequantized);
}
}
}
}
/// <summary>
/// Fills one transform with deterministic signed values spanning threshold, rounding, and clamp behavior.
/// </summary>
private static void FillCoefficients(Span<int> coefficients, int seed)
{
for (int i = 0; i < coefficients.Length; i++)
{
coefficients[i] = (((i * 7919) + (seed * 313)) % 90001) - 45000;
}
coefficients[0] = 0;
coefficients[1] = 1;
coefficients[2] = -1;
coefficients[3] = short.MaxValue;
coefficients[4] = -short.MaxValue;
}
/// <summary>
/// Mirrors av1_quantize_fp_no_qmatrix from current libaom without sharing the production traversal.
/// </summary>
private static ushort QuantizeReference(
ReadOnlySpan<int> coefficients,
Span<int> quantizedCoefficients,
Span<int> dequantizedCoefficients,
Av1TransformSize transformSize,
Av1TransformType transformType,
int qIndex,
int dcDeltaQ,
int acDeltaQ,
Av1BitDepth bitDepth)
{
quantizedCoefficients.Clear();
dequantizedCoefficients.Clear();
int logScale = transformSize.GetScale();
int dcDequantizer = Av1QuantizationLookup.GetDcQuant(qIndex, dcDeltaQ, bitDepth);
int acDequantizer = Av1QuantizationLookup.GetAcQuant(qIndex, acDeltaQ, bitDepth);
int dcQuantizer = (1 << 16) / dcDequantizer;
int acQuantizer = (1 << 16) / acDequantizer;
int dcRounding = RoundPowerOfTwo((64 * dcDequantizer) >> 7, logScale);
int acRounding = RoundPowerOfTwo((64 * acDequantizer) >> 7, logScale);
ReadOnlySpan<short> scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan;
ushort endOfBlock = 0;
for (int scanIndex = 0; scanIndex < scan.Length; scanIndex++)
{
int coefficientIndex = scan[scanIndex];
int coefficient = coefficients[coefficientIndex];
int coefficientSign = coefficient >> 31;
long magnitude = ((long)coefficient ^ coefficientSign) - coefficientSign;
int dequantizer = coefficientIndex == 0 ? dcDequantizer : acDequantizer;
int quantizer = coefficientIndex == 0 ? dcQuantizer : acQuantizer;
int rounding = coefficientIndex == 0 ? dcRounding : acRounding;
int quantizedMagnitude = 0;
if ((magnitude << (1 + logScale)) >= dequantizer)
{
magnitude = Math.Clamp(magnitude + rounding, short.MinValue, short.MaxValue);
quantizedMagnitude = (int)((magnitude * quantizer) >> (16 - logScale));
}
if (quantizedMagnitude != 0)
{
quantizedCoefficients[coefficientIndex] = (quantizedMagnitude ^ coefficientSign) - coefficientSign;
int dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale;
dequantizedCoefficients[coefficientIndex] = (dequantizedMagnitude ^ coefficientSign) - coefficientSign;
endOfBlock = (ushort)(scanIndex + 1);
}
}
return endOfBlock;
}
/// <summary>
/// Applies libaom's positive round-power-of-two operation.
/// </summary>
private static int RoundPowerOfTwo(int value, int shift)
=> shift == 0 ? value : (value + (1 << (shift - 1))) >> shift;
}

2
tests/ImageSharp.Tests/Formats/Heif/Av1/Av1TileDecoderStub.cs

@ -17,6 +17,6 @@ internal class Av1TileDecoderStub : IAv1TileReader, IAv1TileWriter
{ {
} }
public Span<byte> WriteTile(int tileNum) public ReadOnlySpan<byte> GetTileData(int tileNum)
=> this.tileDatas[tileNum]; => this.tileDatas[tileNum];
} }

55
tests/ImageSharp.Tests/Formats/Heif/Av1/ObuFrameHeaderTests.cs

@ -162,7 +162,7 @@ public class ObuFrameHeaderTests
// Assign 2 // Assign 2
Span<byte> encodedBuffer = encoded.ToArray(); Span<byte> encodedBuffer = encoded.ToArray();
IAv1TileReader tileDecoder2 = new Av1TileDecoderStub(); IAv1TileReader tileDecoder2 = new Av1TileDecoderStub();
Av1BitStreamReader reader2 = new(span); Av1BitStreamReader reader2 = new(encodedBuffer);
ObuReader obuReader2 = new(); ObuReader obuReader2 = new();
// Act 2 // Act 2
@ -704,6 +704,59 @@ public class ObuFrameHeaderTests
Assert.Equal(bitStream.Length * 8, reader.BitPosition); Assert.Equal(bitStream.Length * 8, reader.BitPosition);
} }
/// <summary>
/// Verifies non-uniform tile boundaries use the next stored boundary and retain the clipped final mode-info edge.
/// </summary>
[Fact]
public void WriteNonUniformTileBoundariesRoundTrip()
{
ObuSequenceHeader sequenceHeader = GetDefaultSequenceHeader();
ObuFrameHeader frameHeader = GetKeyFrameHeader();
ObuTileGroupHeader tileInfo = frameHeader.TilesInfo;
tileInfo.HasUniformTileSpacing = false;
tileInfo.TileColumnCount = 2;
tileInfo.TileRowCount = 1;
tileInfo.TileSizeBytes = 1;
tileInfo.TileColumnStartModeInfo[0] = 0;
tileInfo.TileColumnStartModeInfo[1] = 64;
tileInfo.TileColumnStartModeInfo[2] = frameHeader.ModeInfoColumnCount;
tileInfo.TileRowStartModeInfo[0] = 0;
tileInfo.TileRowStartModeInfo[1] = frameHeader.ModeInfoRowCount;
Av1TileDecoderStub sourceTiles = new();
sourceTiles.ReadTile([0x80], 0);
sourceTiles.ReadTile([0x80], 1);
using MemoryStream stream = new();
ObuWriter writer = new();
writer.WriteAll(Configuration.Default, stream, sequenceHeader, frameHeader, sourceTiles);
byte[] bitStream = stream.ToArray();
Assert.Equal([0x00, 0x80, 0x80], bitStream[^3..]);
Av1BitStreamReader reader = new(bitStream);
ObuReader obuReader = new();
Av1TileDecoderStub decodedTiles = new();
obuReader.ReadAll(ref reader, bitStream.Length, () => decodedTiles);
ObuTileGroupHeader actual = obuReader.FrameHeader.TilesInfo;
Assert.False(actual.HasUniformTileSpacing);
Assert.Equal(2, actual.TileColumnCount);
Assert.Equal(1, actual.TileRowCount);
Assert.Equal(1, actual.TileSizeBytes);
Assert.Equal(0, actual.TileColumnStartModeInfo[0]);
Assert.Equal(64, actual.TileColumnStartModeInfo[1]);
Assert.Equal(frameHeader.ModeInfoColumnCount, actual.TileColumnStartModeInfo[2]);
Assert.Equal(0, actual.TileRowStartModeInfo[0]);
Assert.Equal(frameHeader.ModeInfoRowCount, actual.TileRowStartModeInfo[1]);
ReadOnlySpan<byte> expectedTileData = [0x80];
Assert.True(decodedTiles.GetTileData(0).SequenceEqual(expectedTileData));
Assert.True(decodedTiles.GetTileData(1).SequenceEqual(expectedTileData));
Assert.Equal(bitStream.Length * 8, reader.BitPosition);
}
/// <summary> /// <summary>
/// Encodes one non-reduced sequence header with a single selected conformance failure. /// Encodes one non-reduced sequence header with a single selected conformance failure.
/// </summary> /// </summary>

39
tests/ImageSharp.Tests/Formats/Heif/HeifEncoderTests.cs

@ -64,13 +64,19 @@ public class HeifEncoderTests
} }
[Fact] [Fact]
public void LegacyJpegRejectsZeroQuality() public void LegacyJpegAcceptsZeroQuality()
{ {
using Image<Rgba32> image = new(1, 1); using Image<Rgba32> image = new(1, 1);
image[0, 0] = new Rgba32(10, 20, 30);
using MemoryStream stream = new(); using MemoryStream stream = new();
HeifEncoder encoder = new() { Quality = 0 }; HeifEncoder encoder = new() { Quality = 0 };
Assert.Throws<NotSupportedException>(() => image.Save(stream, encoder)); image.Save(stream, encoder);
Assert.NotEqual(0, stream.Length);
stream.Position = 0;
using Image<Rgba32> decoded = Image.Load<Rgba32>(stream);
Assert.Equal(image.Size, decoded.Size);
} }
[Fact] [Fact]
@ -81,6 +87,7 @@ public class HeifEncoderTests
HeifEncoder encoder = new() { Lossless = true }; HeifEncoder encoder = new() { Lossless = true };
Assert.Throws<NotSupportedException>(() => image.Save(stream, encoder)); Assert.Throws<NotSupportedException>(() => image.Save(stream, encoder));
Assert.Equal(0, stream.Length);
} }
[Theory] [Theory]
@ -93,6 +100,34 @@ public class HeifEncoderTests
HeifEncoder encoder = new() { BitDepth = bitDepth }; HeifEncoder encoder = new() { BitDepth = bitDepth };
Assert.Throws<NotSupportedException>(() => image.Save(stream, encoder)); Assert.Throws<NotSupportedException>(() => image.Save(stream, encoder));
Assert.Equal(0, stream.Length);
}
[Fact]
public void Av1RejectsEncodingBeforeWritingOutput()
{
using Image<Rgba32> image = new(1, 1);
using MemoryStream stream = new();
HeifEncoder encoder = new() { CompressionMethod = HeifCompressionMethod.Av1 };
Assert.Throws<NotSupportedException>(() => image.Save(stream, encoder));
Assert.Equal(0, stream.Length);
}
[Fact]
public void LegacyJpegEncodingDoesNotMutateSourceHeifMetadata()
{
using Image<Rgba32> image = new(1, 1);
image[0, 0] = new Rgba32(10, 20, 30, 255);
HeifMetadata metadata = image.Metadata.GetHeifMetadata();
metadata.CompressionMethod = HeifCompressionMethod.Av1;
using MemoryStream stream = new();
HeifEncoder encoder = new();
image.Save(stream, encoder);
Assert.Same(metadata, image.Metadata.GetHeifMetadata());
Assert.Equal(HeifCompressionMethod.Av1, metadata.CompressionMethod);
} }
[Theory] [Theory]

Loading…
Cancel
Save