mirror of https://github.com/SixLabors/ImageSharp
Browse Source
Implement zero-bin thresholds, sharpness rounding, reciprocal correction, and dequantization through scalar and SIMD operators. Fix the .NET 11 AVX-512 test switch so narrower hardware paths actually execute. Verified in the current worktree with Release .NET 11 VSTest: 379 focused cases passed. Temporary native comparison matched all quantized coefficients, dequantized coefficients, and end positions in 5184 cases across scalar, 128-, 256-, and 512-bit paths. This is a motion-winner dependency checkpoint; production controller integration and encoder parity remain open.pull/2633/head
6 changed files with 679 additions and 0 deletions
@ -0,0 +1,131 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Six Labors Split License.
|
||||
|
|
||||
|
using System.Runtime.Intrinsics; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; |
||||
|
|
||||
|
internal static partial class Av1ForwardQuantizer |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Quantizes high-bit-depth magnitudes without the sixteen-bit saturation used for byte samples.
|
||||
|
/// </summary>
|
||||
|
public readonly struct HighBitDepthRegularQuantizationOperator : IRegularQuantizationOperator |
||||
|
{ |
||||
|
/// <inheritdoc/>
|
||||
|
public static Vector128<int> Quantize( |
||||
|
Vector128<int> coefficients, |
||||
|
Vector128<int> zeroBin, |
||||
|
Vector128<int> rounding, |
||||
|
Vector128<int> quantizer, |
||||
|
Vector128<int> shift, |
||||
|
Vector128<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector128<int> dequantizedCoefficients) |
||||
|
{ |
||||
|
Vector128<int> sign = coefficients >> 31; |
||||
|
Vector128<int> magnitude = Vector128.Abs(coefficients); |
||||
|
Vector128<int> mask = ~Vector128.GreaterThan(zeroBin, magnitude); |
||||
|
Vector128<int> rounded = magnitude + rounding; |
||||
|
|
||||
|
// Preserve signed products in two widened halves. The reciprocal correction can be negative;
|
||||
|
// an arithmetic Q16 shift restores its implicit leading bit before the second multiplication.
|
||||
|
Vector128<long> lower = Vector128.WidenLower(rounded); |
||||
|
Vector128<long> upper = Vector128.WidenUpper(rounded); |
||||
|
lower += (lower * Vector128.WidenLower(quantizer)) >> 16; |
||||
|
upper += (upper * Vector128.WidenUpper(quantizer)) >> 16; |
||||
|
lower = (lower * Vector128.WidenLower(shift)) >> (16 - logScale); |
||||
|
upper = (upper * Vector128.WidenUpper(shift)) >> (16 - logScale); |
||||
|
Vector128<int> quantizedMagnitude = Vector128.Narrow(lower, upper) & mask; |
||||
|
Vector128<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; |
||||
|
dequantizedCoefficients = (dequantizedMagnitude ^ sign) - sign; |
||||
|
return (quantizedMagnitude ^ sign) - sign; |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
public static Vector256<int> Quantize( |
||||
|
Vector256<int> coefficients, |
||||
|
Vector256<int> zeroBin, |
||||
|
Vector256<int> rounding, |
||||
|
Vector256<int> quantizer, |
||||
|
Vector256<int> shift, |
||||
|
Vector256<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector256<int> dequantizedCoefficients) |
||||
|
{ |
||||
|
Vector256<int> sign = coefficients >> 31; |
||||
|
Vector256<int> magnitude = Vector256.Abs(coefficients); |
||||
|
Vector256<int> mask = ~Vector256.GreaterThan(zeroBin, magnitude); |
||||
|
Vector256<int> rounded = magnitude + rounding; |
||||
|
|
||||
|
// Preserve signed products in two widened halves. The reciprocal correction can be negative;
|
||||
|
// an arithmetic Q16 shift restores its implicit leading bit before the second multiplication.
|
||||
|
Vector256<long> lower = Vector256.WidenLower(rounded); |
||||
|
Vector256<long> upper = Vector256.WidenUpper(rounded); |
||||
|
lower += (lower * Vector256.WidenLower(quantizer)) >> 16; |
||||
|
upper += (upper * Vector256.WidenUpper(quantizer)) >> 16; |
||||
|
lower = (lower * Vector256.WidenLower(shift)) >> (16 - logScale); |
||||
|
upper = (upper * Vector256.WidenUpper(shift)) >> (16 - logScale); |
||||
|
Vector256<int> quantizedMagnitude = Vector256.Narrow(lower, upper) & mask; |
||||
|
Vector256<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; |
||||
|
dequantizedCoefficients = (dequantizedMagnitude ^ sign) - sign; |
||||
|
return (quantizedMagnitude ^ sign) - sign; |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
public static Vector512<int> Quantize( |
||||
|
Vector512<int> coefficients, |
||||
|
Vector512<int> zeroBin, |
||||
|
Vector512<int> rounding, |
||||
|
Vector512<int> quantizer, |
||||
|
Vector512<int> shift, |
||||
|
Vector512<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector512<int> dequantizedCoefficients) |
||||
|
{ |
||||
|
Vector512<int> sign = coefficients >> 31; |
||||
|
Vector512<int> magnitude = Vector512.Abs(coefficients); |
||||
|
Vector512<int> mask = ~Vector512.GreaterThan(zeroBin, magnitude); |
||||
|
Vector512<int> rounded = magnitude + rounding; |
||||
|
|
||||
|
// Preserve signed products in two widened halves. The reciprocal correction can be negative;
|
||||
|
// an arithmetic Q16 shift restores its implicit leading bit before the second multiplication.
|
||||
|
Vector512<long> lower = Vector512.WidenLower(rounded); |
||||
|
Vector512<long> upper = Vector512.WidenUpper(rounded); |
||||
|
lower += (lower * Vector512.WidenLower(quantizer)) >> 16; |
||||
|
upper += (upper * Vector512.WidenUpper(quantizer)) >> 16; |
||||
|
lower = (lower * Vector512.WidenLower(shift)) >> (16 - logScale); |
||||
|
upper = (upper * Vector512.WidenUpper(shift)) >> (16 - logScale); |
||||
|
Vector512<int> quantizedMagnitude = Vector512.Narrow(lower, upper) & mask; |
||||
|
Vector512<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; |
||||
|
dequantizedCoefficients = (dequantizedMagnitude ^ sign) - sign; |
||||
|
return (quantizedMagnitude ^ sign) - sign; |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
public static int Quantize( |
||||
|
int coefficients, |
||||
|
int zeroBin, |
||||
|
int rounding, |
||||
|
int quantizer, |
||||
|
int shift, |
||||
|
int dequantizer, |
||||
|
int logScale, |
||||
|
out int dequantizedCoefficients) |
||||
|
{ |
||||
|
int sign = coefficients >> 31; |
||||
|
int magnitude = (coefficients ^ sign) - sign; |
||||
|
int quantizedMagnitude = 0; |
||||
|
if (magnitude >= zeroBin) |
||||
|
{ |
||||
|
long rounded = (long)magnitude + rounding; |
||||
|
long corrected = rounded + ((rounded * quantizer) >> 16); |
||||
|
quantizedMagnitude = (int)((corrected * shift) >> (16 - logScale)); |
||||
|
} |
||||
|
|
||||
|
int dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; |
||||
|
dequantizedCoefficients = (dequantizedMagnitude ^ sign) - sign; |
||||
|
return (quantizedMagnitude ^ sign) - sign; |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,175 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Six Labors Split License.
|
||||
|
|
||||
|
using System.Runtime.CompilerServices; |
||||
|
using System.Runtime.InteropServices; |
||||
|
using System.Runtime.Intrinsics; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; |
||||
|
|
||||
|
internal static partial class Av1ForwardQuantizer |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Quantizes a transform with the regular zero-bin and reciprocal-correction arithmetic.
|
||||
|
/// </summary>
|
||||
|
/// <param name="coefficients">The raster-order transformed coefficients.</param>
|
||||
|
/// <param name="quantizedCoefficients">The raster-order coding coefficients.</param>
|
||||
|
/// <param name="dequantizedCoefficients">The raster-order reconstruction coefficients.</param>
|
||||
|
/// <param name="transformSize">The transform dimensions.</param>
|
||||
|
/// <param name="transformType">The transform type selecting coefficient scan order.</param>
|
||||
|
/// <param name="qIndex">The base quantizer index.</param>
|
||||
|
/// <param name="dcDeltaQ">The DC index adjustment.</param>
|
||||
|
/// <param name="acDeltaQ">The AC index adjustment.</param>
|
||||
|
/// <param name="bitDepth">The coded sample precision.</param>
|
||||
|
/// <param name="sharpness">The encoder sharpness setting from zero through seven.</param>
|
||||
|
/// <returns>The one-based final nonzero scan position.</returns>
|
||||
|
public static ushort QuantizeRegular( |
||||
|
ReadOnlySpan<int> coefficients, |
||||
|
Span<int> quantizedCoefficients, |
||||
|
Span<int> dequantizedCoefficients, |
||||
|
Av1TransformSize transformSize, |
||||
|
Av1TransformType transformType, |
||||
|
int qIndex, |
||||
|
int dcDeltaQ, |
||||
|
int acDeltaQ, |
||||
|
Av1BitDepth bitDepth, |
||||
|
int sharpness) |
||||
|
=> bitDepth == Av1BitDepth.EightBit |
||||
|
? QuantizeRegular<RegularQuantizationOperator>( |
||||
|
coefficients, quantizedCoefficients, dequantizedCoefficients, transformSize, transformType, qIndex, dcDeltaQ, acDeltaQ, bitDepth, sharpness) |
||||
|
: QuantizeRegular<HighBitDepthRegularQuantizationOperator>( |
||||
|
coefficients, quantizedCoefficients, dequantizedCoefficients, transformSize, transformType, qIndex, dcDeltaQ, acDeltaQ, bitDepth, sharpness); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Traverses regular quantization with one DC coefficient followed by contiguous AC vectors and a scalar tail.
|
||||
|
/// </summary>
|
||||
|
private static ushort QuantizeRegular<TOperator>( |
||||
|
ReadOnlySpan<int> coefficients, |
||||
|
Span<int> quantizedCoefficients, |
||||
|
Span<int> dequantizedCoefficients, |
||||
|
Av1TransformSize transformSize, |
||||
|
Av1TransformType transformType, |
||||
|
int qIndex, |
||||
|
int dcDeltaQ, |
||||
|
int acDeltaQ, |
||||
|
Av1BitDepth bitDepth, |
||||
|
int sharpness) |
||||
|
where TOperator : struct, IRegularQuantizationOperator |
||||
|
{ |
||||
|
int count = transformSize.GetAdjusted().GetSize2d(); |
||||
|
int logScale = transformSize.GetScale(); |
||||
|
int zeroBinFactor = Av1InverseTransformMath.GetQzbinFactor(qIndex, bitDepth); |
||||
|
int roundingFactor = qIndex == 0 ? 64 : sharpness == 0 ? 48 : 64 - (16 * (7 - sharpness) / 7); |
||||
|
int dcDequantizer = Av1QuantizationLookup.GetDcQuant(qIndex, dcDeltaQ, bitDepth); |
||||
|
int acDequantizer = Av1QuantizationLookup.GetAcQuant(qIndex, acDeltaQ, bitDepth); |
||||
|
Av1InverseTransformMath.InvertQuantization(out int dcQuantizer, out int dcShift, dcDequantizer); |
||||
|
Av1InverseTransformMath.InvertQuantization(out int acQuantizer, out int acShift, acDequantizer); |
||||
|
|
||||
|
// Zero-bin constants round twice: once from Q7 and once for the transform scale. Rounding
|
||||
|
// constants first truncate from Q7, then round for that same scale. Combining either pair of
|
||||
|
// shifts would change coefficients at the quantization boundary.
|
||||
|
int dcZeroBin = RoundPowerOfTwo(RoundPowerOfTwo(zeroBinFactor * dcDequantizer, 7), logScale); |
||||
|
int acZeroBin = RoundPowerOfTwo(RoundPowerOfTwo(zeroBinFactor * acDequantizer, 7), logScale); |
||||
|
int dcRounding = RoundPowerOfTwo((roundingFactor * dcDequantizer) >> 7, logScale); |
||||
|
int acRounding = RoundPowerOfTwo((roundingFactor * acDequantizer) >> 7, logScale); |
||||
|
ref int sourceBase = ref MemoryMarshal.GetReference(coefficients); |
||||
|
ref int quantizedBase = ref MemoryMarshal.GetReference(quantizedCoefficients); |
||||
|
ref int dequantizedBase = ref MemoryMarshal.GetReference(dequantizedCoefficients); |
||||
|
|
||||
|
// DC has different constants. Processing it once leaves a uniform AC traversal with no lane masks
|
||||
|
// for DC and no scratch coefficient copy; every output position is overwritten on each candidate.
|
||||
|
quantizedBase = TOperator.Quantize( |
||||
|
sourceBase, dcZeroBin, dcRounding, dcQuantizer, dcShift, dcDequantizer, logScale, out dequantizedBase); |
||||
|
|
||||
|
int index = 1; |
||||
|
|
||||
|
if (Vector512.IsHardwareAccelerated) |
||||
|
{ |
||||
|
Vector512<int> zeroBin = Vector512.Create(acZeroBin); |
||||
|
Vector512<int> rounding = Vector512.Create(acRounding); |
||||
|
Vector512<int> quantizer = Vector512.Create(acQuantizer); |
||||
|
Vector512<int> shift = Vector512.Create(acShift); |
||||
|
Vector512<int> dequantizer = Vector512.Create(acDequantizer); |
||||
|
|
||||
|
// Each lane owns one contiguous coefficient. Narrower tiers resume at the first unread
|
||||
|
// coefficient so unaligned starts and vector tails need neither padding nor overlapping stores.
|
||||
|
for (; index <= count - Vector512<int>.Count; index += Vector512<int>.Count) |
||||
|
{ |
||||
|
Vector512<int> values = Vector512.LoadUnsafe(ref sourceBase, (nuint)index); |
||||
|
Vector512<int> quantized = TOperator.Quantize( |
||||
|
values, zeroBin, rounding, quantizer, shift, dequantizer, logScale, out Vector512<int> dequantized); |
||||
|
|
||||
|
quantized.StoreUnsafe(ref quantizedBase, (nuint)index); |
||||
|
dequantized.StoreUnsafe(ref dequantizedBase, (nuint)index); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
if (Vector256.IsHardwareAccelerated) |
||||
|
{ |
||||
|
Vector256<int> zeroBin = Vector256.Create(acZeroBin); |
||||
|
Vector256<int> rounding = Vector256.Create(acRounding); |
||||
|
Vector256<int> quantizer = Vector256.Create(acQuantizer); |
||||
|
Vector256<int> shift = Vector256.Create(acShift); |
||||
|
Vector256<int> dequantizer = Vector256.Create(acDequantizer); |
||||
|
|
||||
|
// Each lane owns one contiguous coefficient. Narrower tiers resume at the first unread
|
||||
|
// coefficient so unaligned starts and vector tails need neither padding nor overlapping stores.
|
||||
|
for (; index <= count - Vector256<int>.Count; index += Vector256<int>.Count) |
||||
|
{ |
||||
|
Vector256<int> values = Vector256.LoadUnsafe(ref sourceBase, (nuint)index); |
||||
|
Vector256<int> quantized = TOperator.Quantize( |
||||
|
values, zeroBin, rounding, quantizer, shift, dequantizer, logScale, out Vector256<int> dequantized); |
||||
|
|
||||
|
quantized.StoreUnsafe(ref quantizedBase, (nuint)index); |
||||
|
dequantized.StoreUnsafe(ref dequantizedBase, (nuint)index); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
if (Vector128.IsHardwareAccelerated) |
||||
|
{ |
||||
|
Vector128<int> zeroBin = Vector128.Create(acZeroBin); |
||||
|
Vector128<int> rounding = Vector128.Create(acRounding); |
||||
|
Vector128<int> quantizer = Vector128.Create(acQuantizer); |
||||
|
Vector128<int> shift = Vector128.Create(acShift); |
||||
|
Vector128<int> dequantizer = Vector128.Create(acDequantizer); |
||||
|
|
||||
|
// Each lane owns one contiguous coefficient. Narrower tiers resume at the first unread
|
||||
|
// coefficient so unaligned starts and vector tails need neither padding nor overlapping stores.
|
||||
|
for (; index <= count - Vector128<int>.Count; index += Vector128<int>.Count) |
||||
|
{ |
||||
|
Vector128<int> values = Vector128.LoadUnsafe(ref sourceBase, (nuint)index); |
||||
|
Vector128<int> quantized = TOperator.Quantize( |
||||
|
values, zeroBin, rounding, quantizer, shift, dequantizer, logScale, out Vector128<int> dequantized); |
||||
|
|
||||
|
quantized.StoreUnsafe(ref quantizedBase, (nuint)index); |
||||
|
dequantized.StoreUnsafe(ref dequantizedBase, (nuint)index); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
for (; index < count; index++) |
||||
|
{ |
||||
|
Unsafe.Add(ref quantizedBase, index) = TOperator.Quantize( |
||||
|
Unsafe.Add(ref sourceBase, index), |
||||
|
acZeroBin, |
||||
|
acRounding, |
||||
|
acQuantizer, |
||||
|
acShift, |
||||
|
acDequantizer, |
||||
|
logScale, |
||||
|
out Unsafe.Add(ref dequantizedBase, index)); |
||||
|
} |
||||
|
|
||||
|
// Coding and reconstruction retain raster order. Only the end position is reduced in scan order.
|
||||
|
ReadOnlySpan<short> scan = Av1ScanOrderConstants.GetScanOrder(transformSize, transformType).Scan; |
||||
|
for (int scanIndex = count - 1; scanIndex >= 0; scanIndex--) |
||||
|
{ |
||||
|
if (Unsafe.Add(ref quantizedBase, scan[scanIndex]) != 0) |
||||
|
{ |
||||
|
return (ushort)(scanIndex + 1); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
return 0; |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,103 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Six Labors Split License.
|
||||
|
|
||||
|
using System.Runtime.Intrinsics; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; |
||||
|
|
||||
|
internal static partial class Av1ForwardQuantizer |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Applies regular quantization while preserving the precision-specific rounded-magnitude contract.
|
||||
|
/// </summary>
|
||||
|
public interface IRegularQuantizationOperator |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Applies the zero bin, corrected reciprocal, dequantization, and coefficient sign.
|
||||
|
/// </summary>
|
||||
|
/// <param name="coefficients">The signed transform coefficients.</param>
|
||||
|
/// <param name="zeroBin">The inclusive magnitude threshold after transform scaling.</param>
|
||||
|
/// <param name="rounding">The rounded magnitude adjustment.</param>
|
||||
|
/// <param name="quantizer">The signed reciprocal correction below its implicit leading bit.</param>
|
||||
|
/// <param name="shift">The power-of-two reciprocal scale.</param>
|
||||
|
/// <param name="dequantizer">The reconstruction quantizer.</param>
|
||||
|
/// <param name="logScale">The transform coefficient scale.</param>
|
||||
|
/// <param name="dequantizedCoefficients">The signed reconstruction coefficients.</param>
|
||||
|
/// <returns>The signed coding coefficients.</returns>
|
||||
|
static abstract Vector128<int> Quantize( |
||||
|
Vector128<int> coefficients, |
||||
|
Vector128<int> zeroBin, |
||||
|
Vector128<int> rounding, |
||||
|
Vector128<int> quantizer, |
||||
|
Vector128<int> shift, |
||||
|
Vector128<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector128<int> dequantizedCoefficients); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Applies the zero bin, corrected reciprocal, dequantization, and coefficient sign.
|
||||
|
/// </summary>
|
||||
|
/// <param name="coefficients">The signed transform coefficients.</param>
|
||||
|
/// <param name="zeroBin">The inclusive magnitude threshold after transform scaling.</param>
|
||||
|
/// <param name="rounding">The rounded magnitude adjustment.</param>
|
||||
|
/// <param name="quantizer">The signed reciprocal correction below its implicit leading bit.</param>
|
||||
|
/// <param name="shift">The power-of-two reciprocal scale.</param>
|
||||
|
/// <param name="dequantizer">The reconstruction quantizer.</param>
|
||||
|
/// <param name="logScale">The transform coefficient scale.</param>
|
||||
|
/// <param name="dequantizedCoefficients">The signed reconstruction coefficients.</param>
|
||||
|
/// <returns>The signed coding coefficients.</returns>
|
||||
|
static abstract Vector256<int> Quantize( |
||||
|
Vector256<int> coefficients, |
||||
|
Vector256<int> zeroBin, |
||||
|
Vector256<int> rounding, |
||||
|
Vector256<int> quantizer, |
||||
|
Vector256<int> shift, |
||||
|
Vector256<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector256<int> dequantizedCoefficients); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Applies the zero bin, corrected reciprocal, dequantization, and coefficient sign.
|
||||
|
/// </summary>
|
||||
|
/// <param name="coefficients">The signed transform coefficients.</param>
|
||||
|
/// <param name="zeroBin">The inclusive magnitude threshold after transform scaling.</param>
|
||||
|
/// <param name="rounding">The rounded magnitude adjustment.</param>
|
||||
|
/// <param name="quantizer">The signed reciprocal correction below its implicit leading bit.</param>
|
||||
|
/// <param name="shift">The power-of-two reciprocal scale.</param>
|
||||
|
/// <param name="dequantizer">The reconstruction quantizer.</param>
|
||||
|
/// <param name="logScale">The transform coefficient scale.</param>
|
||||
|
/// <param name="dequantizedCoefficients">The signed reconstruction coefficients.</param>
|
||||
|
/// <returns>The signed coding coefficients.</returns>
|
||||
|
static abstract Vector512<int> Quantize( |
||||
|
Vector512<int> coefficients, |
||||
|
Vector512<int> zeroBin, |
||||
|
Vector512<int> rounding, |
||||
|
Vector512<int> quantizer, |
||||
|
Vector512<int> shift, |
||||
|
Vector512<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector512<int> dequantizedCoefficients); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Applies the zero bin, corrected reciprocal, dequantization, and coefficient sign.
|
||||
|
/// </summary>
|
||||
|
/// <param name="coefficients">The signed transform coefficients.</param>
|
||||
|
/// <param name="zeroBin">The inclusive magnitude threshold after transform scaling.</param>
|
||||
|
/// <param name="rounding">The rounded magnitude adjustment.</param>
|
||||
|
/// <param name="quantizer">The signed reciprocal correction below its implicit leading bit.</param>
|
||||
|
/// <param name="shift">The power-of-two reciprocal scale.</param>
|
||||
|
/// <param name="dequantizer">The reconstruction quantizer.</param>
|
||||
|
/// <param name="logScale">The transform coefficient scale.</param>
|
||||
|
/// <param name="dequantizedCoefficients">The signed reconstruction coefficients.</param>
|
||||
|
/// <returns>The signed coding coefficients.</returns>
|
||||
|
static abstract int Quantize( |
||||
|
int coefficients, |
||||
|
int zeroBin, |
||||
|
int rounding, |
||||
|
int quantizer, |
||||
|
int shift, |
||||
|
int dequantizer, |
||||
|
int logScale, |
||||
|
out int dequantizedCoefficients); |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,116 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Six Labors Split License.
|
||||
|
|
||||
|
using System.Runtime.Intrinsics; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; |
||||
|
|
||||
|
internal static partial class Av1ForwardQuantizer |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Quantizes byte-source coefficients with saturated sixteen-bit rounded magnitudes.
|
||||
|
/// </summary>
|
||||
|
public readonly struct RegularQuantizationOperator : IRegularQuantizationOperator |
||||
|
{ |
||||
|
/// <inheritdoc/>
|
||||
|
public static Vector128<int> Quantize( |
||||
|
Vector128<int> coefficients, |
||||
|
Vector128<int> zeroBin, |
||||
|
Vector128<int> rounding, |
||||
|
Vector128<int> quantizer, |
||||
|
Vector128<int> shift, |
||||
|
Vector128<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector128<int> dequantizedCoefficients) |
||||
|
{ |
||||
|
Vector128<int> sign = coefficients >> 31; |
||||
|
Vector128<int> magnitude = Vector128.Abs(coefficients); |
||||
|
Vector128<int> mask = ~Vector128.GreaterThan(zeroBin, magnitude); |
||||
|
Vector128<int> rounded = Vector128.Min(magnitude + rounding, Vector128.Create((int)short.MaxValue)); |
||||
|
|
||||
|
// Saturation bounds both signed products to 32-bit lanes. The first Q16 multiplication
|
||||
|
// restores the reciprocal's implicit leading bit; the second removes its power-of-two scale.
|
||||
|
Vector128<int> corrected = rounded + ((rounded * quantizer) >> 16); |
||||
|
Vector128<int> quantizedMagnitude = ((corrected * shift) >> (16 - logScale)) & mask; |
||||
|
Vector128<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; |
||||
|
dequantizedCoefficients = (dequantizedMagnitude ^ sign) - sign; |
||||
|
return (quantizedMagnitude ^ sign) - sign; |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
public static Vector256<int> Quantize( |
||||
|
Vector256<int> coefficients, |
||||
|
Vector256<int> zeroBin, |
||||
|
Vector256<int> rounding, |
||||
|
Vector256<int> quantizer, |
||||
|
Vector256<int> shift, |
||||
|
Vector256<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector256<int> dequantizedCoefficients) |
||||
|
{ |
||||
|
Vector256<int> sign = coefficients >> 31; |
||||
|
Vector256<int> magnitude = Vector256.Abs(coefficients); |
||||
|
Vector256<int> mask = ~Vector256.GreaterThan(zeroBin, magnitude); |
||||
|
Vector256<int> rounded = Vector256.Min(magnitude + rounding, Vector256.Create((int)short.MaxValue)); |
||||
|
|
||||
|
// Saturation bounds both signed products to 32-bit lanes. The first Q16 multiplication
|
||||
|
// restores the reciprocal's implicit leading bit; the second removes its power-of-two scale.
|
||||
|
Vector256<int> corrected = rounded + ((rounded * quantizer) >> 16); |
||||
|
Vector256<int> quantizedMagnitude = ((corrected * shift) >> (16 - logScale)) & mask; |
||||
|
Vector256<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; |
||||
|
dequantizedCoefficients = (dequantizedMagnitude ^ sign) - sign; |
||||
|
return (quantizedMagnitude ^ sign) - sign; |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
public static Vector512<int> Quantize( |
||||
|
Vector512<int> coefficients, |
||||
|
Vector512<int> zeroBin, |
||||
|
Vector512<int> rounding, |
||||
|
Vector512<int> quantizer, |
||||
|
Vector512<int> shift, |
||||
|
Vector512<int> dequantizer, |
||||
|
int logScale, |
||||
|
out Vector512<int> dequantizedCoefficients) |
||||
|
{ |
||||
|
Vector512<int> sign = coefficients >> 31; |
||||
|
Vector512<int> magnitude = Vector512.Abs(coefficients); |
||||
|
Vector512<int> mask = ~Vector512.GreaterThan(zeroBin, magnitude); |
||||
|
Vector512<int> rounded = Vector512.Min(magnitude + rounding, Vector512.Create((int)short.MaxValue)); |
||||
|
|
||||
|
// Saturation bounds both signed products to 32-bit lanes. The first Q16 multiplication
|
||||
|
// restores the reciprocal's implicit leading bit; the second removes its power-of-two scale.
|
||||
|
Vector512<int> corrected = rounded + ((rounded * quantizer) >> 16); |
||||
|
Vector512<int> quantizedMagnitude = ((corrected * shift) >> (16 - logScale)) & mask; |
||||
|
Vector512<int> dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; |
||||
|
dequantizedCoefficients = (dequantizedMagnitude ^ sign) - sign; |
||||
|
return (quantizedMagnitude ^ sign) - sign; |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
public static int Quantize( |
||||
|
int coefficients, |
||||
|
int zeroBin, |
||||
|
int rounding, |
||||
|
int quantizer, |
||||
|
int shift, |
||||
|
int dequantizer, |
||||
|
int logScale, |
||||
|
out int dequantizedCoefficients) |
||||
|
{ |
||||
|
int sign = coefficients >> 31; |
||||
|
int magnitude = (coefficients ^ sign) - sign; |
||||
|
int quantizedMagnitude = 0; |
||||
|
if (magnitude >= zeroBin) |
||||
|
{ |
||||
|
int rounded = Math.Min(magnitude + rounding, short.MaxValue); |
||||
|
int corrected = rounded + ((rounded * quantizer) >> 16); |
||||
|
quantizedMagnitude = (corrected * shift) >> (16 - logScale); |
||||
|
} |
||||
|
|
||||
|
int dequantizedMagnitude = (quantizedMagnitude * dequantizer) >> logScale; |
||||
|
dequantizedCoefficients = (dequantizedMagnitude ^ sign) - sign; |
||||
|
return (quantizedMagnitude ^ sign) - sign; |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,148 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Six Labors Split License.
|
||||
|
|
||||
|
using System.Runtime.InteropServices; |
||||
|
using System.Runtime.Intrinsics; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline.Quantizers; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1.Transform; |
||||
|
using SixLabors.ImageSharp.Tests.TestUtilities; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Tests.Formats.Heif.Av1; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Verifies regular quantization's precision, scan position, and buffer boundaries across hardware paths.
|
||||
|
/// </summary>
|
||||
|
[Trait("Format", "Avif")] |
||||
|
public class Av1RegularQuantizerTests |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Exercises regular quantization under each available vector width and scalar fallback.
|
||||
|
/// </summary>
|
||||
|
[Fact] |
||||
|
public void RegularQuantizationPreservesCoefficientContracts() |
||||
|
=> FeatureTestRunner.RunWithHwIntrinsicsFeature( |
||||
|
ValidateQuantization, |
||||
|
HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Exports inputs and results for independent native comparison while checking observable storage contracts.
|
||||
|
/// </summary>
|
||||
|
private static void ValidateQuantization() |
||||
|
{ |
||||
|
string vectorWidth = Vector512.IsHardwareAccelerated ? "512" |
||||
|
: Vector256.IsHardwareAccelerated ? "256" |
||||
|
: Vector128.IsHardwareAccelerated ? "128" : "0"; |
||||
|
|
||||
|
string directory = Path.Combine(TestEnvironment.ActualOutputDirectoryFullPath, "Heif", "Av1", "RegularQuantization", vectorWidth); |
||||
|
|
||||
|
Directory.CreateDirectory(directory); |
||||
|
foreach (Av1BitDepth bitDepth in new[] { Av1BitDepth.EightBit, Av1BitDepth.TenBit, Av1BitDepth.TwelveBit }) |
||||
|
{ |
||||
|
foreach (Av1TransformSize size in new[] |
||||
|
{ |
||||
|
Av1TransformSize.Size4x4, Av1TransformSize.Size8x8, Av1TransformSize.Size16x16, |
||||
|
Av1TransformSize.Size32x32, Av1TransformSize.Size64x64, Av1TransformSize.Size8x16 |
||||
|
}) |
||||
|
{ |
||||
|
int count = size.GetAdjusted().GetSize2d(); |
||||
|
int[] input = new int[count + 2]; |
||||
|
int[] quantized = new int[count + 2]; |
||||
|
int[] dequantized = new int[count + 2]; |
||||
|
ReadOnlySpan<short> scan = Av1ScanOrderConstants.GetScanOrder(size, Av1TransformType.DctDct).Scan; |
||||
|
foreach (int qIndex in new[] { 0, 1, 10, 90, 200, 255 }) |
||||
|
{ |
||||
|
foreach (int sharpness in new[] { 0, 3, 7 }) |
||||
|
{ |
||||
|
for (int pattern = 0; pattern < 4; pattern++) |
||||
|
{ |
||||
|
int dcDelta = pattern == 1 ? -20 : pattern == 2 ? 17 : 0; |
||||
|
int acDelta = pattern == 1 ? 13 : pattern == 2 ? -11 : 0; |
||||
|
int bits = bitDepth.GetBitCount(); |
||||
|
int maximum = (1 << (bits + 7)) - 1; |
||||
|
uint state = (uint)(qIndex + 123); |
||||
|
Array.Fill(input, int.MinValue); |
||||
|
Array.Fill(quantized, int.MinValue); |
||||
|
Array.Fill(dequantized, int.MinValue); |
||||
|
for (int index = 0; index < count; index++) |
||||
|
{ |
||||
|
state = unchecked((state * 1664525) + 1013904223); |
||||
|
int magnitude = (int)(state % (uint)maximum); |
||||
|
if (pattern == 1) |
||||
|
{ |
||||
|
int divisor = index == 0 |
||||
|
? Av1QuantizationLookup.GetDcQuant(qIndex, dcDelta, bitDepth) |
||||
|
: Av1QuantizationLookup.GetAcQuant(qIndex, acDelta, bitDepth); |
||||
|
|
||||
|
int threshold = ((Av1InverseTransformMath.GetQzbinFactor(qIndex, bitDepth) * divisor) + 64) >> 7; |
||||
|
int scale = size.GetScale(); |
||||
|
threshold = scale == 0 ? threshold : (threshold + (1 << (scale - 1))) >> scale; |
||||
|
magnitude = threshold + (index % 3) - 1; |
||||
|
} |
||||
|
else if (pattern == 2) |
||||
|
{ |
||||
|
magnitude = maximum; |
||||
|
} |
||||
|
else if (pattern == 3) |
||||
|
{ |
||||
|
// A sparse final scan position exposes stale output coefficients across candidate reuse.
|
||||
|
magnitude = index == scan[count / 3] ? 97 : 0; |
||||
|
} |
||||
|
|
||||
|
input[index + 1] = (index & 1) == 0 ? magnitude : -magnitude; |
||||
|
} |
||||
|
|
||||
|
int[] original = (int[])input.Clone(); |
||||
|
ushort end = Av1ForwardQuantizer.QuantizeRegular( |
||||
|
input.AsSpan(1, count), |
||||
|
quantized.AsSpan(1, count), |
||||
|
dequantized.AsSpan(1, count), |
||||
|
size, |
||||
|
Av1TransformType.DctDct, |
||||
|
qIndex, |
||||
|
dcDelta, |
||||
|
acDelta, |
||||
|
bitDepth, |
||||
|
sharpness); |
||||
|
|
||||
|
Assert.Equal(original, input); |
||||
|
Assert.Equal(int.MinValue, quantized[0]); |
||||
|
Assert.Equal(int.MinValue, quantized[^1]); |
||||
|
Assert.Equal(int.MinValue, dequantized[0]); |
||||
|
Assert.Equal(int.MinValue, dequantized[^1]); |
||||
|
int expectedEnd = 0; |
||||
|
for (int index = 0; index < count; index++) |
||||
|
{ |
||||
|
int level = quantized[index + 1]; |
||||
|
int divisor = index == 0 |
||||
|
? Av1QuantizationLookup.GetDcQuant(qIndex, dcDelta, bitDepth) |
||||
|
: Av1QuantizationLookup.GetAcQuant(qIndex, acDelta, bitDepth); |
||||
|
|
||||
|
// Signed dequantization truncates the magnitude before restoring sign.
|
||||
|
int restored = (Math.Abs(level) * divisor) >> size.GetScale(); |
||||
|
Assert.Equal(level < 0 ? -restored : restored, dequantized[index + 1]); |
||||
|
if (quantized[scan[index] + 1] != 0) |
||||
|
{ |
||||
|
expectedEnd = index + 1; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
Assert.Equal(expectedEnd, end); |
||||
|
using BinaryWriter output = new(File.Create(Path.Combine( |
||||
|
directory, $"{bits}-{(int)size}-{qIndex}-{sharpness}-{pattern}.bin"))); |
||||
|
|
||||
|
foreach (int value in new[] { bits, (int)size, qIndex, dcDelta, acDelta, sharpness, count, end }) |
||||
|
{ |
||||
|
output.Write(value); |
||||
|
} |
||||
|
|
||||
|
output.Write(MemoryMarshal.AsBytes(input.AsSpan(1, count))); |
||||
|
output.Write(MemoryMarshal.AsBytes(quantized.AsSpan(1, count))); |
||||
|
output.Write(MemoryMarshal.AsBytes(dequantized.AsSpan(1, count))); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
Loading…
Reference in new issue