Browse Source

Fix alpha association and SIMD rounding parity

Refactors `Color` to track both exposed and stored alpha representations, adds `ToScaledVector4(PixelAlphaRepresentation)`, and adjusts pixel conversion paths so associated formats preserve canonical values without unnecessary unpremultiply/reassociate loss. This also updates equality/hash behavior to compare canonical scaled values.

SIMD helpers are split into `MultiplyAddEstimate` vs `FusedMultiplyAdd`, with byte-to-float normalization updated to match scalar rounding exactly across vector widths. JPEG converters, resize kernels, and Porter-Duff/associated-alpha blending paths were updated to use the appropriate helper for either fast estimate or strict fused rounding semantics. Tests were expanded substantially (including exhaustive component/alpha cases and fused-order checks), incidental test cleanup was applied, benchmark comment snapshots were refreshed, and one black/white reference output image was updated.
pull/3154/head
James Jackson-South 3 months ago
parent
commit
4b07920c7c
  1. 190
      src/ImageSharp/Color/Color.cs
  2. 3
      src/ImageSharp/Common/Helpers/SimdUtils.Convert.cs
  3. 53
      src/ImageSharp/Common/Helpers/SimdUtils.HwIntrinsics.cs
  4. 54
      src/ImageSharp/Common/Helpers/Vector128Utilities.cs
  5. 45
      src/ImageSharp/Common/Helpers/Vector256Utilities.cs
  6. 50
      src/ImageSharp/Common/Helpers/Vector512Utilities.cs
  7. 2
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector128.cs
  8. 2
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector256.cs
  9. 2
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector512.cs
  10. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector128.cs
  11. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector256.cs
  12. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector512.cs
  13. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector128.cs
  14. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector256.cs
  15. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector512.cs
  16. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector128.cs
  17. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector256.cs
  18. 12
      src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector512.cs
  19. 8
      src/ImageSharp/Formats/Jpeg/Components/FloatingPointDCT.Vector256.cs
  20. 16
      src/ImageSharp/PixelFormats/PixelBlenders/AssociatedAlphaPorterDuffFunctions.cs
  21. 26
      src/ImageSharp/PixelFormats/PixelBlenders/PorterDuffFunctions.cs
  22. 2
      src/ImageSharp/PixelFormats/PixelImplementations/Abgr32.cs
  23. 4
      src/ImageSharp/PixelFormats/PixelImplementations/Abgr32P.cs
  24. 2
      src/ImageSharp/PixelFormats/PixelImplementations/Argb32.cs
  25. 4
      src/ImageSharp/PixelFormats/PixelImplementations/Argb32P.cs
  26. 2
      src/ImageSharp/PixelFormats/PixelImplementations/Bgra32.cs
  27. 4
      src/ImageSharp/PixelFormats/PixelImplementations/Bgra32P.cs
  28. 5
      src/ImageSharp/PixelFormats/PixelImplementations/NormalizedByte4P.cs
  29. 2
      src/ImageSharp/PixelFormats/PixelImplementations/Rgba32.cs
  30. 4
      src/ImageSharp/PixelFormats/PixelImplementations/Rgba32P.cs
  31. 8
      src/ImageSharp/Processing/Processors/Transforms/Resize/ResizeKernel.cs
  32. 16
      tests/ImageSharp.Benchmarks/Bulk/FromVector4.cs
  33. 16
      tests/ImageSharp.Benchmarks/Bulk/ToVector4_Bgra32.cs
  34. 31
      tests/ImageSharp.Tests/Color/ColorTests.cs
  35. 32
      tests/ImageSharp.Tests/Common/SimdUtilsTests.cs
  36. 7
      tests/ImageSharp.Tests/Formats/Icon/Cur/CurEncoderTests.cs
  37. 2
      tests/ImageSharp.Tests/Formats/Qoi/ImageExtensionsTest.cs
  38. 312
      tests/ImageSharp.Tests/PixelFormats/AssociatedAlphaPixelTests.cs
  39. 7
      tests/ImageSharp.Tests/PixelFormats/PixelBlenderTests.cs
  40. 4
      tests/Images/External/ReferenceOutput/Filters/BlackWhiteTest/ApplyBlackWhiteFilter_Rgba32_TestPattern48x48.png

190
src/ImageSharp/Color/Color.cs

@ -20,45 +20,23 @@ namespace SixLabors.ImageSharp;
public readonly partial struct Color : IEquatable<Color>
{
private readonly Vector4 data;
private readonly Vector4 unassociatedData;
private readonly IPixel? boxedHighPrecisionPixel;
private readonly bool isAssociated;
private readonly bool dataIsAssociated;
/// <summary>
/// Initializes a new instance of the <see cref="Color"/> struct.
/// </summary>
/// <param name="vector">The <see cref="Vector4"/> containing the color information.</param>
/// <param name="alphaRepresentation">The alpha representation of <paramref name="vector"/>.</param>
/// <param name="alphaRepresentation">The alpha representation exposed by the color.</param>
/// <param name="dataAlphaRepresentation">The alpha representation of <paramref name="vector"/>.</param>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
private Color(Vector4 vector, PixelAlphaRepresentation alphaRepresentation)
{
Vector4 clamped = Numerics.Clamp(vector, Vector4.Zero, Vector4.One);
Vector4 unassociated = clamped;
if (alphaRepresentation == PixelAlphaRepresentation.Associated)
{
Numerics.UnPremultiply(ref unassociated);
}
this.data = clamped;
this.unassociatedData = unassociated;
this.boxedHighPrecisionPixel = null;
this.isAssociated = alphaRepresentation == PixelAlphaRepresentation.Associated;
}
/// <summary>
/// Initializes a new instance of the <see cref="Color"/> struct.
/// </summary>
/// <param name="vector">The <see cref="Vector4"/> containing the color information.</param>
/// <param name="unassociatedVector">The unassociated representation of <paramref name="vector"/>.</param>
/// <param name="alphaRepresentation">The alpha representation of <paramref name="vector"/>.</param>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
private Color(Vector4 vector, Vector4 unassociatedVector, PixelAlphaRepresentation alphaRepresentation)
private Color(Vector4 vector, PixelAlphaRepresentation alphaRepresentation, PixelAlphaRepresentation dataAlphaRepresentation)
{
this.data = Numerics.Clamp(vector, Vector4.Zero, Vector4.One);
this.unassociatedData = Numerics.Clamp(unassociatedVector, Vector4.Zero, Vector4.One);
this.boxedHighPrecisionPixel = null;
this.isAssociated = alphaRepresentation == PixelAlphaRepresentation.Associated;
this.dataIsAssociated = dataAlphaRepresentation == PixelAlphaRepresentation.Associated;
}
/// <summary>
@ -71,8 +49,8 @@ public readonly partial struct Color : IEquatable<Color>
{
this.boxedHighPrecisionPixel = pixel;
this.data = default;
this.unassociatedData = default;
this.isAssociated = alphaRepresentation == PixelAlphaRepresentation.Associated;
this.dataIsAssociated = this.isAssociated;
}
/// <summary>
@ -118,14 +96,22 @@ public readonly partial struct Color : IEquatable<Color>
// Avoid boxing in case we can convert to Vector4 safely and efficiently
PixelTypeInfo info = TPixel.GetPixelTypeInfo();
if (info.ComponentInfo.HasValue && info.ComponentInfo.Value.GetMaximumComponentPrecision() <= (int)PixelComponentBitDepth.Bit32)
if (info.ComponentInfo.HasValue)
{
Vector4 vector = source.ToScaledVector4();
Vector4 unassociated = info.AlphaRepresentation == PixelAlphaRepresentation.Associated
? PixelOperations<TPixel>.Instance.ToUnassociatedScaledVector4(source)
: vector;
int maximumComponentPrecision = info.ComponentInfo.Value.GetMaximumComponentPrecision();
return new Color(vector, unassociated, info.AlphaRepresentation);
if (maximumComponentPrecision <= (int)PixelComponentBitDepth.Bit32)
{
if (info.AlphaRepresentation == PixelAlphaRepresentation.Associated && maximumComponentPrecision <= (int)PixelComponentBitDepth.Bit8)
{
// Associated formats with at most eight bits per component can be canonicalized without loss by their pixel-specific conversion.
// Higher-precision formats retain their associated values because unassociation can lose representable data.
Vector4 vector = PixelOperations<TPixel>.Instance.ToUnassociatedScaledVector4(source);
return new Color(vector, info.AlphaRepresentation, PixelAlphaRepresentation.Unassociated);
}
return new Color(source.ToScaledVector4(), info.AlphaRepresentation, info.AlphaRepresentation);
}
}
return new Color(source, info.AlphaRepresentation);
@ -137,7 +123,8 @@ public readonly partial struct Color : IEquatable<Color>
/// <param name="source">The unassociated vector to load the color from.</param>
/// <returns>The <see cref="Color"/>.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Color FromScaledVector(Vector4 source) => new(source, PixelAlphaRepresentation.Unassociated);
public static Color FromScaledVector(Vector4 source)
=> new(source, PixelAlphaRepresentation.Unassociated, PixelAlphaRepresentation.Unassociated);
/// <summary>
/// Creates a <see cref="Color"/> from a generic scaled <see cref="Vector4"/> with the specified alpha representation.
@ -146,7 +133,8 @@ public readonly partial struct Color : IEquatable<Color>
/// <param name="alphaRepresentation">The alpha representation of <paramref name="source"/>.</param>
/// <returns>The <see cref="Color"/>.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Color FromScaledVector(Vector4 source, PixelAlphaRepresentation alphaRepresentation) => new(source, alphaRepresentation);
public static Color FromScaledVector(Vector4 source, PixelAlphaRepresentation alphaRepresentation)
=> new(source, alphaRepresentation, alphaRepresentation);
/// <summary>
/// Bulk converts a span of generic scaled <see cref="Vector4"/> to a span of <see cref="Color"/>.
@ -190,23 +178,38 @@ public readonly partial struct Color : IEquatable<Color>
// Avoid boxing in case we can convert to Vector4 safely and efficiently
PixelTypeInfo info = TPixel.GetPixelTypeInfo();
if (info.ComponentInfo.HasValue && info.ComponentInfo.Value.GetMaximumComponentPrecision() <= (int)PixelComponentBitDepth.Bit32)
if (info.ComponentInfo.HasValue)
{
bool isAssociated = info.AlphaRepresentation == PixelAlphaRepresentation.Associated;
int maximumComponentPrecision = info.ComponentInfo.Value.GetMaximumComponentPrecision();
for (int i = 0; i < source.Length; i++)
if (maximumComponentPrecision <= (int)PixelComponentBitDepth.Bit32)
{
Vector4 vector = source[i].ToScaledVector4();
Vector4 unassociated = isAssociated ? PixelOperations<TPixel>.Instance.ToUnassociatedScaledVector4(source[i]) : vector;
destination[i] = new Color(vector, unassociated, info.AlphaRepresentation);
if (info.AlphaRepresentation == PixelAlphaRepresentation.Associated && maximumComponentPrecision <= (int)PixelComponentBitDepth.Bit8)
{
// Match the scalar conversion by retaining exact unassociated values from the format-specific operation.
PixelOperations<TPixel> operations = PixelOperations<TPixel>.Instance;
for (int i = 0; i < source.Length; i++)
{
Vector4 vector = operations.ToUnassociatedScaledVector4(source[i]);
destination[i] = new Color(vector, info.AlphaRepresentation, PixelAlphaRepresentation.Unassociated);
}
return;
}
for (int i = 0; i < source.Length; i++)
{
destination[i] = new Color(source[i].ToScaledVector4(), info.AlphaRepresentation, info.AlphaRepresentation);
}
return;
}
}
else
for (int i = 0; i < source.Length; i++)
{
for (int i = 0; i < source.Length; i++)
{
destination[i] = new Color(source[i], info.AlphaRepresentation);
}
destination[i] = new Color(source[i], info.AlphaRepresentation);
}
}
@ -349,26 +352,9 @@ public readonly partial struct Color : IEquatable<Color>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public Color WithAlpha(float alpha)
{
Vector4 v = this.ToScaledVector4();
float clampedAlpha = Numerics.Clamp(alpha, 0, 1);
if (this.isAssociated)
{
Vector4 unassociated = this.boxedHighPrecisionPixel is null ? this.unassociatedData : v;
if (this.boxedHighPrecisionPixel is not null)
{
Numerics.UnPremultiply(ref unassociated);
}
unassociated.W = clampedAlpha;
Vector4 associated = unassociated;
Numerics.Premultiply(ref associated);
return new Color(associated, unassociated, PixelAlphaRepresentation.Associated);
}
v.W = clampedAlpha;
return FromScaledVector(v, PixelAlphaRepresentation.Unassociated);
Vector4 vector = this.ToScaledVector4(PixelAlphaRepresentation.Unassociated);
vector.W = Numerics.Clamp(alpha, 0, 1);
return new Color(vector, this.AlphaRepresentation, PixelAlphaRepresentation.Unassociated);
}
/// <summary>
@ -411,17 +397,22 @@ public readonly partial struct Color : IEquatable<Color>
return pixel;
}
Vector4 vector = this.boxedHighPrecisionPixel is null
? this.isAssociated ? this.unassociatedData : this.data
: this.boxedHighPrecisionPixel.ToScaledVector4();
Vector4 vector = this.boxedHighPrecisionPixel?.ToScaledVector4() ?? this.data;
PixelOperations<TPixel> operations = PixelOperations<TPixel>.Instance;
if (this.isAssociated && this.boxedHighPrecisionPixel is not null)
if (this.dataIsAssociated)
{
if (operations is AssociatedAlphaPixelOperations<TPixel> associatedOperations)
{
// Preserve associated components directly while allowing the destination to quantize alpha to its own storage grid.
return associatedOperations.FromAssociatedScaledVector4(vector);
}
Numerics.UnPremultiply(ref vector);
}
// Destination pixel operations own association because RGB must use the alpha value representable by that destination format.
return PixelOperations<TPixel>.Instance.FromUnassociatedScaledVector4(vector);
// Unassociated input lets an associated destination quantize alpha before it multiplies the color components.
return operations.FromUnassociatedScaledVector4(vector);
}
/// <summary>
@ -434,12 +425,55 @@ public readonly partial struct Color : IEquatable<Color>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public Vector4 ToScaledVector4()
{
if (this.boxedHighPrecisionPixel is null)
Vector4 vector = this.boxedHighPrecisionPixel?.ToScaledVector4() ?? this.data;
if (this.dataIsAssociated == this.isAssociated)
{
return vector;
}
if (this.isAssociated)
{
Numerics.Premultiply(ref vector);
}
else
{
Numerics.UnPremultiply(ref vector);
}
return vector;
}
/// <summary>
/// Expands the color into a generic ("scaled") <see cref="Vector4"/> using the specified alpha representation,
/// with values scaled and clamped between <value>0</value> and <value>1</value>.
/// </summary>
/// <param name="alphaRepresentation">
/// The alpha representation to apply. <see cref="PixelAlphaRepresentation.Associated"/> returns color components
/// multiplied by alpha; other representations return color components independent of alpha.
/// </param>
/// <returns>The <see cref="Vector4"/>.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public Vector4 ToScaledVector4(PixelAlphaRepresentation alphaRepresentation)
{
bool targetIsAssociated = alphaRepresentation == PixelAlphaRepresentation.Associated;
Vector4 vector = this.boxedHighPrecisionPixel?.ToScaledVector4() ?? this.data;
if (this.dataIsAssociated == targetIsAssociated)
{
return vector;
}
if (targetIsAssociated)
{
return this.data;
Numerics.Premultiply(ref vector);
}
else
{
Numerics.UnPremultiply(ref vector);
}
return this.boxedHighPrecisionPixel.ToScaledVector4();
return vector;
}
/// <summary>
@ -467,7 +501,7 @@ public readonly partial struct Color : IEquatable<Color>
{
if (this.boxedHighPrecisionPixel is null && other.boxedHighPrecisionPixel is null)
{
return this.data == other.data && this.isAssociated == other.isAssociated;
return this.isAssociated == other.isAssociated && this.ToScaledVector4() == other.ToScaledVector4();
}
return this.isAssociated == other.isAssociated
@ -483,7 +517,7 @@ public readonly partial struct Color : IEquatable<Color>
{
if (this.boxedHighPrecisionPixel is null)
{
return HashCode.Combine(this.data, this.isAssociated);
return HashCode.Combine(this.ToScaledVector4(), this.isAssociated);
}
return HashCode.Combine(this.boxedHighPrecisionPixel.ToScaledVector4(), this.isAssociated);

3
src/ImageSharp/Common/Helpers/SimdUtils.Convert.cs

@ -70,14 +70,13 @@ internal static partial class SimdUtils
[MethodImpl(MethodImplOptions.NoInlining)]
private static void ConvertByteToNormalizedFloatRemainder(ReadOnlySpan<byte> source, Span<float> destination)
{
const float scale = 1F / byte.MaxValue;
ref byte sBase = ref MemoryMarshal.GetReference(source);
ref float dBase = ref MemoryMarshal.GetReference(destination);
for (int i = 0; i < source.Length; i++)
{
// Match the SIMD conversion so one span cannot contain different float representations of the same byte value.
Unsafe.Add(ref dBase, (uint)i) = Unsafe.Add(ref sBase, (uint)i) * scale;
Unsafe.Add(ref dBase, (uint)i) = Unsafe.Add(ref sBase, (uint)i) / (float)byte.MaxValue;
}
}

53
src/ImageSharp/Common/Helpers/SimdUtils.HwIntrinsics.cs

@ -711,6 +711,10 @@ internal static partial class SimdUtils
ReadOnlySpan<byte> source,
Span<float> destination)
{
const double reciprocal = 1D / byte.MaxValue;
const float reciprocalHigh = (float)reciprocal;
const float reciprocalLow = (float)(reciprocal - reciprocalHigh);
if (Vector512.IsHardwareAccelerated && Avx512F.IsSupported)
{
DebugVerifySpanInput(source, destination, Vector512<byte>.Count);
@ -719,6 +723,8 @@ internal static partial class SimdUtils
ref byte sourceBase = ref MemoryMarshal.GetReference(source);
ref Vector512<float> destinationBase = ref Unsafe.As<float, Vector512<float>>(ref MemoryMarshal.GetReference(destination));
Vector512<float> high = Vector512.Create(reciprocalHigh);
Vector512<float> low = Vector512.Create(reciprocalLow);
for (nuint i = 0; i < n; i++)
{
@ -728,11 +734,16 @@ internal static partial class SimdUtils
Vector512<int> i2 = Avx512F.ConvertToVector512Int32(Vector128.LoadUnsafe(ref sourceBase, si + (nuint)(Vector512<int>.Count * 2)));
Vector512<int> i3 = Avx512F.ConvertToVector512Int32(Vector128.LoadUnsafe(ref sourceBase, si + (nuint)(Vector512<int>.Count * 3)));
// Declare multiplier on each line. Codegen is better.
Vector512<float> f0 = Vector512.Create(1 / (float)byte.MaxValue) * Avx512F.ConvertToVector512Single(i0);
Vector512<float> f1 = Vector512.Create(1 / (float)byte.MaxValue) * Avx512F.ConvertToVector512Single(i1);
Vector512<float> f2 = Vector512.Create(1 / (float)byte.MaxValue) * Avx512F.ConvertToVector512Single(i2);
Vector512<float> f3 = Vector512.Create(1 / (float)byte.MaxValue) * Avx512F.ConvertToVector512Single(i3);
Vector512<float> f0 = Avx512F.ConvertToVector512Single(i0);
Vector512<float> f1 = Avx512F.ConvertToVector512Single(i1);
Vector512<float> f2 = Avx512F.ConvertToVector512Single(i2);
Vector512<float> f3 = Avx512F.ConvertToVector512Single(i3);
// The residual term restores the correctly rounded byte / 255F result without paying for vector division.
f0 = Vector512_.FusedMultiplyAdd(f0, high, f0 * low);
f1 = Vector512_.FusedMultiplyAdd(f1, high, f1 * low);
f2 = Vector512_.FusedMultiplyAdd(f2, high, f2 * low);
f3 = Vector512_.FusedMultiplyAdd(f3, high, f3 * low);
ref Vector512<float> d = ref Unsafe.Add(ref destinationBase, i * 4);
@ -750,6 +761,8 @@ internal static partial class SimdUtils
ref byte sourceBase = ref MemoryMarshal.GetReference(source);
ref Vector256<float> destinationBase = ref Unsafe.As<float, Vector256<float>>(ref MemoryMarshal.GetReference(destination));
Vector256<float> high = Vector256.Create(reciprocalHigh);
Vector256<float> low = Vector256.Create(reciprocalLow);
for (nuint i = 0; i < n; i++)
{
@ -762,11 +775,15 @@ internal static partial class SimdUtils
ref ulong refULong = ref Unsafe.As<byte, ulong>(ref Unsafe.Add(ref sourceBase, si));
Vector256<int> i3 = Avx2.ConvertToVector256Int32(Vector128.CreateScalarUnsafe(Unsafe.Add(ref refULong, 3)).AsByte());
// Declare multiplier on each line. Codegen is better.
Vector256<float> f0 = Vector256.Create(1 / (float)byte.MaxValue) * Avx.ConvertToVector256Single(i0);
Vector256<float> f1 = Vector256.Create(1 / (float)byte.MaxValue) * Avx.ConvertToVector256Single(i1);
Vector256<float> f2 = Vector256.Create(1 / (float)byte.MaxValue) * Avx.ConvertToVector256Single(i2);
Vector256<float> f3 = Vector256.Create(1 / (float)byte.MaxValue) * Avx.ConvertToVector256Single(i3);
Vector256<float> f0 = Avx.ConvertToVector256Single(i0);
Vector256<float> f1 = Avx.ConvertToVector256Single(i1);
Vector256<float> f2 = Avx.ConvertToVector256Single(i2);
Vector256<float> f3 = Avx.ConvertToVector256Single(i3);
f0 = Vector256_.FusedMultiplyAdd(f0, high, f0 * low);
f1 = Vector256_.FusedMultiplyAdd(f1, high, f1 * low);
f2 = Vector256_.FusedMultiplyAdd(f2, high, f2 * low);
f3 = Vector256_.FusedMultiplyAdd(f3, high, f3 * low);
ref Vector256<float> d = ref Unsafe.Add(ref destinationBase, i * 4);
@ -785,7 +802,8 @@ internal static partial class SimdUtils
ref byte sourceBase = ref MemoryMarshal.GetReference(source);
ref Vector128<float> destinationBase = ref Unsafe.As<float, Vector128<float>>(ref MemoryMarshal.GetReference(destination));
Vector128<float> scale = Vector128.Create(1 / (float)byte.MaxValue);
Vector128<float> high = Vector128.Create(reciprocalHigh);
Vector128<float> low = Vector128.Create(reciprocalLow);
for (nuint i = 0; i < n; i++)
{
@ -810,10 +828,15 @@ internal static partial class SimdUtils
(i2, i3) = Vector128.Widen(s1.AsInt16());
}
Vector128<float> f0 = scale * Vector128.ConvertToSingle(i0);
Vector128<float> f1 = scale * Vector128.ConvertToSingle(i1);
Vector128<float> f2 = scale * Vector128.ConvertToSingle(i2);
Vector128<float> f3 = scale * Vector128.ConvertToSingle(i3);
Vector128<float> f0 = Vector128.ConvertToSingle(i0);
Vector128<float> f1 = Vector128.ConvertToSingle(i1);
Vector128<float> f2 = Vector128.ConvertToSingle(i2);
Vector128<float> f3 = Vector128.ConvertToSingle(i3);
f0 = Vector128_.FusedMultiplyAdd(f0, high, f0 * low);
f1 = Vector128_.FusedMultiplyAdd(f1, high, f1 * low);
f2 = Vector128_.FusedMultiplyAdd(f2, high, f2 * low);
f3 = Vector128_.FusedMultiplyAdd(f3, high, f3 * low);
ref Vector128<float> d = ref Unsafe.Add(ref destinationBase, i * 4);

54
src/ImageSharp/Common/Helpers/Vector128Utilities.cs

@ -363,30 +363,58 @@ internal static class Vector128_
}
/// <summary>
/// Performs a multiplication and an addition of the <see cref="Vector128{Single}"/>.
/// Computes an estimate of (<paramref name="left"/> * <paramref name="right"/>) + <paramref name="addend"/>.
/// </summary>
/// <remarks>ret = (vm0 * vm1) + va</remarks>
/// <param name="va">The vector to add to the intermediate result.</param>
/// <param name="vm0">The first vector to multiply.</param>
/// <param name="vm1">The second vector to multiply.</param>
/// <returns>The <see cref="Vector256{T}"/>.</returns>
/// <param name="left">The first vector to multiply.</param>
/// <param name="right">The second vector to multiply.</param>
/// <param name="addend">The vector to add to the product.</param>
/// <returns>An estimate of the multiplication and addition result.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector128<float> MultiplyAdd(
Vector128<float> va,
Vector128<float> vm0,
Vector128<float> vm1)
public static Vector128<float> MultiplyAddEstimate(Vector128<float> left, Vector128<float> right, Vector128<float> addend)
{
if (Fma.IsSupported)
{
return Fma.MultiplyAdd(vm1, vm0, va);
return Fma.MultiplyAdd(left, right, addend);
}
if (AdvSimd.IsSupported)
{
return AdvSimd.FusedMultiplyAdd(va, vm0, vm1);
return AdvSimd.FusedMultiplyAdd(addend, left, right);
}
return va + (vm0 * vm1);
return (left * right) + addend;
}
/// <summary>
/// Computes (<paramref name="left"/> * <paramref name="right"/>) + <paramref name="addend"/>, rounded as one ternary operation.
/// </summary>
/// <param name="left">The first vector to multiply.</param>
/// <param name="right">The second vector to multiply.</param>
/// <param name="addend">The vector to add to the product.</param>
/// <returns>The fused multiplication and addition result.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector128<float> FusedMultiplyAdd(Vector128<float> left, Vector128<float> right, Vector128<float> addend)
{
if (Fma.IsSupported)
{
return Fma.MultiplyAdd(left, right, addend);
}
if (AdvSimd.IsSupported)
{
return AdvSimd.FusedMultiplyAdd(addend, left, right);
}
// WebAssembly SIMD has no exact fused multiply-add, so match the runtime fallback by preserving fused rounding per element.
Vector64<float> lower = Vector64.Create(
MathF.FusedMultiplyAdd(left.GetElement(0), right.GetElement(0), addend.GetElement(0)),
MathF.FusedMultiplyAdd(left.GetElement(1), right.GetElement(1), addend.GetElement(1)));
Vector64<float> upper = Vector64.Create(
MathF.FusedMultiplyAdd(left.GetElement(2), right.GetElement(2), addend.GetElement(2)),
MathF.FusedMultiplyAdd(left.GetElement(3), right.GetElement(3), addend.GetElement(3)));
return Vector128.Create(lower, upper);
}
/// <summary>

45
src/ImageSharp/Common/Helpers/Vector256Utilities.cs

@ -114,25 +114,46 @@ internal static class Vector256_
}
/// <summary>
/// Performs a multiplication and an addition of the <see cref="Vector256{Single}"/>.
/// Computes an estimate of (<paramref name="left"/> * <paramref name="right"/>) + <paramref name="addend"/>.
/// </summary>
/// <remarks>ret = (vm0 * vm1) + va</remarks>
/// <param name="va">The vector to add to the intermediate result.</param>
/// <param name="vm0">The first vector to multiply.</param>
/// <param name="vm1">The second vector to multiply.</param>
/// <returns>The <see cref="Vector256{T}"/>.</returns>
/// <param name="left">The first vector to multiply.</param>
/// <param name="right">The second vector to multiply.</param>
/// <param name="addend">The vector to add to the product.</param>
/// <returns>An estimate of the multiplication and addition result.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector256<float> MultiplyAdd(
Vector256<float> va,
Vector256<float> vm0,
Vector256<float> vm1)
public static Vector256<float> MultiplyAddEstimate(Vector256<float> left, Vector256<float> right, Vector256<float> addend)
{
if (Fma.IsSupported)
{
return Fma.MultiplyAdd(left, right, addend);
}
Vector128<float> lower = Vector128_.MultiplyAddEstimate(left.GetLower(), right.GetLower(), addend.GetLower());
Vector128<float> upper = Vector128_.MultiplyAddEstimate(left.GetUpper(), right.GetUpper(), addend.GetUpper());
return Vector256.Create(lower, upper);
}
/// <summary>
/// Computes (<paramref name="left"/> * <paramref name="right"/>) + <paramref name="addend"/>, rounded as one ternary operation.
/// </summary>
/// <param name="left">The first vector to multiply.</param>
/// <param name="right">The second vector to multiply.</param>
/// <param name="addend">The vector to add to the product.</param>
/// <returns>The fused multiplication and addition result.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector256<float> FusedMultiplyAdd(Vector256<float> left, Vector256<float> right, Vector256<float> addend)
{
if (Fma.IsSupported)
{
return Fma.MultiplyAdd(vm0, vm1, va);
return Fma.MultiplyAdd(left, right, addend);
}
return va + (vm0 * vm1);
// Match the runtime fallback by recursively applying the same fused contract to both halves.
Vector128<float> lower = Vector128_.FusedMultiplyAdd(left.GetLower(), right.GetLower(), addend.GetLower());
Vector128<float> upper = Vector128_.FusedMultiplyAdd(left.GetUpper(), right.GetUpper(), addend.GetUpper());
return Vector256.Create(lower, upper);
}
/// <summary>

50
src/ImageSharp/Common/Helpers/Vector512Utilities.cs

@ -86,19 +86,47 @@ internal static class Vector512_
=> Avx512F.RoundScale(vector, 0b0000_1000);
/// <summary>
/// Performs a multiplication and an addition of the <see cref="Vector512{Single}"/>.
/// Computes an estimate of (<paramref name="left"/> * <paramref name="right"/>) + <paramref name="addend"/>.
/// </summary>
/// <remarks>ret = (vm0 * vm1) + va</remarks>
/// <param name="va">The vector to add to the intermediate result.</param>
/// <param name="vm0">The first vector to multiply.</param>
/// <param name="vm1">The second vector to multiply.</param>
/// <returns>The <see cref="Vector256{T}"/>.</returns>
/// <param name="left">The first vector to multiply.</param>
/// <param name="right">The second vector to multiply.</param>
/// <param name="addend">The vector to add to the product.</param>
/// <returns>An estimate of the multiplication and addition result.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector512<float> MultiplyAdd(
Vector512<float> va,
Vector512<float> vm0,
Vector512<float> vm1)
=> Avx512F.FusedMultiplyAdd(vm0, vm1, va);
public static Vector512<float> MultiplyAddEstimate(Vector512<float> left, Vector512<float> right, Vector512<float> addend)
{
if (Avx512F.IsSupported)
{
return Avx512F.FusedMultiplyAdd(left, right, addend);
}
Vector256<float> lower = Vector256_.MultiplyAddEstimate(left.GetLower(), right.GetLower(), addend.GetLower());
Vector256<float> upper = Vector256_.MultiplyAddEstimate(left.GetUpper(), right.GetUpper(), addend.GetUpper());
return Vector512.Create(lower, upper);
}
/// <summary>
/// Computes (<paramref name="left"/> * <paramref name="right"/>) + <paramref name="addend"/>, rounded as one ternary operation.
/// </summary>
/// <param name="left">The first vector to multiply.</param>
/// <param name="right">The second vector to multiply.</param>
/// <param name="addend">The vector to add to the product.</param>
/// <returns>The fused multiplication and addition result.</returns>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector512<float> FusedMultiplyAdd(Vector512<float> left, Vector512<float> right, Vector512<float> addend)
{
if (Avx512F.IsSupported)
{
return Avx512F.FusedMultiplyAdd(left, right, addend);
}
// Match the runtime fallback by recursively applying the same fused contract to both halves.
Vector256<float> lower = Vector256_.FusedMultiplyAdd(left.GetLower(), right.GetLower(), addend.GetLower());
Vector256<float> upper = Vector256_.FusedMultiplyAdd(left.GetUpper(), right.GetUpper(), addend.GetUpper());
return Vector512.Create(lower, upper);
}
/// <summary>
/// Performs a multiplication and a negated addition of the <see cref="Vector512{Single}"/>.

2
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector128.cs

@ -74,7 +74,7 @@ internal abstract partial class JpegColorConverterBase
ref Vector128<float> b = ref Unsafe.Add(ref srcBlue, i);
// luminosity = (0.299 * r) + (0.587 * g) + (0.114 * b)
Unsafe.Add(ref destLuminance, i) = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Unsafe.Add(ref destLuminance, i) = Vector128_.MultiplyAddEstimate(f0299, r, Vector128_.MultiplyAddEstimate(f0587, g, f0114 * b));
}
}
}

2
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector256.cs

@ -74,7 +74,7 @@ internal abstract partial class JpegColorConverterBase
ref Vector256<float> b = ref Unsafe.Add(ref srcBlue, i);
// luminosity = (0.299 * r) + (0.587 * g) + (0.114 * b)
Unsafe.Add(ref destLuminance, i) = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Unsafe.Add(ref destLuminance, i) = Vector256_.MultiplyAddEstimate(f0299, r, Vector256_.MultiplyAddEstimate(f0587, g, f0114 * b));
}
}
}

2
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector512.cs

@ -74,7 +74,7 @@ internal abstract partial class JpegColorConverterBase
ref Vector512<float> b = ref Unsafe.Add(ref srcBlue, i);
// luminosity = (0.299 * r) + (0.587 * g) + (0.114 * b)
Unsafe.Add(ref destLuminance, i) = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Unsafe.Add(ref destLuminance, i) = Vector512_.MultiplyAddEstimate(f0299, r, Vector512_.MultiplyAddEstimate(f0587, g, f0114 * b));
}
}

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector128.cs

@ -53,9 +53,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector128<float> r = Vector128_.MultiplyAdd(y, cr, rCrMult) * scaledK;
Vector128<float> g = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult) * scaledK;
Vector128<float> b = Vector128_.MultiplyAdd(y, cb, bCbMult) * scaledK;
Vector128<float> r = Vector128_.MultiplyAddEstimate(cr, rCrMult, y) * scaledK;
Vector128<float> g = Vector128_.MultiplyAddEstimate(cr, gCrMult, Vector128_.MultiplyAddEstimate(cb, gCbMult, y)) * scaledK;
Vector128<float> b = Vector128_.MultiplyAddEstimate(cb, bCbMult, y) * scaledK;
c0 = r;
c1 = g;
@ -117,9 +117,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector128<float> y = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector128<float> cb = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector128<float> cr = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector128<float> y = Vector128_.MultiplyAddEstimate(f0299, r, Vector128_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector128<float> cb = chromaOffset + Vector128_.MultiplyAddEstimate(fn0168736, r, Vector128_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector128<float> cr = chromaOffset + Vector128_.MultiplyAddEstimate(f05, r, Vector128_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y * maxSampleValue;
Unsafe.Add(ref destCb, i) = chromaOffset + (cb * maxSampleValue);

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector256.cs

@ -53,9 +53,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector256<float> r = Vector256_.MultiplyAdd(y, cr, rCrMult) * scaledK;
Vector256<float> g = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult) * scaledK;
Vector256<float> b = Vector256_.MultiplyAdd(y, cb, bCbMult) * scaledK;
Vector256<float> r = Vector256_.MultiplyAddEstimate(cr, rCrMult, y) * scaledK;
Vector256<float> g = Vector256_.MultiplyAddEstimate(cr, gCrMult, Vector256_.MultiplyAddEstimate(cb, gCbMult, y)) * scaledK;
Vector256<float> b = Vector256_.MultiplyAddEstimate(cb, bCbMult, y) * scaledK;
c0 = r;
c1 = g;
@ -117,9 +117,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector256<float> y = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector256<float> cb = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector256<float> cr = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector256<float> y = Vector256_.MultiplyAddEstimate(f0299, r, Vector256_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector256<float> cb = chromaOffset + Vector256_.MultiplyAddEstimate(fn0168736, r, Vector256_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector256<float> cr = chromaOffset + Vector256_.MultiplyAddEstimate(f05, r, Vector256_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y * maxSampleValue;
Unsafe.Add(ref destCb, i) = chromaOffset + (cb * maxSampleValue);

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector512.cs

@ -57,9 +57,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector512<float> r = Vector512_.MultiplyAdd(y, cr, rCrMult) * scaledK;
Vector512<float> g = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult) * scaledK;
Vector512<float> b = Vector512_.MultiplyAdd(y, cb, bCbMult) * scaledK;
Vector512<float> r = Vector512_.MultiplyAddEstimate(cr, rCrMult, y) * scaledK;
Vector512<float> g = Vector512_.MultiplyAddEstimate(cr, gCrMult, Vector512_.MultiplyAddEstimate(cb, gCbMult, y)) * scaledK;
Vector512<float> b = Vector512_.MultiplyAddEstimate(cb, bCbMult, y) * scaledK;
c0 = r;
c1 = g;
@ -128,9 +128,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector512<float> y = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector512<float> cb = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector512<float> cr = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector512<float> y = Vector512_.MultiplyAddEstimate(f0299, r, Vector512_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector512<float> cb = chromaOffset + Vector512_.MultiplyAddEstimate(fn0168736, r, Vector512_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector512<float> cr = chromaOffset + Vector512_.MultiplyAddEstimate(f05, r, Vector512_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y * maxSampleValue;
Unsafe.Add(ref destCb, i) = chromaOffset + (cb * maxSampleValue);

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector128.cs

@ -53,9 +53,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector128<float> r = Vector128_.MultiplyAdd(y, cr, rCrMult);
Vector128<float> g = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult);
Vector128<float> b = Vector128_.MultiplyAdd(y, cb, bCbMult);
Vector128<float> r = Vector128_.MultiplyAddEstimate(cr, rCrMult, y);
Vector128<float> g = Vector128_.MultiplyAddEstimate(cr, gCrMult, Vector128_.MultiplyAddEstimate(cb, gCbMult, y));
Vector128<float> b = Vector128_.MultiplyAddEstimate(cb, bCbMult, y);
r = Vector128_.RoundToNearestInteger(r) * scale;
g = Vector128_.RoundToNearestInteger(g) * scale;
@ -108,9 +108,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector128<float> y = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector128<float> cb = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector128<float> cr = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector128<float> y = Vector128_.MultiplyAddEstimate(f0299, r, Vector128_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector128<float> cb = chromaOffset + Vector128_.MultiplyAddEstimate(fn0168736, r, Vector128_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector128<float> cr = chromaOffset + Vector128_.MultiplyAddEstimate(f05, r, Vector128_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y;
Unsafe.Add(ref destCb, i) = cb;

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector256.cs

@ -53,9 +53,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector256<float> r = Vector256_.MultiplyAdd(y, cr, rCrMult);
Vector256<float> g = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult);
Vector256<float> b = Vector256_.MultiplyAdd(y, cb, bCbMult);
Vector256<float> r = Vector256_.MultiplyAddEstimate(cr, rCrMult, y);
Vector256<float> g = Vector256_.MultiplyAddEstimate(cr, gCrMult, Vector256_.MultiplyAddEstimate(cb, gCbMult, y));
Vector256<float> b = Vector256_.MultiplyAddEstimate(cb, bCbMult, y);
r = Vector256_.RoundToNearestInteger(r) * scale;
g = Vector256_.RoundToNearestInteger(g) * scale;
@ -108,9 +108,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector256<float> y = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector256<float> cb = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector256<float> cr = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector256<float> y = Vector256_.MultiplyAddEstimate(f0299, r, Vector256_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector256<float> cb = chromaOffset + Vector256_.MultiplyAddEstimate(fn0168736, r, Vector256_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector256<float> cr = chromaOffset + Vector256_.MultiplyAddEstimate(f05, r, Vector256_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y;
Unsafe.Add(ref destCb, i) = cb;

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector512.cs

@ -56,9 +56,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector512<float> r = Vector512_.MultiplyAdd(y, cr, rCrMult);
Vector512<float> g = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult);
Vector512<float> b = Vector512_.MultiplyAdd(y, cb, bCbMult);
Vector512<float> r = Vector512_.MultiplyAddEstimate(cr, rCrMult, y);
Vector512<float> g = Vector512_.MultiplyAddEstimate(cr, gCrMult, Vector512_.MultiplyAddEstimate(cb, gCbMult, y));
Vector512<float> b = Vector512_.MultiplyAddEstimate(cb, bCbMult, y);
r = Vector512_.RoundToNearestInteger(r) * scale;
g = Vector512_.RoundToNearestInteger(g) * scale;
@ -107,9 +107,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector512<float> y = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector512<float> cb = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector512<float> cr = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector512<float> y = Vector512_.MultiplyAddEstimate(f0299, r, Vector512_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector512<float> cb = chromaOffset + Vector512_.MultiplyAddEstimate(fn0168736, r, Vector512_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector512<float> cr = chromaOffset + Vector512_.MultiplyAddEstimate(f05, r, Vector512_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y;
Unsafe.Add(ref destCb, i) = cb;

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector128.cs

@ -58,9 +58,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector128<float> r = Vector128_.MultiplyAdd(y, cr, rCrMult);
Vector128<float> g = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult);
Vector128<float> b = Vector128_.MultiplyAdd(y, cb, bCbMult);
Vector128<float> r = Vector128_.MultiplyAddEstimate(cr, rCrMult, y);
Vector128<float> g = Vector128_.MultiplyAddEstimate(cr, gCrMult, Vector128_.MultiplyAddEstimate(cb, gCbMult, y));
Vector128<float> b = Vector128_.MultiplyAddEstimate(cb, bCbMult, y);
r = max - Vector128_.RoundToNearestInteger(r);
g = max - Vector128_.RoundToNearestInteger(g);
@ -122,9 +122,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector128<float> y = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector128<float> cb = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector128<float> cr = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector128<float> y = Vector128_.MultiplyAddEstimate(f0299, r, Vector128_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector128<float> cb = chromaOffset + Vector128_.MultiplyAddEstimate(fn0168736, r, Vector128_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector128<float> cr = chromaOffset + Vector128_.MultiplyAddEstimate(f05, r, Vector128_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y;
Unsafe.Add(ref destCb, i) = cb;

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector256.cs

@ -58,9 +58,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector256<float> r = Vector256_.MultiplyAdd(y, cr, rCrMult);
Vector256<float> g = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult);
Vector256<float> b = Vector256_.MultiplyAdd(y, cb, bCbMult);
Vector256<float> r = Vector256_.MultiplyAddEstimate(cr, rCrMult, y);
Vector256<float> g = Vector256_.MultiplyAddEstimate(cr, gCrMult, Vector256_.MultiplyAddEstimate(cb, gCbMult, y));
Vector256<float> b = Vector256_.MultiplyAddEstimate(cb, bCbMult, y);
r = max - Vector256_.RoundToNearestInteger(r);
g = max - Vector256_.RoundToNearestInteger(g);
@ -122,9 +122,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector256<float> y = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector256<float> cb = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector256<float> cr = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector256<float> y = Vector256_.MultiplyAddEstimate(f0299, r, Vector256_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector256<float> cb = chromaOffset + Vector256_.MultiplyAddEstimate(fn0168736, r, Vector256_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector256<float> cr = chromaOffset + Vector256_.MultiplyAddEstimate(f05, r, Vector256_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y;
Unsafe.Add(ref destCb, i) = cb;

12
src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector512.cs

@ -58,9 +58,9 @@ internal abstract partial class JpegColorConverterBase
// r = y + (1.402F * cr);
// g = y - (0.344136F * cb) - (0.714136F * cr);
// b = y + (1.772F * cb);
Vector512<float> r = Vector512_.MultiplyAdd(y, cr, rCrMult);
Vector512<float> g = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult);
Vector512<float> b = Vector512_.MultiplyAdd(y, cb, bCbMult);
Vector512<float> r = Vector512_.MultiplyAddEstimate(cr, rCrMult, y);
Vector512<float> g = Vector512_.MultiplyAddEstimate(cr, gCrMult, Vector512_.MultiplyAddEstimate(cb, gCbMult, y));
Vector512<float> b = Vector512_.MultiplyAddEstimate(cb, bCbMult, y);
r = max - Vector512_.RoundToNearestInteger(r);
g = max - Vector512_.RoundToNearestInteger(g);
@ -126,9 +126,9 @@ internal abstract partial class JpegColorConverterBase
// y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b)
// cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b)
// cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b)
Vector512<float> y = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f0114 * b, f0587, g), f0299, r);
Vector512<float> cb = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r);
Vector512<float> cr = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r);
Vector512<float> y = Vector512_.MultiplyAddEstimate(f0299, r, Vector512_.MultiplyAddEstimate(f0587, g, f0114 * b));
Vector512<float> cb = chromaOffset + Vector512_.MultiplyAddEstimate(fn0168736, r, Vector512_.MultiplyAddEstimate(fn0331264, g, f05 * b));
Vector512<float> cr = chromaOffset + Vector512_.MultiplyAddEstimate(f05, r, Vector512_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b));
Unsafe.Add(ref destY, i) = y;
Unsafe.Add(ref destCb, i) = cb;

8
src/ImageSharp/Formats/Jpeg/Components/FloatingPointDCT.Vector256.cs

@ -55,8 +55,8 @@ internal static partial class FloatingPointDCT
tmp12 = tmp6 + tmp7;
Vector256<float> z5 = (tmp10 - tmp12) * Vector256.Create(0.382683433f); // mm256_F_0_3826
Vector256<float> z2 = Vector256_.MultiplyAdd(z5, Vector256.Create(0.541196100f), tmp10); // mm256_F_0_5411
Vector256<float> z4 = Vector256_.MultiplyAdd(z5, Vector256.Create(1.306562965f), tmp12); // mm256_F_1_3065
Vector256<float> z2 = Vector256_.MultiplyAddEstimate(Vector256.Create(0.541196100f), tmp10, z5); // mm256_F_0_5411
Vector256<float> z4 = Vector256_.MultiplyAddEstimate(Vector256.Create(1.306562965f), tmp12, z5); // mm256_F_1_3065
Vector256<float> z3 = tmp11 * mm256_F_0_7071;
Vector256<float> z11 = tmp7 + z3;
@ -122,8 +122,8 @@ internal static partial class FloatingPointDCT
z5 = (z10 + z12) * Vector256.Create(1.847759065f); // mm256_F_1_8477
tmp10 = Vector256_.MultiplyAdd(z5, z12, Vector256.Create(-1.082392200f)); // mm256_F_n1_0823
tmp12 = Vector256_.MultiplyAdd(z5, z10, Vector256.Create(-2.613125930f)); // mm256_F_n2_6131
tmp10 = Vector256_.MultiplyAddEstimate(z12, Vector256.Create(-1.082392200f), z5); // mm256_F_n1_0823
tmp12 = Vector256_.MultiplyAddEstimate(z10, Vector256.Create(-2.613125930f), z5); // mm256_F_n2_6131
tmp6 = tmp12 - tmp7;
tmp5 = tmp11 - tmp6;

16
src/ImageSharp/PixelFormats/PixelBlenders/AssociatedAlphaPorterDuffFunctions.cs

@ -398,7 +398,9 @@ internal static partial class AssociatedAlphaPorterDuffFunctions
// Atop discards source-only color and retains the destination alpha unchanged.
Vector4 sourceAlpha = Numerics.PermuteW(source);
Vector4 destinationAlpha = Numerics.PermuteW(destination);
Vector4 result = (destination * (Vector4.One - sourceAlpha)) + overlap;
Vector4 coefficient = Vector4.One - sourceAlpha;
Vector4 result = Vector128_.FusedMultiplyAdd(destination.AsVector128(), coefficient.AsVector128(), overlap.AsVector128()).AsVector4();
return Numerics.WithW(result, destinationAlpha);
}
@ -408,7 +410,8 @@ internal static partial class AssociatedAlphaPorterDuffFunctions
{
Vector256<float> sourceAlpha = Avx.Permute(source, ShuffleAlphaControl);
Vector256<float> destinationAlpha = Avx.Permute(destination, ShuffleAlphaControl);
Vector256<float> result = (destination * (Vector256.Create(1F) - sourceAlpha)) + overlap;
Vector256<float> result = Vector256_.FusedMultiplyAdd(destination, Vector256.Create(1F) - sourceAlpha, overlap);
return Avx.Blend(result, destinationAlpha, BlendAlphaControl);
}
@ -418,7 +421,8 @@ internal static partial class AssociatedAlphaPorterDuffFunctions
{
Vector512<float> sourceAlpha = Vector512_.ShuffleNative(source, ShuffleAlphaControl);
Vector512<float> destinationAlpha = Vector512_.ShuffleNative(destination, ShuffleAlphaControl);
Vector512<float> result = (destination * (Vector512.Create(1F) - sourceAlpha)) + overlap;
Vector512<float> result = Vector512_.FusedMultiplyAdd(destination, Vector512.Create(1F) - sourceAlpha, overlap);
return Vector512.ConditionalSelect(AlphaMask512(), destinationAlpha, result);
}
@ -506,18 +510,18 @@ internal static partial class AssociatedAlphaPorterDuffFunctions
public static Vector4 BlendWithCoverage(Vector4 backdrop, Vector4 source, float coverage)
{
// Use the same fused operation as the wider paths so exact midpoints cannot change across vector widths.
return Vector128_.MultiplyAdd(backdrop.AsVector128(), (source - backdrop).AsVector128(), Vector128.Create(coverage)).AsVector4();
return Vector128_.FusedMultiplyAdd((source - backdrop).AsVector128(), Vector128.Create(coverage), backdrop.AsVector128()).AsVector4();
}
/// <inheritdoc cref="BlendWithCoverage(Vector4, Vector4, float)" />
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector256<float> BlendWithCoverage(Vector256<float> backdrop, Vector256<float> source, Vector256<float> coverage)
=> Vector256_.MultiplyAdd(backdrop, source - backdrop, coverage);
=> Vector256_.FusedMultiplyAdd(source - backdrop, coverage, backdrop);
/// <inheritdoc cref="BlendWithCoverage(Vector4, Vector4, float)" />
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public static Vector512<float> BlendWithCoverage(Vector512<float> backdrop, Vector512<float> source, Vector512<float> coverage)
=> Vector512_.MultiplyAdd(backdrop, source - backdrop, coverage);
=> Vector512_.FusedMultiplyAdd(source - backdrop, coverage, backdrop);
/// <summary>
/// Calculates one associated Overlay overlap component without recovering either straight component.

26
src/ImageSharp/PixelFormats/PixelBlenders/PorterDuffFunctions.cs

@ -340,7 +340,7 @@ internal static partial class PorterDuffFunctions
Vector4 sourcePremultiplied = Numerics.WithW(source * sourceAlpha, sourceAlpha);
// Use the same fused operation as the wider paths so exact midpoints cannot change across vector widths.
Vector4 result = Vector128_.MultiplyAdd(backdropPremultiplied.AsVector128(), (sourcePremultiplied - backdropPremultiplied).AsVector128(), Vector128.Create(coverage)).AsVector4();
Vector4 result = Vector128_.MultiplyAddEstimate((sourcePremultiplied - backdropPremultiplied).AsVector128(), Vector128.Create(coverage), backdropPremultiplied.AsVector128()).AsVector4();
Numerics.UnPremultiply(ref result);
return result;
@ -360,7 +360,7 @@ internal static partial class PorterDuffFunctions
Vector256<float> sourceAlpha = Avx.Permute(source, ShuffleAlphaControl);
Vector256<float> backdropPremultiplied = Avx.Blend(backdrop * backdropAlpha, backdropAlpha, BlendAlphaControl);
Vector256<float> sourcePremultiplied = Avx.Blend(source * sourceAlpha, sourceAlpha, BlendAlphaControl);
Vector256<float> result = Vector256_.MultiplyAdd(backdropPremultiplied, sourcePremultiplied - backdropPremultiplied, coverage);
Vector256<float> result = Vector256_.MultiplyAddEstimate(sourcePremultiplied - backdropPremultiplied, coverage, backdropPremultiplied);
return Numerics.UnPremultiply(result, Avx.Permute(result, ShuffleAlphaControl));
}
@ -380,7 +380,7 @@ internal static partial class PorterDuffFunctions
Vector512<float> alphaMask = AlphaMask512();
Vector512<float> backdropPremultiplied = Vector512.ConditionalSelect(alphaMask, backdropAlpha, backdrop * backdropAlpha);
Vector512<float> sourcePremultiplied = Vector512.ConditionalSelect(alphaMask, sourceAlpha, source * sourceAlpha);
Vector512<float> result = Vector512_.MultiplyAdd(backdropPremultiplied, sourcePremultiplied - backdropPremultiplied, coverage);
Vector512<float> result = Vector512_.MultiplyAddEstimate(sourcePremultiplied - backdropPremultiplied, coverage, backdropPremultiplied);
return Numerics.UnPremultiply(result, Vector512_.ShuffleNative(result, ShuffleAlphaControl));
}
@ -483,8 +483,8 @@ internal static partial class PorterDuffFunctions
// calculate final color
Vector256<float> color = destination * dstW;
color = Vector256_.MultiplyAdd(color, source, srcW);
color = Vector256_.MultiplyAdd(color, blend, blendW);
color = Vector256_.MultiplyAddEstimate(source, srcW, color);
color = Vector256_.MultiplyAddEstimate(blend, blendW, color);
// unpremultiply
return Numerics.UnPremultiply(color, alpha);
@ -513,8 +513,8 @@ internal static partial class PorterDuffFunctions
// calculate final color
Vector512<float> color = destination * dstW;
color = Vector512_.MultiplyAdd(color, source, srcW);
color = Vector512_.MultiplyAdd(color, blend, blendW);
color = Vector512_.MultiplyAddEstimate(source, srcW, color);
color = Vector512_.MultiplyAddEstimate(blend, blendW, color);
// unpremultiply
return Numerics.UnPremultiply(color, alpha);
@ -567,7 +567,7 @@ internal static partial class PorterDuffFunctions
Vector256<float> dstW = alpha - blendW;
// calculate final color
Vector256<float> color = Vector256_.MultiplyAdd(Avx.Multiply(blend, blendW), destination, dstW);
Vector256<float> color = Vector256_.MultiplyAddEstimate(destination, dstW, Avx.Multiply(blend, blendW));
// unpremultiply
return Numerics.UnPremultiply(color, alpha);
@ -592,7 +592,7 @@ internal static partial class PorterDuffFunctions
Vector512<float> dstW = alpha - blendW;
// calculate final color
Vector512<float> color = Vector512_.MultiplyAdd(blend * blendW, destination, dstW);
Vector512<float> color = Vector512_.MultiplyAddEstimate(destination, dstW, blend * blendW);
// unpremultiply
return Numerics.UnPremultiply(color, alpha);
@ -751,8 +751,8 @@ internal static partial class PorterDuffFunctions
Vector256<float> dstW = vOne - sW;
// calculate alpha
Vector256<float> alpha = Vector256_.MultiplyAdd(Avx.Multiply(dW, dstW), sW, srcW);
Vector256<float> color = Vector256_.MultiplyAdd(Avx.Multiply(Avx.Multiply(dW, destination), dstW), Avx.Multiply(sW, source), srcW);
Vector256<float> alpha = Vector256_.MultiplyAddEstimate(sW, srcW, Avx.Multiply(dW, dstW));
Vector256<float> color = Vector256_.MultiplyAddEstimate(Avx.Multiply(sW, source), srcW, Avx.Multiply(Avx.Multiply(dW, destination), dstW));
// unpremultiply
return Numerics.UnPremultiply(color, alpha);
@ -776,8 +776,8 @@ internal static partial class PorterDuffFunctions
Vector512<float> dstW = vOne - sW;
// calculate alpha
Vector512<float> alpha = Vector512_.MultiplyAdd(dW * dstW, sW, srcW);
Vector512<float> color = Vector512_.MultiplyAdd((dW * destination) * dstW, sW * source, srcW);
Vector512<float> alpha = Vector512_.MultiplyAddEstimate(sW, srcW, dW * dstW);
Vector512<float> color = Vector512_.MultiplyAddEstimate(sW * source, srcW, (dW * destination) * dstW);
// unpremultiply
return Numerics.UnPremultiply(color, alpha);

2
src/ImageSharp/PixelFormats/PixelImplementations/Abgr32.cs

@ -182,7 +182,7 @@ public partial struct Abgr32 : IPixel<Abgr32>, IPackedVector<uint>
/// <inheritdoc/>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / MaxBytes;
public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue;
/// <inheritdoc />
public static PixelTypeInfo GetPixelTypeInfo()

4
src/ImageSharp/PixelFormats/PixelImplementations/Abgr32P.cs

@ -19,8 +19,6 @@ namespace SixLabors.ImageSharp.PixelFormats;
[StructLayout(LayoutKind.Sequential)]
public partial struct Abgr32P : IPixel<Abgr32P>, IPackedVector<uint>
{
private const float ByteScale = 1F / byte.MaxValue;
private static readonly Vector4 Half = new(0.5F);
private static readonly Vector4 MaxBytes = new(byte.MaxValue);
@ -147,7 +145,7 @@ public partial struct Abgr32P : IPixel<Abgr32P>, IPackedVector<uint>
=> Rgba32.FromScaledVector4(Vector4Converters.AssociatedRgbaCompatible.ToUnassociatedVector4(this.R, this.G, this.B, this.A));
/// <inheritdoc />
public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) * ByteScale;
public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue;
/// <inheritdoc />
public readonly Vector4 ToVector4() => this.ToScaledVector4();

2
src/ImageSharp/PixelFormats/PixelImplementations/Argb32.cs

@ -175,7 +175,7 @@ public partial struct Argb32 : IPixel<Argb32>, IPackedVector<uint>
/// <inheritdoc/>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / MaxBytes;
public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue;
/// <inheritdoc />
public static PixelTypeInfo GetPixelTypeInfo()

4
src/ImageSharp/PixelFormats/PixelImplementations/Argb32P.cs

@ -19,8 +19,6 @@ namespace SixLabors.ImageSharp.PixelFormats;
[StructLayout(LayoutKind.Sequential)]
public partial struct Argb32P : IPixel<Argb32P>, IPackedVector<uint>
{
private const float ByteScale = 1F / byte.MaxValue;
private static readonly Vector4 Half = new(0.5F);
private static readonly Vector4 MaxBytes = new(byte.MaxValue);
@ -147,7 +145,7 @@ public partial struct Argb32P : IPixel<Argb32P>, IPackedVector<uint>
=> Rgba32.FromScaledVector4(Vector4Converters.AssociatedRgbaCompatible.ToUnassociatedVector4(this.R, this.G, this.B, this.A));
/// <inheritdoc />
public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) * ByteScale;
public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue;
/// <inheritdoc />
public readonly Vector4 ToVector4() => this.ToScaledVector4();

2
src/ImageSharp/PixelFormats/PixelImplementations/Bgra32.cs

@ -124,7 +124,7 @@ public partial struct Bgra32 : IPixel<Bgra32>, IPackedVector<uint>
/// <inheritdoc/>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / MaxBytes;
public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue;
/// <inheritdoc />
public static PixelTypeInfo GetPixelTypeInfo()

4
src/ImageSharp/PixelFormats/PixelImplementations/Bgra32P.cs

@ -19,8 +19,6 @@ namespace SixLabors.ImageSharp.PixelFormats;
[StructLayout(LayoutKind.Sequential)]
public partial struct Bgra32P : IPixel<Bgra32P>, IPackedVector<uint>
{
private const float ByteScale = 1F / byte.MaxValue;
private static readonly Vector4 Half = new(0.5F);
private static readonly Vector4 MaxBytes = new(byte.MaxValue);
@ -147,7 +145,7 @@ public partial struct Bgra32P : IPixel<Bgra32P>, IPackedVector<uint>
=> Rgba32.FromScaledVector4(Vector4Converters.AssociatedRgbaCompatible.ToUnassociatedVector4(this.R, this.G, this.B, this.A));
/// <inheritdoc />
public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) * ByteScale;
public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue;
/// <inheritdoc />
public readonly Vector4 ToVector4() => this.ToScaledVector4();

5
src/ImageSharp/PixelFormats/PixelImplementations/NormalizedByte4P.cs

@ -158,10 +158,7 @@ public partial struct NormalizedByte4P : IPixel<NormalizedByte4P>, IPackedVector
/// </summary>
/// <param name="source">The unassociated scaled vector.</param>
/// <returns>The associated pixel.</returns>
private static NormalizedByte4P FromUnassociatedScaledVector4(Vector4 source)
{
return FromScaledVector4(Associate(source));
}
private static NormalizedByte4P FromUnassociatedScaledVector4(Vector4 source) => FromScaledVector4(Associate(source));
/// <summary>
/// Converts an unassociated scaled vector to the associated representation of a signed-normalized-byte destination.

2
src/ImageSharp/PixelFormats/PixelImplementations/Rgba32.cs

@ -220,7 +220,7 @@ public partial struct Rgba32 : IPixel<Rgba32>, IPackedVector<uint>
/// <inheritdoc/>
[MethodImpl(MethodImplOptions.AggressiveInlining)]
public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / MaxBytes;
public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue;
/// <inheritdoc />
public static PixelTypeInfo GetPixelTypeInfo()

4
src/ImageSharp/PixelFormats/PixelImplementations/Rgba32P.cs

@ -19,8 +19,6 @@ namespace SixLabors.ImageSharp.PixelFormats;
[StructLayout(LayoutKind.Sequential)]
public partial struct Rgba32P : IPixel<Rgba32P>, IPackedVector<uint>
{
private const float ByteScale = 1F / byte.MaxValue;
private static readonly Vector4 Half = new(0.5F);
private static readonly Vector4 MaxBytes = new(byte.MaxValue);
@ -147,7 +145,7 @@ public partial struct Rgba32P : IPixel<Rgba32P>, IPackedVector<uint>
=> Rgba32.FromScaledVector4(Vector4Converters.AssociatedRgbaCompatible.ToUnassociatedVector4(this.R, this.G, this.B, this.A));
/// <inheritdoc />
public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) * ByteScale;
public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue;
/// <inheritdoc />
public readonly Vector4 ToVector4() => this.ToScaledVector4();

8
src/ImageSharp/Processing/Processors/Transforms/Resize/ResizeKernel.cs

@ -104,8 +104,8 @@ internal readonly unsafe struct ResizeKernel
Vector256<float> pixels256_0 = Unsafe.As<Vector4, Vector256<float>>(ref rowStartRef);
Vector256<float> pixels256_1 = Unsafe.As<Vector4, Vector256<float>>(ref Unsafe.Add(ref rowStartRef, (nuint)2));
result256_0 = Vector256_.MultiplyAdd(result256_0, Vector256.Load(bufferStart), pixels256_0);
result256_1 = Vector256_.MultiplyAdd(result256_1, Vector256.Load(bufferStart + 8), pixels256_1);
result256_0 = Vector256_.MultiplyAddEstimate(Vector256.Load(bufferStart), pixels256_0, result256_0);
result256_1 = Vector256_.MultiplyAddEstimate(Vector256.Load(bufferStart + 8), pixels256_1, result256_1);
bufferStart += 16;
rowStartRef = ref Unsafe.Add(ref rowStartRef, (nuint)4);
@ -116,7 +116,7 @@ internal readonly unsafe struct ResizeKernel
if ((this.Length & 3) >= 2)
{
Vector256<float> pixels256_0 = Unsafe.As<Vector4, Vector256<float>>(ref rowStartRef);
result256_0 = Vector256_.MultiplyAdd(result256_0, Vector256.Load(bufferStart), pixels256_0);
result256_0 = Vector256_.MultiplyAddEstimate(Vector256.Load(bufferStart), pixels256_0, result256_0);
bufferStart += 8;
rowStartRef = ref Unsafe.Add(ref rowStartRef, (nuint)2);
@ -127,7 +127,7 @@ internal readonly unsafe struct ResizeKernel
if ((this.Length & 1) != 0)
{
Vector128<float> pixels128 = Unsafe.As<Vector4, Vector128<float>>(ref rowStartRef);
result128 = Vector128_.MultiplyAdd(result128, Vector128.Load(bufferStart), pixels128);
result128 = Vector128_.MultiplyAddEstimate(Vector128.Load(bufferStart), pixels128, result128);
}
return result128.AsVector4();

16
tests/ImageSharp.Benchmarks/Bulk/FromVector4.cs

@ -159,13 +159,13 @@ public class FromVector4Bgra32P : FromVector4<Bgra32P>
AMD RYZEN AI MAX+ 395 w/ Radeon 8060S 3.00GHz, 1 CPU, 32 logical and 16 physical cores
.NET 8.0.28, X64 RyuJIT x86-64-v4
| Method | Count | Mean | Error | StdDev | Ratio | RatioSD | Code Size | Allocated |
|---------------------------- |------ |------------:|-----------:|----------:|------:|--------:|----------:|----------:|
| PixelOperations_Base | 64 | 59.73 ns | 19.644 ns | 1.077 ns | 1.00 | 0.02 | 1,152 B | - |
| PixelOperations_Specialized | 64 | 50.86 ns | 7.372 ns | 0.404 ns | 0.85 | 0.01 | 2,975 B | - |
| PixelOperations_Base | 256 | 214.03 ns | 70.798 ns | 3.881 ns | 1.00 | 0.02 | 1,152 B | - |
| PixelOperations_Specialized | 256 | 70.21 ns | 17.210 ns | 0.943 ns | 0.33 | 0.01 | 2,992 B | - |
| PixelOperations_Base | 2048 | 1,855.02 ns | 443.677 ns | 24.319 ns | 1.00 | 0.02 | 1,152 B | - |
| PixelOperations_Specialized | 2048 | 272.82 ns | 39.302 ns | 2.154 ns | 0.15 | 0.00 | 2,992 B | - |
| Method | Count | Mean | Error | StdDev | Ratio | Code Size | Allocated |
|---------------------------- |------ |------------:|---------:|----------:|------:|----------:|----------:|
| PixelOperations_Base | 64 | 53.35 ns | 0.208 ns | 0.311 ns | 1.00 | 1,152 B | - |
| PixelOperations_Specialized | 64 | 46.61 ns | 0.195 ns | 0.280 ns | 0.87 | 2,975 B | - |
| PixelOperations_Base | 256 | 212.11 ns | 0.719 ns | 1.076 ns | 1.00 | 1,152 B | - |
| PixelOperations_Specialized | 256 | 77.12 ns | 0.824 ns | 1.233 ns | 0.36 | 2,992 B | - |
| PixelOperations_Base | 2048 | 1,435.87 ns | 9.402 ns | 13.781 ns | 1.00 | 1,152 B | - |
| PixelOperations_Specialized | 2048 | 250.89 ns | 9.354 ns | 13.416 ns | 0.17 | 2,992 B | - |
*/
}

16
tests/ImageSharp.Benchmarks/Bulk/ToVector4_Bgra32.cs

@ -56,12 +56,12 @@ public class ToVector4_Bgra32P : ToVector4<Bgra32P>
// AMD RYZEN AI MAX+ 395 w/ Radeon 8060S 3.00GHz, 1 CPU, 32 logical and 16 physical cores
// .NET 8.0.28, X64 RyuJIT x86-64-v4
//
// | Method | Count | Mean | Error | StdDev | Ratio | Code Size | Allocated |
// |---------------------------- |------ |------------:|-----------:|---------:|------:|----------:|----------:|
// | PixelOperations_Base | 64 | 58.08 ns | 10.122 ns | 0.555 ns | 1.00 | 932 B | - |
// | PixelOperations_Specialized | 64 | 79.63 ns | 9.080 ns | 0.498 ns | 1.37 | 3,688 B | - |
// | PixelOperations_Base | 256 | 224.99 ns | 46.742 ns | 2.562 ns | 1.00 | 958 B | - |
// | PixelOperations_Specialized | 256 | 99.41 ns | 9.728 ns | 0.533 ns | 0.44 | 3,688 B | - |
// | PixelOperations_Base | 2048 | 1,745.77 ns | 143.244 ns | 7.852 ns | 1.00 | 958 B | - |
// | PixelOperations_Specialized | 2048 | 293.00 ns | 44.137 ns | 2.419 ns | 0.17 | 3,704 B | - |
// | Method | Count | Mean | Error | StdDev | Ratio | RatioSD | Code Size | Allocated |
// |---------------------------- |------ |------------:|---------:|---------:|------:|--------:|----------:|----------:|
// | PixelOperations_Base | 64 | 58.05 ns | 3.204 ns | 4.795 ns | 1.01 | 0.11 | 932 B | - |
// | PixelOperations_Specialized | 64 | 84.78 ns | 0.467 ns | 0.669 ns | 1.47 | 0.12 | 3,731 B | - |
// | PixelOperations_Base | 256 | 208.93 ns | 0.345 ns | 0.484 ns | 1.00 | 0.00 | 958 B | - |
// | PixelOperations_Specialized | 256 | 106.50 ns | 0.280 ns | 0.410 ns | 0.51 | 0.00 | 3,738 B | - |
// | PixelOperations_Base | 2048 | 1,617.39 ns | 3.380 ns | 4.848 ns | 1.00 | 0.00 | 958 B | - |
// | PixelOperations_Specialized | 2048 | 269.90 ns | 0.666 ns | 0.976 ns | 0.17 | 0.00 | 3,735 B | - |
}

31
tests/ImageSharp.Tests/Color/ColorTests.cs

@ -1,6 +1,7 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using System.Numerics;
using SixLabors.ImageSharp.PixelFormats;
namespace SixLabors.ImageSharp.Tests;
@ -40,6 +41,36 @@ public partial class ColorTests
Assert.Equal("80402080", color.ToHex());
}
[Fact]
public void ToScaledVector4ConvertsUnassociatedToAssociated()
{
Color color = Color.FromScaledVector(new Vector4(1F, 0.5F, 0.25F, 0.5F));
Vector4 actual = color.ToScaledVector4(PixelAlphaRepresentation.Associated);
Assert.Equal(new Vector4(0.5F, 0.25F, 0.125F, 0.5F), actual);
}
[Fact]
public void ToScaledVector4ConvertsAssociatedToUnassociated()
{
Color color = Color.FromScaledVector(new Vector4(0.5F, 0.25F, 0.125F, 0.5F), PixelAlphaRepresentation.Associated);
Vector4 actual = color.ToScaledVector4(PixelAlphaRepresentation.Unassociated);
Assert.Equal(new Vector4(1F, 0.5F, 0.25F, 0.5F), actual);
}
[Fact]
public void EqualityDoesNotDependOnInternalAssociationStorage()
{
Color fromPixel = Color.FromPixel(new Rgba32P(64, 32, 16, 128));
Color fromVector = Color.FromScaledVector(fromPixel.ToScaledVector4(), PixelAlphaRepresentation.Associated);
Assert.Equal(fromPixel, fromVector);
Assert.Equal(fromPixel.GetHashCode(), fromVector.GetHashCode());
}
[Theory]
[InlineData(false)]
[InlineData(true)]

32
tests/ImageSharp.Tests/Common/SimdUtilsTests.cs

@ -3,8 +3,10 @@
using System.Numerics;
using System.Runtime.CompilerServices;
using System.Runtime.Intrinsics;
using System.Runtime.Intrinsics.Arm;
using System.Runtime.Intrinsics.X86;
using SixLabors.ImageSharp.Common.Helpers;
using SixLabors.ImageSharp.PixelFormats;
using SixLabors.ImageSharp.Tests.TestUtilities;
using Xunit.Abstractions;
@ -133,7 +135,7 @@ public partial class SimdUtilsTests
FeatureTestRunner.RunWithHwIntrinsicsFeature(
RunTest,
count,
HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX2);
HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX);
}
[Theory]
@ -142,13 +144,37 @@ public partial class SimdUtilsTests
count,
(s, d) => SimdUtils.ByteToNormalizedFloat(s.Span, d.Span));
[Fact]
public void VectorMultiplyAddUsesRuntimeOrderAndFusedContract() =>
FeatureTestRunner.RunWithHwIntrinsicsFeature(
RunVectorMultiplyAddUsesRuntimeOrderAndFusedContract,
HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic);
private static void RunVectorMultiplyAddUsesRuntimeOrderAndFusedContract()
{
Assert.Equal(Vector128.Create(11F), Vector128_.MultiplyAddEstimate(Vector128.Create(2F), Vector128.Create(3F), Vector128.Create(5F)));
Assert.Equal(Vector256.Create(11F), Vector256_.MultiplyAddEstimate(Vector256.Create(2F), Vector256.Create(3F), Vector256.Create(5F)));
Assert.Equal(Vector512.Create(11F), Vector512_.MultiplyAddEstimate(Vector512.Create(2F), Vector512.Create(3F), Vector512.Create(5F)));
// These associated-alpha components produce an exact midpoint that a separate multiply and add rounds incorrectly.
float left = (68F / byte.MaxValue) * .625F;
float right = 1F - (160F / byte.MaxValue);
float addend = ((50F / byte.MaxValue) + (50F / byte.MaxValue)) * left;
float expected = MathF.FusedMultiplyAdd(left, right, addend);
Assert.Equal(Vector128.Create(expected), Vector128_.FusedMultiplyAdd(Vector128.Create(left), Vector128.Create(right), Vector128.Create(addend)));
Assert.Equal(Vector256.Create(expected), Vector256_.FusedMultiplyAdd(Vector256.Create(left), Vector256.Create(right), Vector256.Create(addend)));
Assert.Equal(Vector512.Create(expected), Vector512_.FusedMultiplyAdd(Vector512.Create(left), Vector512.Create(right), Vector512.Create(addend)));
}
private static void TestImpl_BulkConvertByteToNormalizedFloat(
int count,
Action<Memory<byte>, Memory<float>> convert)
{
byte[] source = new Random(count).GenerateRandomByteArray(count);
byte[] source = Enumerable.Range(0, count).Select(i => (byte)i).ToArray();
float[] result = new float[count];
float[] expected = source.Select(b => b * (1F / byte.MaxValue)).ToArray();
// A double-precision oracle keeps this expectation independent from either production implementation.
float[] expected = source.Select(b => (float)(b / (double)byte.MaxValue)).ToArray();
convert(source, result);

7
tests/ImageSharp.Tests/Formats/Icon/Cur/CurEncoderTests.cs

@ -4,7 +4,6 @@
using SixLabors.ImageSharp.Formats;
using SixLabors.ImageSharp.Formats.Cur;
using SixLabors.ImageSharp.Formats.Ico;
using SixLabors.ImageSharp.Formats.Png;
using SixLabors.ImageSharp.PixelFormats;
using SixLabors.ImageSharp.Tests.TestUtilities.ImageComparison;
using static SixLabors.ImageSharp.Tests.TestImages.Cur;
@ -120,12 +119,6 @@ public class CurEncoderTests
for (int x = 0; x < accessor.Width; x++)
{
if (expectedColor != rowSpan[x])
{
int xx = 0;
}
Assert.Equal(expectedColor, rowSpan[x]);
}
}

2
tests/ImageSharp.Tests/Formats/Qoi/ImageExtensionsTest.cs

@ -1,8 +1,8 @@
// Copyright (c) Six Labors.
// Licensed under the Six Labors Split License.
using SixLabors.ImageSharp.Formats.Qoi;
using SixLabors.ImageSharp.Formats;
using SixLabors.ImageSharp.Formats.Qoi;
using SixLabors.ImageSharp.PixelFormats;
namespace SixLabors.ImageSharp.Tests.Formats.Qoi;

312
tests/ImageSharp.Tests/PixelFormats/AssociatedAlphaPixelTests.cs

@ -5,7 +5,6 @@ using System.Numerics;
using System.Runtime.CompilerServices;
using System.Runtime.InteropServices;
using SixLabors.ImageSharp.PixelFormats;
using SixLabors.ImageSharp.PixelFormats.PixelBlenders;
namespace SixLabors.ImageSharp.Tests.PixelFormats;
@ -17,8 +16,6 @@ namespace SixLabors.ImageSharp.Tests.PixelFormats;
public abstract class AssociatedAlphaPixelTests<TPixel>
where TPixel : unmanaged, IPixel<TPixel>
{
private static readonly ApproximateFloatComparer VectorComparer = new(.005F);
/// <summary>
/// Gets the color channels described by the pixel format.
/// </summary>
@ -42,75 +39,6 @@ public abstract class AssociatedAlphaPixelTests<TPixel>
Assert.Equal(expectedComponentPrecision, componentInfo.GetComponentPrecision(i));
}
}
[Fact]
public void ScaledVectorConversionsUseAssociatedComponents()
{
Vector4 associated = new(.25F, .125F, .0625F, .5F);
TPixel pixel = TPixel.FromScaledVector4(associated);
Assert.Equal(associated, pixel.ToScaledVector4(), VectorComparer);
}
[Fact]
public void FromRgba32AssociatesColorComponents()
{
Rgba32 source = new(192, 128, 64, 128);
Vector4 expected = source.ToScaledVector4();
Numerics.Premultiply(ref expected);
TPixel pixel = TPixel.FromRgba32(source);
Assert.Equal(expected, pixel.ToScaledVector4(), VectorComparer);
}
[Fact]
public void ToRgba32ReturnsUnassociatedColorComponents()
{
Rgba32 expected = new(192, 128, 64, 128);
TPixel pixel = TPixel.FromRgba32(expected);
Rgba32 actual = pixel.ToRgba32();
AssertRgba32Equal(expected, actual, 3);
}
[Fact]
public void ColorConversionsPreserveUnassociatedColor()
{
Rgba32 expected = new(192, 128, 64, 128);
TPixel source = TPixel.FromRgba32(expected);
Color color = Color.FromPixel(source);
Rgba32 actual = color.ToPixel<Rgba32>();
TPixel roundTrip = color.ToPixel<TPixel>();
AssertRgba32Equal(expected, actual, 3);
Assert.Equal(source.ToScaledVector4(), roundTrip.ToScaledVector4(), VectorComparer);
}
[Fact]
public void ScalarBlendingUsesUnassociatedColorValues()
{
TPixel background = TPixel.FromRgba32(new Rgba32(200, 40, 80, 160));
TPixel source = TPixel.FromRgba32(new Rgba32(20, 180, 100, 96));
PixelBlender<TPixel> associatedBlender = PixelOperations<TPixel>.Instance.GetPixelBlender(PixelColorBlendingMode.Normal, PixelAlphaCompositionMode.SrcOver);
PixelBlender<Rgba32> unassociatedBlender = new DefaultPixelBlenders<Rgba32>.NormalSrcOver();
Rgba32 expected = unassociatedBlender.Blend(background.ToRgba32(), source.ToRgba32(), .75F);
Rgba32 actual = associatedBlender.Blend(background, source, .75F).ToRgba32();
AssertRgba32Equal(expected, actual, 4);
}
private static void AssertRgba32Equal(Rgba32 expected, Rgba32 actual, int tolerance)
{
Assert.InRange(Math.Abs(expected.R - actual.R), 0, tolerance);
Assert.InRange(Math.Abs(expected.G - actual.G), 0, tolerance);
Assert.InRange(Math.Abs(expected.B - actual.B), 0, tolerance);
Assert.InRange(Math.Abs(expected.A - actual.A), 0, tolerance);
}
}
/// <summary>
@ -205,6 +133,29 @@ public class HalfVector4PTests : AssociatedAlphaPixelTests<HalfVector4P>
Assert.Equal(new HalfVector4(associated).PackedValue, new HalfVector4P(associated).PackedValue);
}
[Fact]
public void ColorRoundTripDoesNotIntroduceAdditionalAssociationLoss()
{
const ulong zeroComponent = 0xBC00;
// This is the first valid Half component/alpha pair for which an unassociate/reassociate round trip
// selects the adjacent component value instead of the direct scaled-vector conversion result.
HalfVector4P source = new()
{
PackedValue = 0x8BF5 | (zeroComponent << 16) | (zeroComponent << 32) | (0x05DFUL << 48)
};
HalfVector4P expected = HalfVector4P.FromScaledVector4(source.ToScaledVector4());
Color color = Color.FromPixel(source);
Color[] bulkColors = new Color[1];
Color.FromPixel<HalfVector4P>(new[] { source }, bulkColors);
Assert.Equal(source.ToScaledVector4(), color.ToScaledVector4());
Assert.Equal(expected, color.ToPixel<HalfVector4P>());
Assert.Equal(source.ToScaledVector4(), bulkColors[0].ToScaledVector4());
Assert.Equal(expected, bulkColors[0].ToPixel<HalfVector4P>());
}
}
/// <summary>
@ -248,17 +199,19 @@ public class AssociatedAlphaPackedPixelConversionTests
private static void AssertLosslessRoundTrip<TIntermediate>()
where TIntermediate : unmanaged, IPixel<TIntermediate>
{
Rgba32P[] expected =
[
new(0, 0, 0, 0),
new(1, 2, 3, 4),
new(31, 63, 95, 127),
new(64, 128, 192, 255),
new(255, 255, 255, 255),
];
Rgba32P[] expected = new Rgba32P[byte.MaxValue * 4 + 4];
TIntermediate[] intermediate = new TIntermediate[expected.Length];
Rgba32P[] actual = new Rgba32P[expected.Length];
for (int component = 0; component <= byte.MaxValue; component++)
{
int index = component * 4;
expected[index] = new Rgba32P((byte)component, 0, 0, byte.MaxValue);
expected[index + 1] = new Rgba32P(0, (byte)component, 0, byte.MaxValue);
expected[index + 2] = new Rgba32P(0, 0, (byte)component, byte.MaxValue);
expected[index + 3] = new Rgba32P(0, 0, 0, (byte)component);
}
PixelOperations<TIntermediate>.Instance.From<Rgba32P>(Configuration.Default, expected, intermediate);
PixelOperations<Rgba32P>.Instance.From<TIntermediate>(Configuration.Default, intermediate, actual);
@ -273,13 +226,16 @@ public class AssociatedAlphaPackedPixelConversionTests
private static void AssertScalarAndBulkAssociatedVectorsAreEqual<TPixel>(Func<byte, byte, byte, byte, TPixel> createPixel)
where TPixel : unmanaged, IPixel<TPixel>
{
TPixel[] pixels = new TPixel[64];
TPixel[] pixels = new TPixel[byte.MaxValue * 4 + 4];
Vector4[] actual = new Vector4[pixels.Length];
for (int i = 0; i < pixels.Length; i++)
for (int component = 0; component <= byte.MaxValue; component++)
{
int component = i * 4;
pixels[i] = createPixel((byte)component, (byte)(component + 1), (byte)(component + 2), (byte)(component + 3));
int index = component * 4;
pixels[index] = createPixel((byte)component, 0, 0, byte.MaxValue);
pixels[index + 1] = createPixel(0, (byte)component, 0, byte.MaxValue);
pixels[index + 2] = createPixel(0, 0, (byte)component, byte.MaxValue);
pixels[index + 3] = createPixel(0, 0, 0, (byte)component);
}
PixelOperations<TPixel>.Instance.ToAssociatedScaledVector4(Configuration.Default, pixels, actual);
@ -319,6 +275,110 @@ public class AssociatedAlphaPackedPixelConversionTests
/// </summary>
public class AssociatedToUnassociatedPackedPixelConversionTests
{
[Fact]
public void Rgba32ToRgba32PMatchesExactAssociationForEveryComponentAndAlpha()
=> AssertUnsignedByteAssociation((red, green, blue, alpha) => new Rgba32P(red, green, blue, alpha));
[Fact]
public void Rgba32ToBgra32PMatchesExactAssociationForEveryComponentAndAlpha()
=> AssertUnsignedByteAssociation((red, green, blue, alpha) => new Bgra32P(red, green, blue, alpha));
[Fact]
public void Rgba32ToArgb32PMatchesExactAssociationForEveryComponentAndAlpha()
=> AssertUnsignedByteAssociation((red, green, blue, alpha) => new Argb32P(red, green, blue, alpha));
[Fact]
public void Rgba32ToAbgr32PMatchesExactAssociationForEveryComponentAndAlpha()
=> AssertUnsignedByteAssociation((red, green, blue, alpha) => new Abgr32P(red, green, blue, alpha));
[Fact]
public void Rgba32ToNormalizedByte4PMatchesExactAssociationForEveryComponentAndAlpha()
{
const int pairCount = 65536;
const int channelCount = 3;
Rgba32[] source = new Rgba32[pairCount * channelCount];
NormalizedByte4P[] expected = new NormalizedByte4P[source.Length];
NormalizedByte4P[] actualBulk = new NormalizedByte4P[source.Length];
int index = 0;
for (int alpha = 0; alpha <= byte.MaxValue; alpha++)
{
// NormalizedByte4 has 254 intervals from -1 through 1, so alpha is first rounded to that destination grid.
int destinationAlpha = ((alpha * 254) + 127) / byte.MaxValue;
for (int unassociated = 0; unassociated <= byte.MaxValue; unassociated++)
{
// Association uses the alpha representable by the destination, then rounds the result to the same 254-interval grid.
int associated = ((unassociated * destinationAlpha) + 127) / byte.MaxValue;
source[index] = new Rgba32((byte)unassociated, 0, 0, (byte)alpha);
expected[index++] = CreateNormalizedByte4P(associated, 0, 0, destinationAlpha);
source[index] = new Rgba32(0, (byte)unassociated, 0, (byte)alpha);
expected[index++] = CreateNormalizedByte4P(0, associated, 0, destinationAlpha);
source[index] = new Rgba32(0, 0, (byte)unassociated, (byte)alpha);
expected[index++] = CreateNormalizedByte4P(0, 0, associated, destinationAlpha);
}
}
PixelOperations<NormalizedByte4P>.Instance.From<Rgba32>(Configuration.Default, source, actualBulk);
Assert.Equal(expected, actualBulk);
for (int i = 0; i < source.Length; i++)
{
Assert.Equal(expected[i], NormalizedByte4P.FromRgba32(source[i]));
Assert.Equal(expected[i], Color.FromPixel(source[i]).ToPixel<NormalizedByte4P>());
}
}
[Fact]
public void Rgba32ToHalfVector4PMatchesExactAssociationForEveryComponentAndAlpha()
{
const int pairCount = 65536;
const int channelCount = 3;
// HalfVector4 maps scaled zero to native -1, whose IEEE 754 binary16 representation is 0xBC00.
const ulong zeroComponentBits = 0xBC00;
Rgba32[] source = new Rgba32[pairCount * channelCount];
HalfVector4P[] expected = new HalfVector4P[source.Length];
HalfVector4P[] actualBulk = new HalfVector4P[source.Length];
int index = 0;
for (int alpha = 0; alpha <= byte.MaxValue; alpha++)
{
// HalfVector4 stores normalized values over -1 through 1. Association must use the alpha value
// recovered from the destination half, rather than the higher-precision source alpha.
float normalizedAlpha = (float)(alpha / (double)byte.MaxValue);
float nativeAlpha = (normalizedAlpha * 2F) - 1F;
ushort alphaBits = BitConverter.HalfToUInt16Bits((Half)nativeAlpha);
float destinationAlpha = ((float)BitConverter.UInt16BitsToHalf(alphaBits) + 1F) * .5F;
for (int unassociated = 0; unassociated <= byte.MaxValue; unassociated++)
{
float normalizedComponent = (float)(unassociated / (double)byte.MaxValue);
float associated = normalizedComponent * destinationAlpha;
ushort associatedBits = BitConverter.HalfToUInt16Bits((Half)((associated * 2F) - 1F));
ulong alphaPacked = (ulong)alphaBits << 48;
source[index] = new Rgba32((byte)unassociated, 0, 0, (byte)alpha);
expected[index++] = new HalfVector4P { PackedValue = associatedBits | (zeroComponentBits << 16) | (zeroComponentBits << 32) | alphaPacked };
source[index] = new Rgba32(0, (byte)unassociated, 0, (byte)alpha);
expected[index++] = new HalfVector4P { PackedValue = zeroComponentBits | ((ulong)associatedBits << 16) | (zeroComponentBits << 32) | alphaPacked };
source[index] = new Rgba32(0, 0, (byte)unassociated, (byte)alpha);
expected[index++] = new HalfVector4P { PackedValue = zeroComponentBits | (zeroComponentBits << 16) | ((ulong)associatedBits << 32) | alphaPacked };
}
}
PixelOperations<HalfVector4P>.Instance.From<Rgba32>(Configuration.Default, source, actualBulk);
for (int i = 0; i < source.Length; i++)
{
Assert.Equal(expected[i], HalfVector4P.FromRgba32(source[i]));
Assert.Equal(expected[i], Color.FromPixel(source[i]).ToPixel<HalfVector4P>());
Assert.Equal(expected[i], actualBulk[i]);
}
}
[Fact]
public void Rgba32PToRgba32ScalarRoundTripPreservesEveryValidAssociatedComponent()
{
@ -344,6 +404,11 @@ public class AssociatedToUnassociatedPackedPixelConversionTests
[Fact]
public void ColorFromRgba32PPreservesEveryValidAssociatedComponent()
{
const int pairCount = 32896;
Rgba32P[] sourcePixels = new Rgba32P[pairCount];
Color[] bulkColors = new Color[pairCount];
int index = 0;
for (int alpha = 0; alpha <= byte.MaxValue; alpha++)
{
for (int associated = 0; associated <= alpha; associated++)
@ -351,11 +416,19 @@ public class AssociatedToUnassociatedPackedPixelConversionTests
byte unassociated = alpha == 0 ? (byte)0 : (byte)(((associated * byte.MaxValue) + (alpha / 2)) / alpha);
Rgba32P source = new((byte)associated, 0, 0, (byte)alpha);
Color color = Color.FromPixel(source);
sourcePixels[index++] = source;
Assert.Equal(new Rgba32(unassociated, 0, 0, (byte)alpha), color.ToPixel<Rgba32>());
Assert.Equal(source, color.ToPixel<Rgba32P>());
}
}
Color.FromPixel<Rgba32P>(sourcePixels, bulkColors);
for (int i = 0; i < sourcePixels.Length; i++)
{
Assert.Equal(sourcePixels[i], bulkColors[i].ToPixel<Rgba32P>());
}
}
[Fact]
@ -397,6 +470,11 @@ public class AssociatedToUnassociatedPackedPixelConversionTests
[Fact]
public void ColorFromNormalizedByte4PPreservesEveryValidAssociatedComponent()
{
const int pairCount = 32640;
NormalizedByte4P[] sourcePixels = new NormalizedByte4P[pairCount];
Color[] bulkColors = new Color[pairCount];
int index = 0;
for (int alpha = 0; alpha < byte.MaxValue; alpha++)
{
for (int associated = 0; associated <= alpha; associated++)
@ -405,11 +483,19 @@ public class AssociatedToUnassociatedPackedPixelConversionTests
byte unassociatedAlpha = (byte)(((alpha * byte.MaxValue) + 127) / 254);
NormalizedByte4P source = CreateNormalizedByte4P(associated, 0, 0, alpha);
Color color = Color.FromPixel(source);
sourcePixels[index++] = source;
Assert.Equal(new Rgba32(unassociated, 0, 0, unassociatedAlpha), color.ToPixel<Rgba32>());
Assert.Equal(source, color.ToPixel<NormalizedByte4P>());
}
}
Color.FromPixel<NormalizedByte4P>(sourcePixels, bulkColors);
for (int i = 0; i < sourcePixels.Length; i++)
{
Assert.Equal(sourcePixels[i], bulkColors[i].ToPixel<NormalizedByte4P>());
}
}
[Fact]
@ -444,6 +530,15 @@ public class AssociatedToUnassociatedPackedPixelConversionTests
Assert.Equal(expectedUnassociated, actualUnassociated);
Assert.Equal(source, actualRoundTrip);
for (int i = 0; i < source.Length; i++)
{
Bgra32 expected = expectedUnassociated[i];
// Scalar conversion and Color must preserve the same exact canonical value proven for the bulk path.
Assert.Equal(new Rgba32(expected.R, expected.G, expected.B, expected.A), source[i].ToRgba32());
Assert.Equal(expected, Color.FromPixel(source[i]).ToPixel<Bgra32>());
}
}
private static void AssertUnsignedByteBulkRoundTrip<TPixel>(Func<byte, byte, byte, byte, TPixel> createPixel)
@ -479,6 +574,53 @@ public class AssociatedToUnassociatedPackedPixelConversionTests
Assert.Equal(expectedUnassociated, actualUnassociated);
Assert.Equal(source, actualRoundTrip);
for (int i = 0; i < source.Length; i++)
{
Bgra32 expected = expectedUnassociated[i];
// Scalar conversion and Color must use the same exact canonical byte as the independently calculated bulk oracle.
Assert.Equal(new Rgba32(expected.R, expected.G, expected.B, expected.A), source[i].ToRgba32());
Assert.Equal(expected, Color.FromPixel(source[i]).ToPixel<Bgra32>());
}
}
private static void AssertUnsignedByteAssociation<TPixel>(Func<byte, byte, byte, byte, TPixel> createPixel)
where TPixel : unmanaged, IPixel<TPixel>
{
const int pairCount = 65536;
const int channelCount = 3;
Rgba32[] source = new Rgba32[pairCount * channelCount];
TPixel[] expected = new TPixel[source.Length];
TPixel[] actualBulk = new TPixel[source.Length];
int index = 0;
for (int alpha = 0; alpha <= byte.MaxValue; alpha++)
{
for (int unassociated = 0; unassociated <= byte.MaxValue; unassociated++)
{
// The destination stores round(unassociated * alpha / 255). The integer numerator adds half
// the divisor so the expected value is independent of the floating-point implementation.
byte associated = (byte)(((unassociated * alpha) + 127) / byte.MaxValue);
source[index] = new Rgba32((byte)unassociated, 0, 0, (byte)alpha);
expected[index++] = createPixel(associated, 0, 0, (byte)alpha);
source[index] = new Rgba32(0, (byte)unassociated, 0, (byte)alpha);
expected[index++] = createPixel(0, associated, 0, (byte)alpha);
source[index] = new Rgba32(0, 0, (byte)unassociated, (byte)alpha);
expected[index++] = createPixel(0, 0, associated, (byte)alpha);
}
}
PixelOperations<TPixel>.Instance.From<Rgba32>(Configuration.Default, source, actualBulk);
Assert.Equal(expected, actualBulk);
for (int i = 0; i < source.Length; i++)
{
Assert.Equal(expected[i], TPixel.FromRgba32(source[i]));
Assert.Equal(expected[i], Color.FromPixel(source[i]).ToPixel<TPixel>());
}
}
private static NormalizedByte4P CreateNormalizedByte4P(int red, int green, int blue, int alpha)
@ -638,7 +780,7 @@ public class AssociatedDestinationAlphaQuantizationTests
foreach (byte component in components)
{
int expectedRed = ((component * expectedAlpha) + 127) / byte.MaxValue;
Vector4 expected = new Vector4(expectedRed, 0, 0, expectedAlpha) * (1F / byte.MaxValue);
Vector4 expected = new Vector4(expectedRed, 0, 0, expectedAlpha) / byte.MaxValue;
Assert.Equal(expected, TPixel.FromRgba64(source[index]).ToScaledVector4());
Assert.Equal(expected, actualBulk[index].ToScaledVector4());

7
tests/ImageSharp.Tests/PixelFormats/PixelBlenderTests.cs

@ -229,7 +229,12 @@ public class PixelBlenderTests
HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic);
[Fact]
public void AssociatedHardLightDestAtopRoundsExactMidpointAwayFromZero()
public void AssociatedHardLightDestAtopRoundsExactMidpointAwayFromZero() =>
FeatureTestRunner.RunWithHwIntrinsicsFeature(
RunAssociatedHardLightDestAtopRoundsExactMidpointAwayFromZero,
HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic);
private static void RunAssociatedHardLightDestAtopRoundsExactMidpointAwayFromZero()
{
Rgba32P background = Rgba32P.FromRgba32(new Rgba32(220, 80, 40, 160));
Rgba32P source = Rgba32P.FromRgba32(new Rgba32(20, 180, 120, 96));

4
tests/Images/External/ReferenceOutput/Filters/BlackWhiteTest/ApplyBlackWhiteFilter_Rgba32_TestPattern48x48.png

@ -1,3 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:65d3a099dd24fcee2fc86c63b8664ebd81d7b3c530168da5f4e50ee0ca925496
size 759
oid sha256:d76ae5ac1a0aebb9d76b2fa5c6b1fc109e5fcbdc6c41d9681a76ade9c86fab79
size 748

Loading…
Cancel
Save