From 4b07920c7c1ea48aa82ad31815954d9289698370 Mon Sep 17 00:00:00 2001 From: James Jackson-South Date: Wed, 15 Jul 2026 11:59:05 +1000 Subject: [PATCH] Fix alpha association and SIMD rounding parity Refactors `Color` to track both exposed and stored alpha representations, adds `ToScaledVector4(PixelAlphaRepresentation)`, and adjusts pixel conversion paths so associated formats preserve canonical values without unnecessary unpremultiply/reassociate loss. This also updates equality/hash behavior to compare canonical scaled values. SIMD helpers are split into `MultiplyAddEstimate` vs `FusedMultiplyAdd`, with byte-to-float normalization updated to match scalar rounding exactly across vector widths. JPEG converters, resize kernels, and Porter-Duff/associated-alpha blending paths were updated to use the appropriate helper for either fast estimate or strict fused rounding semantics. Tests were expanded substantially (including exhaustive component/alpha cases and fused-order checks), incidental test cleanup was applied, benchmark comment snapshots were refreshed, and one black/white reference output image was updated. --- src/ImageSharp/Color/Color.cs | 190 ++++++----- .../Common/Helpers/SimdUtils.Convert.cs | 3 +- .../Common/Helpers/SimdUtils.HwIntrinsics.cs | 53 ++- .../Common/Helpers/Vector128Utilities.cs | 54 ++- .../Common/Helpers/Vector256Utilities.cs | 45 ++- .../Common/Helpers/Vector512Utilities.cs | 50 ++- .../JpegColorConverter.GrayScaleVector128.cs | 2 +- .../JpegColorConverter.GrayScaleVector256.cs | 2 +- .../JpegColorConverter.GrayScaleVector512.cs | 2 +- .../JpegColorConverter.TiffYccKVector128.cs | 12 +- .../JpegColorConverter.TiffYccKVector256.cs | 12 +- .../JpegColorConverter.TiffYccKVector512.cs | 12 +- .../JpegColorConverter.YCbCrVector128.cs | 12 +- .../JpegColorConverter.YCbCrVector256.cs | 12 +- .../JpegColorConverter.YCbCrVector512.cs | 12 +- .../JpegColorConverter.YccKVector128.cs | 12 +- .../JpegColorConverter.YccKVector256.cs | 12 +- .../JpegColorConverter.YccKVector512.cs | 12 +- .../Components/FloatingPointDCT.Vector256.cs | 8 +- .../AssociatedAlphaPorterDuffFunctions.cs | 16 +- .../PixelBlenders/PorterDuffFunctions.cs | 26 +- .../PixelImplementations/Abgr32.cs | 2 +- .../PixelImplementations/Abgr32P.cs | 4 +- .../PixelImplementations/Argb32.cs | 2 +- .../PixelImplementations/Argb32P.cs | 4 +- .../PixelImplementations/Bgra32.cs | 2 +- .../PixelImplementations/Bgra32P.cs | 4 +- .../PixelImplementations/NormalizedByte4P.cs | 5 +- .../PixelImplementations/Rgba32.cs | 2 +- .../PixelImplementations/Rgba32P.cs | 4 +- .../Transforms/Resize/ResizeKernel.cs | 8 +- .../ImageSharp.Benchmarks/Bulk/FromVector4.cs | 16 +- .../Bulk/ToVector4_Bgra32.cs | 16 +- tests/ImageSharp.Tests/Color/ColorTests.cs | 31 ++ .../ImageSharp.Tests/Common/SimdUtilsTests.cs | 32 +- .../Formats/Icon/Cur/CurEncoderTests.cs | 7 - .../Formats/Qoi/ImageExtensionsTest.cs | 2 +- .../PixelFormats/AssociatedAlphaPixelTests.cs | 312 +++++++++++++----- .../PixelFormats/PixelBlenderTests.cs | 7 +- ...ackWhiteFilter_Rgba32_TestPattern48x48.png | 4 +- 40 files changed, 673 insertions(+), 350 deletions(-) diff --git a/src/ImageSharp/Color/Color.cs b/src/ImageSharp/Color/Color.cs index 7969f4728d..9ce0ecfc3a 100644 --- a/src/ImageSharp/Color/Color.cs +++ b/src/ImageSharp/Color/Color.cs @@ -20,45 +20,23 @@ namespace SixLabors.ImageSharp; public readonly partial struct Color : IEquatable { private readonly Vector4 data; - private readonly Vector4 unassociatedData; private readonly IPixel? boxedHighPrecisionPixel; private readonly bool isAssociated; + private readonly bool dataIsAssociated; /// /// Initializes a new instance of the struct. /// /// The containing the color information. - /// The alpha representation of . + /// The alpha representation exposed by the color. + /// The alpha representation of . [MethodImpl(MethodImplOptions.AggressiveInlining)] - private Color(Vector4 vector, PixelAlphaRepresentation alphaRepresentation) - { - Vector4 clamped = Numerics.Clamp(vector, Vector4.Zero, Vector4.One); - Vector4 unassociated = clamped; - - if (alphaRepresentation == PixelAlphaRepresentation.Associated) - { - Numerics.UnPremultiply(ref unassociated); - } - - this.data = clamped; - this.unassociatedData = unassociated; - this.boxedHighPrecisionPixel = null; - this.isAssociated = alphaRepresentation == PixelAlphaRepresentation.Associated; - } - - /// - /// Initializes a new instance of the struct. - /// - /// The containing the color information. - /// The unassociated representation of . - /// The alpha representation of . - [MethodImpl(MethodImplOptions.AggressiveInlining)] - private Color(Vector4 vector, Vector4 unassociatedVector, PixelAlphaRepresentation alphaRepresentation) + private Color(Vector4 vector, PixelAlphaRepresentation alphaRepresentation, PixelAlphaRepresentation dataAlphaRepresentation) { this.data = Numerics.Clamp(vector, Vector4.Zero, Vector4.One); - this.unassociatedData = Numerics.Clamp(unassociatedVector, Vector4.Zero, Vector4.One); this.boxedHighPrecisionPixel = null; this.isAssociated = alphaRepresentation == PixelAlphaRepresentation.Associated; + this.dataIsAssociated = dataAlphaRepresentation == PixelAlphaRepresentation.Associated; } /// @@ -71,8 +49,8 @@ public readonly partial struct Color : IEquatable { this.boxedHighPrecisionPixel = pixel; this.data = default; - this.unassociatedData = default; this.isAssociated = alphaRepresentation == PixelAlphaRepresentation.Associated; + this.dataIsAssociated = this.isAssociated; } /// @@ -118,14 +96,22 @@ public readonly partial struct Color : IEquatable // Avoid boxing in case we can convert to Vector4 safely and efficiently PixelTypeInfo info = TPixel.GetPixelTypeInfo(); - if (info.ComponentInfo.HasValue && info.ComponentInfo.Value.GetMaximumComponentPrecision() <= (int)PixelComponentBitDepth.Bit32) + if (info.ComponentInfo.HasValue) { - Vector4 vector = source.ToScaledVector4(); - Vector4 unassociated = info.AlphaRepresentation == PixelAlphaRepresentation.Associated - ? PixelOperations.Instance.ToUnassociatedScaledVector4(source) - : vector; + int maximumComponentPrecision = info.ComponentInfo.Value.GetMaximumComponentPrecision(); - return new Color(vector, unassociated, info.AlphaRepresentation); + if (maximumComponentPrecision <= (int)PixelComponentBitDepth.Bit32) + { + if (info.AlphaRepresentation == PixelAlphaRepresentation.Associated && maximumComponentPrecision <= (int)PixelComponentBitDepth.Bit8) + { + // Associated formats with at most eight bits per component can be canonicalized without loss by their pixel-specific conversion. + // Higher-precision formats retain their associated values because unassociation can lose representable data. + Vector4 vector = PixelOperations.Instance.ToUnassociatedScaledVector4(source); + return new Color(vector, info.AlphaRepresentation, PixelAlphaRepresentation.Unassociated); + } + + return new Color(source.ToScaledVector4(), info.AlphaRepresentation, info.AlphaRepresentation); + } } return new Color(source, info.AlphaRepresentation); @@ -137,7 +123,8 @@ public readonly partial struct Color : IEquatable /// The unassociated vector to load the color from. /// The . [MethodImpl(MethodImplOptions.AggressiveInlining)] - public static Color FromScaledVector(Vector4 source) => new(source, PixelAlphaRepresentation.Unassociated); + public static Color FromScaledVector(Vector4 source) + => new(source, PixelAlphaRepresentation.Unassociated, PixelAlphaRepresentation.Unassociated); /// /// Creates a from a generic scaled with the specified alpha representation. @@ -146,7 +133,8 @@ public readonly partial struct Color : IEquatable /// The alpha representation of . /// The . [MethodImpl(MethodImplOptions.AggressiveInlining)] - public static Color FromScaledVector(Vector4 source, PixelAlphaRepresentation alphaRepresentation) => new(source, alphaRepresentation); + public static Color FromScaledVector(Vector4 source, PixelAlphaRepresentation alphaRepresentation) + => new(source, alphaRepresentation, alphaRepresentation); /// /// Bulk converts a span of generic scaled to a span of . @@ -190,23 +178,38 @@ public readonly partial struct Color : IEquatable // Avoid boxing in case we can convert to Vector4 safely and efficiently PixelTypeInfo info = TPixel.GetPixelTypeInfo(); - if (info.ComponentInfo.HasValue && info.ComponentInfo.Value.GetMaximumComponentPrecision() <= (int)PixelComponentBitDepth.Bit32) + if (info.ComponentInfo.HasValue) { - bool isAssociated = info.AlphaRepresentation == PixelAlphaRepresentation.Associated; + int maximumComponentPrecision = info.ComponentInfo.Value.GetMaximumComponentPrecision(); - for (int i = 0; i < source.Length; i++) + if (maximumComponentPrecision <= (int)PixelComponentBitDepth.Bit32) { - Vector4 vector = source[i].ToScaledVector4(); - Vector4 unassociated = isAssociated ? PixelOperations.Instance.ToUnassociatedScaledVector4(source[i]) : vector; - destination[i] = new Color(vector, unassociated, info.AlphaRepresentation); + if (info.AlphaRepresentation == PixelAlphaRepresentation.Associated && maximumComponentPrecision <= (int)PixelComponentBitDepth.Bit8) + { + // Match the scalar conversion by retaining exact unassociated values from the format-specific operation. + PixelOperations operations = PixelOperations.Instance; + + for (int i = 0; i < source.Length; i++) + { + Vector4 vector = operations.ToUnassociatedScaledVector4(source[i]); + destination[i] = new Color(vector, info.AlphaRepresentation, PixelAlphaRepresentation.Unassociated); + } + + return; + } + + for (int i = 0; i < source.Length; i++) + { + destination[i] = new Color(source[i].ToScaledVector4(), info.AlphaRepresentation, info.AlphaRepresentation); + } + + return; } } - else + + for (int i = 0; i < source.Length; i++) { - for (int i = 0; i < source.Length; i++) - { - destination[i] = new Color(source[i], info.AlphaRepresentation); - } + destination[i] = new Color(source[i], info.AlphaRepresentation); } } @@ -349,26 +352,9 @@ public readonly partial struct Color : IEquatable [MethodImpl(MethodImplOptions.AggressiveInlining)] public Color WithAlpha(float alpha) { - Vector4 v = this.ToScaledVector4(); - float clampedAlpha = Numerics.Clamp(alpha, 0, 1); - - if (this.isAssociated) - { - Vector4 unassociated = this.boxedHighPrecisionPixel is null ? this.unassociatedData : v; - - if (this.boxedHighPrecisionPixel is not null) - { - Numerics.UnPremultiply(ref unassociated); - } - - unassociated.W = clampedAlpha; - Vector4 associated = unassociated; - Numerics.Premultiply(ref associated); - return new Color(associated, unassociated, PixelAlphaRepresentation.Associated); - } - - v.W = clampedAlpha; - return FromScaledVector(v, PixelAlphaRepresentation.Unassociated); + Vector4 vector = this.ToScaledVector4(PixelAlphaRepresentation.Unassociated); + vector.W = Numerics.Clamp(alpha, 0, 1); + return new Color(vector, this.AlphaRepresentation, PixelAlphaRepresentation.Unassociated); } /// @@ -411,17 +397,22 @@ public readonly partial struct Color : IEquatable return pixel; } - Vector4 vector = this.boxedHighPrecisionPixel is null - ? this.isAssociated ? this.unassociatedData : this.data - : this.boxedHighPrecisionPixel.ToScaledVector4(); + Vector4 vector = this.boxedHighPrecisionPixel?.ToScaledVector4() ?? this.data; + PixelOperations operations = PixelOperations.Instance; - if (this.isAssociated && this.boxedHighPrecisionPixel is not null) + if (this.dataIsAssociated) { + if (operations is AssociatedAlphaPixelOperations associatedOperations) + { + // Preserve associated components directly while allowing the destination to quantize alpha to its own storage grid. + return associatedOperations.FromAssociatedScaledVector4(vector); + } + Numerics.UnPremultiply(ref vector); } - // Destination pixel operations own association because RGB must use the alpha value representable by that destination format. - return PixelOperations.Instance.FromUnassociatedScaledVector4(vector); + // Unassociated input lets an associated destination quantize alpha before it multiplies the color components. + return operations.FromUnassociatedScaledVector4(vector); } /// @@ -434,12 +425,55 @@ public readonly partial struct Color : IEquatable [MethodImpl(MethodImplOptions.AggressiveInlining)] public Vector4 ToScaledVector4() { - if (this.boxedHighPrecisionPixel is null) + Vector4 vector = this.boxedHighPrecisionPixel?.ToScaledVector4() ?? this.data; + + if (this.dataIsAssociated == this.isAssociated) + { + return vector; + } + + if (this.isAssociated) + { + Numerics.Premultiply(ref vector); + } + else + { + Numerics.UnPremultiply(ref vector); + } + + return vector; + } + + /// + /// Expands the color into a generic ("scaled") using the specified alpha representation, + /// with values scaled and clamped between 0 and 1. + /// + /// + /// The alpha representation to apply. returns color components + /// multiplied by alpha; other representations return color components independent of alpha. + /// + /// The . + [MethodImpl(MethodImplOptions.AggressiveInlining)] + public Vector4 ToScaledVector4(PixelAlphaRepresentation alphaRepresentation) + { + bool targetIsAssociated = alphaRepresentation == PixelAlphaRepresentation.Associated; + Vector4 vector = this.boxedHighPrecisionPixel?.ToScaledVector4() ?? this.data; + + if (this.dataIsAssociated == targetIsAssociated) + { + return vector; + } + + if (targetIsAssociated) { - return this.data; + Numerics.Premultiply(ref vector); + } + else + { + Numerics.UnPremultiply(ref vector); } - return this.boxedHighPrecisionPixel.ToScaledVector4(); + return vector; } /// @@ -467,7 +501,7 @@ public readonly partial struct Color : IEquatable { if (this.boxedHighPrecisionPixel is null && other.boxedHighPrecisionPixel is null) { - return this.data == other.data && this.isAssociated == other.isAssociated; + return this.isAssociated == other.isAssociated && this.ToScaledVector4() == other.ToScaledVector4(); } return this.isAssociated == other.isAssociated @@ -483,7 +517,7 @@ public readonly partial struct Color : IEquatable { if (this.boxedHighPrecisionPixel is null) { - return HashCode.Combine(this.data, this.isAssociated); + return HashCode.Combine(this.ToScaledVector4(), this.isAssociated); } return HashCode.Combine(this.boxedHighPrecisionPixel.ToScaledVector4(), this.isAssociated); diff --git a/src/ImageSharp/Common/Helpers/SimdUtils.Convert.cs b/src/ImageSharp/Common/Helpers/SimdUtils.Convert.cs index 2efa3f5450..3a66be520b 100644 --- a/src/ImageSharp/Common/Helpers/SimdUtils.Convert.cs +++ b/src/ImageSharp/Common/Helpers/SimdUtils.Convert.cs @@ -70,14 +70,13 @@ internal static partial class SimdUtils [MethodImpl(MethodImplOptions.NoInlining)] private static void ConvertByteToNormalizedFloatRemainder(ReadOnlySpan source, Span destination) { - const float scale = 1F / byte.MaxValue; ref byte sBase = ref MemoryMarshal.GetReference(source); ref float dBase = ref MemoryMarshal.GetReference(destination); for (int i = 0; i < source.Length; i++) { // Match the SIMD conversion so one span cannot contain different float representations of the same byte value. - Unsafe.Add(ref dBase, (uint)i) = Unsafe.Add(ref sBase, (uint)i) * scale; + Unsafe.Add(ref dBase, (uint)i) = Unsafe.Add(ref sBase, (uint)i) / (float)byte.MaxValue; } } diff --git a/src/ImageSharp/Common/Helpers/SimdUtils.HwIntrinsics.cs b/src/ImageSharp/Common/Helpers/SimdUtils.HwIntrinsics.cs index cb172d145e..503a64b90d 100644 --- a/src/ImageSharp/Common/Helpers/SimdUtils.HwIntrinsics.cs +++ b/src/ImageSharp/Common/Helpers/SimdUtils.HwIntrinsics.cs @@ -711,6 +711,10 @@ internal static partial class SimdUtils ReadOnlySpan source, Span destination) { + const double reciprocal = 1D / byte.MaxValue; + const float reciprocalHigh = (float)reciprocal; + const float reciprocalLow = (float)(reciprocal - reciprocalHigh); + if (Vector512.IsHardwareAccelerated && Avx512F.IsSupported) { DebugVerifySpanInput(source, destination, Vector512.Count); @@ -719,6 +723,8 @@ internal static partial class SimdUtils ref byte sourceBase = ref MemoryMarshal.GetReference(source); ref Vector512 destinationBase = ref Unsafe.As>(ref MemoryMarshal.GetReference(destination)); + Vector512 high = Vector512.Create(reciprocalHigh); + Vector512 low = Vector512.Create(reciprocalLow); for (nuint i = 0; i < n; i++) { @@ -728,11 +734,16 @@ internal static partial class SimdUtils Vector512 i2 = Avx512F.ConvertToVector512Int32(Vector128.LoadUnsafe(ref sourceBase, si + (nuint)(Vector512.Count * 2))); Vector512 i3 = Avx512F.ConvertToVector512Int32(Vector128.LoadUnsafe(ref sourceBase, si + (nuint)(Vector512.Count * 3))); - // Declare multiplier on each line. Codegen is better. - Vector512 f0 = Vector512.Create(1 / (float)byte.MaxValue) * Avx512F.ConvertToVector512Single(i0); - Vector512 f1 = Vector512.Create(1 / (float)byte.MaxValue) * Avx512F.ConvertToVector512Single(i1); - Vector512 f2 = Vector512.Create(1 / (float)byte.MaxValue) * Avx512F.ConvertToVector512Single(i2); - Vector512 f3 = Vector512.Create(1 / (float)byte.MaxValue) * Avx512F.ConvertToVector512Single(i3); + Vector512 f0 = Avx512F.ConvertToVector512Single(i0); + Vector512 f1 = Avx512F.ConvertToVector512Single(i1); + Vector512 f2 = Avx512F.ConvertToVector512Single(i2); + Vector512 f3 = Avx512F.ConvertToVector512Single(i3); + + // The residual term restores the correctly rounded byte / 255F result without paying for vector division. + f0 = Vector512_.FusedMultiplyAdd(f0, high, f0 * low); + f1 = Vector512_.FusedMultiplyAdd(f1, high, f1 * low); + f2 = Vector512_.FusedMultiplyAdd(f2, high, f2 * low); + f3 = Vector512_.FusedMultiplyAdd(f3, high, f3 * low); ref Vector512 d = ref Unsafe.Add(ref destinationBase, i * 4); @@ -750,6 +761,8 @@ internal static partial class SimdUtils ref byte sourceBase = ref MemoryMarshal.GetReference(source); ref Vector256 destinationBase = ref Unsafe.As>(ref MemoryMarshal.GetReference(destination)); + Vector256 high = Vector256.Create(reciprocalHigh); + Vector256 low = Vector256.Create(reciprocalLow); for (nuint i = 0; i < n; i++) { @@ -762,11 +775,15 @@ internal static partial class SimdUtils ref ulong refULong = ref Unsafe.As(ref Unsafe.Add(ref sourceBase, si)); Vector256 i3 = Avx2.ConvertToVector256Int32(Vector128.CreateScalarUnsafe(Unsafe.Add(ref refULong, 3)).AsByte()); - // Declare multiplier on each line. Codegen is better. - Vector256 f0 = Vector256.Create(1 / (float)byte.MaxValue) * Avx.ConvertToVector256Single(i0); - Vector256 f1 = Vector256.Create(1 / (float)byte.MaxValue) * Avx.ConvertToVector256Single(i1); - Vector256 f2 = Vector256.Create(1 / (float)byte.MaxValue) * Avx.ConvertToVector256Single(i2); - Vector256 f3 = Vector256.Create(1 / (float)byte.MaxValue) * Avx.ConvertToVector256Single(i3); + Vector256 f0 = Avx.ConvertToVector256Single(i0); + Vector256 f1 = Avx.ConvertToVector256Single(i1); + Vector256 f2 = Avx.ConvertToVector256Single(i2); + Vector256 f3 = Avx.ConvertToVector256Single(i3); + + f0 = Vector256_.FusedMultiplyAdd(f0, high, f0 * low); + f1 = Vector256_.FusedMultiplyAdd(f1, high, f1 * low); + f2 = Vector256_.FusedMultiplyAdd(f2, high, f2 * low); + f3 = Vector256_.FusedMultiplyAdd(f3, high, f3 * low); ref Vector256 d = ref Unsafe.Add(ref destinationBase, i * 4); @@ -785,7 +802,8 @@ internal static partial class SimdUtils ref byte sourceBase = ref MemoryMarshal.GetReference(source); ref Vector128 destinationBase = ref Unsafe.As>(ref MemoryMarshal.GetReference(destination)); - Vector128 scale = Vector128.Create(1 / (float)byte.MaxValue); + Vector128 high = Vector128.Create(reciprocalHigh); + Vector128 low = Vector128.Create(reciprocalLow); for (nuint i = 0; i < n; i++) { @@ -810,10 +828,15 @@ internal static partial class SimdUtils (i2, i3) = Vector128.Widen(s1.AsInt16()); } - Vector128 f0 = scale * Vector128.ConvertToSingle(i0); - Vector128 f1 = scale * Vector128.ConvertToSingle(i1); - Vector128 f2 = scale * Vector128.ConvertToSingle(i2); - Vector128 f3 = scale * Vector128.ConvertToSingle(i3); + Vector128 f0 = Vector128.ConvertToSingle(i0); + Vector128 f1 = Vector128.ConvertToSingle(i1); + Vector128 f2 = Vector128.ConvertToSingle(i2); + Vector128 f3 = Vector128.ConvertToSingle(i3); + + f0 = Vector128_.FusedMultiplyAdd(f0, high, f0 * low); + f1 = Vector128_.FusedMultiplyAdd(f1, high, f1 * low); + f2 = Vector128_.FusedMultiplyAdd(f2, high, f2 * low); + f3 = Vector128_.FusedMultiplyAdd(f3, high, f3 * low); ref Vector128 d = ref Unsafe.Add(ref destinationBase, i * 4); diff --git a/src/ImageSharp/Common/Helpers/Vector128Utilities.cs b/src/ImageSharp/Common/Helpers/Vector128Utilities.cs index 7ea77fadfb..cba34e1165 100644 --- a/src/ImageSharp/Common/Helpers/Vector128Utilities.cs +++ b/src/ImageSharp/Common/Helpers/Vector128Utilities.cs @@ -363,30 +363,58 @@ internal static class Vector128_ } /// - /// Performs a multiplication and an addition of the . + /// Computes an estimate of ( * ) + . /// - /// ret = (vm0 * vm1) + va - /// The vector to add to the intermediate result. - /// The first vector to multiply. - /// The second vector to multiply. - /// The . + /// The first vector to multiply. + /// The second vector to multiply. + /// The vector to add to the product. + /// An estimate of the multiplication and addition result. [MethodImpl(MethodImplOptions.AggressiveInlining)] - public static Vector128 MultiplyAdd( - Vector128 va, - Vector128 vm0, - Vector128 vm1) + public static Vector128 MultiplyAddEstimate(Vector128 left, Vector128 right, Vector128 addend) { if (Fma.IsSupported) { - return Fma.MultiplyAdd(vm1, vm0, va); + return Fma.MultiplyAdd(left, right, addend); } if (AdvSimd.IsSupported) { - return AdvSimd.FusedMultiplyAdd(va, vm0, vm1); + return AdvSimd.FusedMultiplyAdd(addend, left, right); } - return va + (vm0 * vm1); + return (left * right) + addend; + } + + /// + /// Computes ( * ) + , rounded as one ternary operation. + /// + /// The first vector to multiply. + /// The second vector to multiply. + /// The vector to add to the product. + /// The fused multiplication and addition result. + [MethodImpl(MethodImplOptions.AggressiveInlining)] + public static Vector128 FusedMultiplyAdd(Vector128 left, Vector128 right, Vector128 addend) + { + if (Fma.IsSupported) + { + return Fma.MultiplyAdd(left, right, addend); + } + + if (AdvSimd.IsSupported) + { + return AdvSimd.FusedMultiplyAdd(addend, left, right); + } + + // WebAssembly SIMD has no exact fused multiply-add, so match the runtime fallback by preserving fused rounding per element. + Vector64 lower = Vector64.Create( + MathF.FusedMultiplyAdd(left.GetElement(0), right.GetElement(0), addend.GetElement(0)), + MathF.FusedMultiplyAdd(left.GetElement(1), right.GetElement(1), addend.GetElement(1))); + + Vector64 upper = Vector64.Create( + MathF.FusedMultiplyAdd(left.GetElement(2), right.GetElement(2), addend.GetElement(2)), + MathF.FusedMultiplyAdd(left.GetElement(3), right.GetElement(3), addend.GetElement(3))); + + return Vector128.Create(lower, upper); } /// diff --git a/src/ImageSharp/Common/Helpers/Vector256Utilities.cs b/src/ImageSharp/Common/Helpers/Vector256Utilities.cs index caef699dd1..e900465b5e 100644 --- a/src/ImageSharp/Common/Helpers/Vector256Utilities.cs +++ b/src/ImageSharp/Common/Helpers/Vector256Utilities.cs @@ -114,25 +114,46 @@ internal static class Vector256_ } /// - /// Performs a multiplication and an addition of the . + /// Computes an estimate of ( * ) + . /// - /// ret = (vm0 * vm1) + va - /// The vector to add to the intermediate result. - /// The first vector to multiply. - /// The second vector to multiply. - /// The . + /// The first vector to multiply. + /// The second vector to multiply. + /// The vector to add to the product. + /// An estimate of the multiplication and addition result. [MethodImpl(MethodImplOptions.AggressiveInlining)] - public static Vector256 MultiplyAdd( - Vector256 va, - Vector256 vm0, - Vector256 vm1) + public static Vector256 MultiplyAddEstimate(Vector256 left, Vector256 right, Vector256 addend) + { + if (Fma.IsSupported) + { + return Fma.MultiplyAdd(left, right, addend); + } + + Vector128 lower = Vector128_.MultiplyAddEstimate(left.GetLower(), right.GetLower(), addend.GetLower()); + Vector128 upper = Vector128_.MultiplyAddEstimate(left.GetUpper(), right.GetUpper(), addend.GetUpper()); + + return Vector256.Create(lower, upper); + } + + /// + /// Computes ( * ) + , rounded as one ternary operation. + /// + /// The first vector to multiply. + /// The second vector to multiply. + /// The vector to add to the product. + /// The fused multiplication and addition result. + [MethodImpl(MethodImplOptions.AggressiveInlining)] + public static Vector256 FusedMultiplyAdd(Vector256 left, Vector256 right, Vector256 addend) { if (Fma.IsSupported) { - return Fma.MultiplyAdd(vm0, vm1, va); + return Fma.MultiplyAdd(left, right, addend); } - return va + (vm0 * vm1); + // Match the runtime fallback by recursively applying the same fused contract to both halves. + Vector128 lower = Vector128_.FusedMultiplyAdd(left.GetLower(), right.GetLower(), addend.GetLower()); + Vector128 upper = Vector128_.FusedMultiplyAdd(left.GetUpper(), right.GetUpper(), addend.GetUpper()); + + return Vector256.Create(lower, upper); } /// diff --git a/src/ImageSharp/Common/Helpers/Vector512Utilities.cs b/src/ImageSharp/Common/Helpers/Vector512Utilities.cs index 17f9e9c51e..b1af0b2af8 100644 --- a/src/ImageSharp/Common/Helpers/Vector512Utilities.cs +++ b/src/ImageSharp/Common/Helpers/Vector512Utilities.cs @@ -86,19 +86,47 @@ internal static class Vector512_ => Avx512F.RoundScale(vector, 0b0000_1000); /// - /// Performs a multiplication and an addition of the . + /// Computes an estimate of ( * ) + . /// - /// ret = (vm0 * vm1) + va - /// The vector to add to the intermediate result. - /// The first vector to multiply. - /// The second vector to multiply. - /// The . + /// The first vector to multiply. + /// The second vector to multiply. + /// The vector to add to the product. + /// An estimate of the multiplication and addition result. [MethodImpl(MethodImplOptions.AggressiveInlining)] - public static Vector512 MultiplyAdd( - Vector512 va, - Vector512 vm0, - Vector512 vm1) - => Avx512F.FusedMultiplyAdd(vm0, vm1, va); + public static Vector512 MultiplyAddEstimate(Vector512 left, Vector512 right, Vector512 addend) + { + if (Avx512F.IsSupported) + { + return Avx512F.FusedMultiplyAdd(left, right, addend); + } + + Vector256 lower = Vector256_.MultiplyAddEstimate(left.GetLower(), right.GetLower(), addend.GetLower()); + Vector256 upper = Vector256_.MultiplyAddEstimate(left.GetUpper(), right.GetUpper(), addend.GetUpper()); + + return Vector512.Create(lower, upper); + } + + /// + /// Computes ( * ) + , rounded as one ternary operation. + /// + /// The first vector to multiply. + /// The second vector to multiply. + /// The vector to add to the product. + /// The fused multiplication and addition result. + [MethodImpl(MethodImplOptions.AggressiveInlining)] + public static Vector512 FusedMultiplyAdd(Vector512 left, Vector512 right, Vector512 addend) + { + if (Avx512F.IsSupported) + { + return Avx512F.FusedMultiplyAdd(left, right, addend); + } + + // Match the runtime fallback by recursively applying the same fused contract to both halves. + Vector256 lower = Vector256_.FusedMultiplyAdd(left.GetLower(), right.GetLower(), addend.GetLower()); + Vector256 upper = Vector256_.FusedMultiplyAdd(left.GetUpper(), right.GetUpper(), addend.GetUpper()); + + return Vector512.Create(lower, upper); + } /// /// Performs a multiplication and a negated addition of the . diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector128.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector128.cs index 633080706b..3c4a64f806 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector128.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector128.cs @@ -74,7 +74,7 @@ internal abstract partial class JpegColorConverterBase ref Vector128 b = ref Unsafe.Add(ref srcBlue, i); // luminosity = (0.299 * r) + (0.587 * g) + (0.114 * b) - Unsafe.Add(ref destLuminance, i) = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); + Unsafe.Add(ref destLuminance, i) = Vector128_.MultiplyAddEstimate(f0299, r, Vector128_.MultiplyAddEstimate(f0587, g, f0114 * b)); } } } diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector256.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector256.cs index 0b2b8a119f..3d98d1effc 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector256.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector256.cs @@ -74,7 +74,7 @@ internal abstract partial class JpegColorConverterBase ref Vector256 b = ref Unsafe.Add(ref srcBlue, i); // luminosity = (0.299 * r) + (0.587 * g) + (0.114 * b) - Unsafe.Add(ref destLuminance, i) = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); + Unsafe.Add(ref destLuminance, i) = Vector256_.MultiplyAddEstimate(f0299, r, Vector256_.MultiplyAddEstimate(f0587, g, f0114 * b)); } } } diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector512.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector512.cs index 0a58b0196b..96126ac9da 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector512.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.GrayScaleVector512.cs @@ -74,7 +74,7 @@ internal abstract partial class JpegColorConverterBase ref Vector512 b = ref Unsafe.Add(ref srcBlue, i); // luminosity = (0.299 * r) + (0.587 * g) + (0.114 * b) - Unsafe.Add(ref destLuminance, i) = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); + Unsafe.Add(ref destLuminance, i) = Vector512_.MultiplyAddEstimate(f0299, r, Vector512_.MultiplyAddEstimate(f0587, g, f0114 * b)); } } diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector128.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector128.cs index b360c373ad..ae3391d42a 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector128.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector128.cs @@ -53,9 +53,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector128 r = Vector128_.MultiplyAdd(y, cr, rCrMult) * scaledK; - Vector128 g = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult) * scaledK; - Vector128 b = Vector128_.MultiplyAdd(y, cb, bCbMult) * scaledK; + Vector128 r = Vector128_.MultiplyAddEstimate(cr, rCrMult, y) * scaledK; + Vector128 g = Vector128_.MultiplyAddEstimate(cr, gCrMult, Vector128_.MultiplyAddEstimate(cb, gCbMult, y)) * scaledK; + Vector128 b = Vector128_.MultiplyAddEstimate(cb, bCbMult, y) * scaledK; c0 = r; c1 = g; @@ -117,9 +117,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector128 y = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector128 cb = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector128 cr = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector128 y = Vector128_.MultiplyAddEstimate(f0299, r, Vector128_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector128 cb = chromaOffset + Vector128_.MultiplyAddEstimate(fn0168736, r, Vector128_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector128 cr = chromaOffset + Vector128_.MultiplyAddEstimate(f05, r, Vector128_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y * maxSampleValue; Unsafe.Add(ref destCb, i) = chromaOffset + (cb * maxSampleValue); diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector256.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector256.cs index f996522d36..ec6b228ca4 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector256.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector256.cs @@ -53,9 +53,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector256 r = Vector256_.MultiplyAdd(y, cr, rCrMult) * scaledK; - Vector256 g = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult) * scaledK; - Vector256 b = Vector256_.MultiplyAdd(y, cb, bCbMult) * scaledK; + Vector256 r = Vector256_.MultiplyAddEstimate(cr, rCrMult, y) * scaledK; + Vector256 g = Vector256_.MultiplyAddEstimate(cr, gCrMult, Vector256_.MultiplyAddEstimate(cb, gCbMult, y)) * scaledK; + Vector256 b = Vector256_.MultiplyAddEstimate(cb, bCbMult, y) * scaledK; c0 = r; c1 = g; @@ -117,9 +117,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector256 y = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector256 cb = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector256 cr = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector256 y = Vector256_.MultiplyAddEstimate(f0299, r, Vector256_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector256 cb = chromaOffset + Vector256_.MultiplyAddEstimate(fn0168736, r, Vector256_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector256 cr = chromaOffset + Vector256_.MultiplyAddEstimate(f05, r, Vector256_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y * maxSampleValue; Unsafe.Add(ref destCb, i) = chromaOffset + (cb * maxSampleValue); diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector512.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector512.cs index 47168a739d..6a0fb93e27 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector512.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.TiffYccKVector512.cs @@ -57,9 +57,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector512 r = Vector512_.MultiplyAdd(y, cr, rCrMult) * scaledK; - Vector512 g = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult) * scaledK; - Vector512 b = Vector512_.MultiplyAdd(y, cb, bCbMult) * scaledK; + Vector512 r = Vector512_.MultiplyAddEstimate(cr, rCrMult, y) * scaledK; + Vector512 g = Vector512_.MultiplyAddEstimate(cr, gCrMult, Vector512_.MultiplyAddEstimate(cb, gCbMult, y)) * scaledK; + Vector512 b = Vector512_.MultiplyAddEstimate(cb, bCbMult, y) * scaledK; c0 = r; c1 = g; @@ -128,9 +128,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector512 y = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector512 cb = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector512 cr = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector512 y = Vector512_.MultiplyAddEstimate(f0299, r, Vector512_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector512 cb = chromaOffset + Vector512_.MultiplyAddEstimate(fn0168736, r, Vector512_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector512 cr = chromaOffset + Vector512_.MultiplyAddEstimate(f05, r, Vector512_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y * maxSampleValue; Unsafe.Add(ref destCb, i) = chromaOffset + (cb * maxSampleValue); diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector128.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector128.cs index 01c8508edc..37847a6e67 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector128.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector128.cs @@ -53,9 +53,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector128 r = Vector128_.MultiplyAdd(y, cr, rCrMult); - Vector128 g = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult); - Vector128 b = Vector128_.MultiplyAdd(y, cb, bCbMult); + Vector128 r = Vector128_.MultiplyAddEstimate(cr, rCrMult, y); + Vector128 g = Vector128_.MultiplyAddEstimate(cr, gCrMult, Vector128_.MultiplyAddEstimate(cb, gCbMult, y)); + Vector128 b = Vector128_.MultiplyAddEstimate(cb, bCbMult, y); r = Vector128_.RoundToNearestInteger(r) * scale; g = Vector128_.RoundToNearestInteger(g) * scale; @@ -108,9 +108,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector128 y = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector128 cb = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector128 cr = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector128 y = Vector128_.MultiplyAddEstimate(f0299, r, Vector128_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector128 cb = chromaOffset + Vector128_.MultiplyAddEstimate(fn0168736, r, Vector128_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector128 cr = chromaOffset + Vector128_.MultiplyAddEstimate(f05, r, Vector128_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y; Unsafe.Add(ref destCb, i) = cb; diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector256.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector256.cs index 8fdf1004d8..fbccf88e20 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector256.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector256.cs @@ -53,9 +53,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector256 r = Vector256_.MultiplyAdd(y, cr, rCrMult); - Vector256 g = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult); - Vector256 b = Vector256_.MultiplyAdd(y, cb, bCbMult); + Vector256 r = Vector256_.MultiplyAddEstimate(cr, rCrMult, y); + Vector256 g = Vector256_.MultiplyAddEstimate(cr, gCrMult, Vector256_.MultiplyAddEstimate(cb, gCbMult, y)); + Vector256 b = Vector256_.MultiplyAddEstimate(cb, bCbMult, y); r = Vector256_.RoundToNearestInteger(r) * scale; g = Vector256_.RoundToNearestInteger(g) * scale; @@ -108,9 +108,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector256 y = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector256 cb = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector256 cr = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector256 y = Vector256_.MultiplyAddEstimate(f0299, r, Vector256_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector256 cb = chromaOffset + Vector256_.MultiplyAddEstimate(fn0168736, r, Vector256_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector256 cr = chromaOffset + Vector256_.MultiplyAddEstimate(f05, r, Vector256_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y; Unsafe.Add(ref destCb, i) = cb; diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector512.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector512.cs index 33cb2496c6..3879171752 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector512.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YCbCrVector512.cs @@ -56,9 +56,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector512 r = Vector512_.MultiplyAdd(y, cr, rCrMult); - Vector512 g = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult); - Vector512 b = Vector512_.MultiplyAdd(y, cb, bCbMult); + Vector512 r = Vector512_.MultiplyAddEstimate(cr, rCrMult, y); + Vector512 g = Vector512_.MultiplyAddEstimate(cr, gCrMult, Vector512_.MultiplyAddEstimate(cb, gCbMult, y)); + Vector512 b = Vector512_.MultiplyAddEstimate(cb, bCbMult, y); r = Vector512_.RoundToNearestInteger(r) * scale; g = Vector512_.RoundToNearestInteger(g) * scale; @@ -107,9 +107,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector512 y = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector512 cb = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector512 cr = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector512 y = Vector512_.MultiplyAddEstimate(f0299, r, Vector512_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector512 cb = chromaOffset + Vector512_.MultiplyAddEstimate(fn0168736, r, Vector512_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector512 cr = chromaOffset + Vector512_.MultiplyAddEstimate(f05, r, Vector512_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y; Unsafe.Add(ref destCb, i) = cb; diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector128.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector128.cs index 67b7ee0dc6..279c100b2c 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector128.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector128.cs @@ -58,9 +58,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector128 r = Vector128_.MultiplyAdd(y, cr, rCrMult); - Vector128 g = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult); - Vector128 b = Vector128_.MultiplyAdd(y, cb, bCbMult); + Vector128 r = Vector128_.MultiplyAddEstimate(cr, rCrMult, y); + Vector128 g = Vector128_.MultiplyAddEstimate(cr, gCrMult, Vector128_.MultiplyAddEstimate(cb, gCbMult, y)); + Vector128 b = Vector128_.MultiplyAddEstimate(cb, bCbMult, y); r = max - Vector128_.RoundToNearestInteger(r); g = max - Vector128_.RoundToNearestInteger(g); @@ -122,9 +122,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector128 y = Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector128 cb = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector128 cr = chromaOffset + Vector128_.MultiplyAdd(Vector128_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector128 y = Vector128_.MultiplyAddEstimate(f0299, r, Vector128_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector128 cb = chromaOffset + Vector128_.MultiplyAddEstimate(fn0168736, r, Vector128_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector128 cr = chromaOffset + Vector128_.MultiplyAddEstimate(f05, r, Vector128_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y; Unsafe.Add(ref destCb, i) = cb; diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector256.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector256.cs index 3a586d25fa..1bebdfcf44 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector256.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector256.cs @@ -58,9 +58,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector256 r = Vector256_.MultiplyAdd(y, cr, rCrMult); - Vector256 g = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult); - Vector256 b = Vector256_.MultiplyAdd(y, cb, bCbMult); + Vector256 r = Vector256_.MultiplyAddEstimate(cr, rCrMult, y); + Vector256 g = Vector256_.MultiplyAddEstimate(cr, gCrMult, Vector256_.MultiplyAddEstimate(cb, gCbMult, y)); + Vector256 b = Vector256_.MultiplyAddEstimate(cb, bCbMult, y); r = max - Vector256_.RoundToNearestInteger(r); g = max - Vector256_.RoundToNearestInteger(g); @@ -122,9 +122,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector256 y = Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector256 cb = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector256 cr = chromaOffset + Vector256_.MultiplyAdd(Vector256_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector256 y = Vector256_.MultiplyAddEstimate(f0299, r, Vector256_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector256 cb = chromaOffset + Vector256_.MultiplyAddEstimate(fn0168736, r, Vector256_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector256 cr = chromaOffset + Vector256_.MultiplyAddEstimate(f05, r, Vector256_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y; Unsafe.Add(ref destCb, i) = cb; diff --git a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector512.cs b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector512.cs index b81a833cde..9c0e1ab747 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector512.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/ColorConverters/JpegColorConverter.YccKVector512.cs @@ -58,9 +58,9 @@ internal abstract partial class JpegColorConverterBase // r = y + (1.402F * cr); // g = y - (0.344136F * cb) - (0.714136F * cr); // b = y + (1.772F * cb); - Vector512 r = Vector512_.MultiplyAdd(y, cr, rCrMult); - Vector512 g = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(y, cb, gCbMult), cr, gCrMult); - Vector512 b = Vector512_.MultiplyAdd(y, cb, bCbMult); + Vector512 r = Vector512_.MultiplyAddEstimate(cr, rCrMult, y); + Vector512 g = Vector512_.MultiplyAddEstimate(cr, gCrMult, Vector512_.MultiplyAddEstimate(cb, gCbMult, y)); + Vector512 b = Vector512_.MultiplyAddEstimate(cb, bCbMult, y); r = max - Vector512_.RoundToNearestInteger(r); g = max - Vector512_.RoundToNearestInteger(g); @@ -126,9 +126,9 @@ internal abstract partial class JpegColorConverterBase // y = 0 + (0.299 * r) + (0.587 * g) + (0.114 * b) // cb = 128 - (0.168736 * r) - (0.331264 * g) + (0.5 * b) // cr = 128 + (0.5 * r) - (0.418688 * g) - (0.081312 * b) - Vector512 y = Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f0114 * b, f0587, g), f0299, r); - Vector512 cb = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(f05 * b, fn0331264, g), fn0168736, r); - Vector512 cr = chromaOffset + Vector512_.MultiplyAdd(Vector512_.MultiplyAdd(fn0081312F * b, fn0418688, g), f05, r); + Vector512 y = Vector512_.MultiplyAddEstimate(f0299, r, Vector512_.MultiplyAddEstimate(f0587, g, f0114 * b)); + Vector512 cb = chromaOffset + Vector512_.MultiplyAddEstimate(fn0168736, r, Vector512_.MultiplyAddEstimate(fn0331264, g, f05 * b)); + Vector512 cr = chromaOffset + Vector512_.MultiplyAddEstimate(f05, r, Vector512_.MultiplyAddEstimate(fn0418688, g, fn0081312F * b)); Unsafe.Add(ref destY, i) = y; Unsafe.Add(ref destCb, i) = cb; diff --git a/src/ImageSharp/Formats/Jpeg/Components/FloatingPointDCT.Vector256.cs b/src/ImageSharp/Formats/Jpeg/Components/FloatingPointDCT.Vector256.cs index bcd8c70431..7c342d7a0b 100644 --- a/src/ImageSharp/Formats/Jpeg/Components/FloatingPointDCT.Vector256.cs +++ b/src/ImageSharp/Formats/Jpeg/Components/FloatingPointDCT.Vector256.cs @@ -55,8 +55,8 @@ internal static partial class FloatingPointDCT tmp12 = tmp6 + tmp7; Vector256 z5 = (tmp10 - tmp12) * Vector256.Create(0.382683433f); // mm256_F_0_3826 - Vector256 z2 = Vector256_.MultiplyAdd(z5, Vector256.Create(0.541196100f), tmp10); // mm256_F_0_5411 - Vector256 z4 = Vector256_.MultiplyAdd(z5, Vector256.Create(1.306562965f), tmp12); // mm256_F_1_3065 + Vector256 z2 = Vector256_.MultiplyAddEstimate(Vector256.Create(0.541196100f), tmp10, z5); // mm256_F_0_5411 + Vector256 z4 = Vector256_.MultiplyAddEstimate(Vector256.Create(1.306562965f), tmp12, z5); // mm256_F_1_3065 Vector256 z3 = tmp11 * mm256_F_0_7071; Vector256 z11 = tmp7 + z3; @@ -122,8 +122,8 @@ internal static partial class FloatingPointDCT z5 = (z10 + z12) * Vector256.Create(1.847759065f); // mm256_F_1_8477 - tmp10 = Vector256_.MultiplyAdd(z5, z12, Vector256.Create(-1.082392200f)); // mm256_F_n1_0823 - tmp12 = Vector256_.MultiplyAdd(z5, z10, Vector256.Create(-2.613125930f)); // mm256_F_n2_6131 + tmp10 = Vector256_.MultiplyAddEstimate(z12, Vector256.Create(-1.082392200f), z5); // mm256_F_n1_0823 + tmp12 = Vector256_.MultiplyAddEstimate(z10, Vector256.Create(-2.613125930f), z5); // mm256_F_n2_6131 tmp6 = tmp12 - tmp7; tmp5 = tmp11 - tmp6; diff --git a/src/ImageSharp/PixelFormats/PixelBlenders/AssociatedAlphaPorterDuffFunctions.cs b/src/ImageSharp/PixelFormats/PixelBlenders/AssociatedAlphaPorterDuffFunctions.cs index 4f6af32c8a..d7b5260d36 100644 --- a/src/ImageSharp/PixelFormats/PixelBlenders/AssociatedAlphaPorterDuffFunctions.cs +++ b/src/ImageSharp/PixelFormats/PixelBlenders/AssociatedAlphaPorterDuffFunctions.cs @@ -398,7 +398,9 @@ internal static partial class AssociatedAlphaPorterDuffFunctions // Atop discards source-only color and retains the destination alpha unchanged. Vector4 sourceAlpha = Numerics.PermuteW(source); Vector4 destinationAlpha = Numerics.PermuteW(destination); - Vector4 result = (destination * (Vector4.One - sourceAlpha)) + overlap; + Vector4 coefficient = Vector4.One - sourceAlpha; + Vector4 result = Vector128_.FusedMultiplyAdd(destination.AsVector128(), coefficient.AsVector128(), overlap.AsVector128()).AsVector4(); + return Numerics.WithW(result, destinationAlpha); } @@ -408,7 +410,8 @@ internal static partial class AssociatedAlphaPorterDuffFunctions { Vector256 sourceAlpha = Avx.Permute(source, ShuffleAlphaControl); Vector256 destinationAlpha = Avx.Permute(destination, ShuffleAlphaControl); - Vector256 result = (destination * (Vector256.Create(1F) - sourceAlpha)) + overlap; + Vector256 result = Vector256_.FusedMultiplyAdd(destination, Vector256.Create(1F) - sourceAlpha, overlap); + return Avx.Blend(result, destinationAlpha, BlendAlphaControl); } @@ -418,7 +421,8 @@ internal static partial class AssociatedAlphaPorterDuffFunctions { Vector512 sourceAlpha = Vector512_.ShuffleNative(source, ShuffleAlphaControl); Vector512 destinationAlpha = Vector512_.ShuffleNative(destination, ShuffleAlphaControl); - Vector512 result = (destination * (Vector512.Create(1F) - sourceAlpha)) + overlap; + Vector512 result = Vector512_.FusedMultiplyAdd(destination, Vector512.Create(1F) - sourceAlpha, overlap); + return Vector512.ConditionalSelect(AlphaMask512(), destinationAlpha, result); } @@ -506,18 +510,18 @@ internal static partial class AssociatedAlphaPorterDuffFunctions public static Vector4 BlendWithCoverage(Vector4 backdrop, Vector4 source, float coverage) { // Use the same fused operation as the wider paths so exact midpoints cannot change across vector widths. - return Vector128_.MultiplyAdd(backdrop.AsVector128(), (source - backdrop).AsVector128(), Vector128.Create(coverage)).AsVector4(); + return Vector128_.FusedMultiplyAdd((source - backdrop).AsVector128(), Vector128.Create(coverage), backdrop.AsVector128()).AsVector4(); } /// [MethodImpl(MethodImplOptions.AggressiveInlining)] public static Vector256 BlendWithCoverage(Vector256 backdrop, Vector256 source, Vector256 coverage) - => Vector256_.MultiplyAdd(backdrop, source - backdrop, coverage); + => Vector256_.FusedMultiplyAdd(source - backdrop, coverage, backdrop); /// [MethodImpl(MethodImplOptions.AggressiveInlining)] public static Vector512 BlendWithCoverage(Vector512 backdrop, Vector512 source, Vector512 coverage) - => Vector512_.MultiplyAdd(backdrop, source - backdrop, coverage); + => Vector512_.FusedMultiplyAdd(source - backdrop, coverage, backdrop); /// /// Calculates one associated Overlay overlap component without recovering either straight component. diff --git a/src/ImageSharp/PixelFormats/PixelBlenders/PorterDuffFunctions.cs b/src/ImageSharp/PixelFormats/PixelBlenders/PorterDuffFunctions.cs index 1e5fc8df91..46d97ad3dc 100644 --- a/src/ImageSharp/PixelFormats/PixelBlenders/PorterDuffFunctions.cs +++ b/src/ImageSharp/PixelFormats/PixelBlenders/PorterDuffFunctions.cs @@ -340,7 +340,7 @@ internal static partial class PorterDuffFunctions Vector4 sourcePremultiplied = Numerics.WithW(source * sourceAlpha, sourceAlpha); // Use the same fused operation as the wider paths so exact midpoints cannot change across vector widths. - Vector4 result = Vector128_.MultiplyAdd(backdropPremultiplied.AsVector128(), (sourcePremultiplied - backdropPremultiplied).AsVector128(), Vector128.Create(coverage)).AsVector4(); + Vector4 result = Vector128_.MultiplyAddEstimate((sourcePremultiplied - backdropPremultiplied).AsVector128(), Vector128.Create(coverage), backdropPremultiplied.AsVector128()).AsVector4(); Numerics.UnPremultiply(ref result); return result; @@ -360,7 +360,7 @@ internal static partial class PorterDuffFunctions Vector256 sourceAlpha = Avx.Permute(source, ShuffleAlphaControl); Vector256 backdropPremultiplied = Avx.Blend(backdrop * backdropAlpha, backdropAlpha, BlendAlphaControl); Vector256 sourcePremultiplied = Avx.Blend(source * sourceAlpha, sourceAlpha, BlendAlphaControl); - Vector256 result = Vector256_.MultiplyAdd(backdropPremultiplied, sourcePremultiplied - backdropPremultiplied, coverage); + Vector256 result = Vector256_.MultiplyAddEstimate(sourcePremultiplied - backdropPremultiplied, coverage, backdropPremultiplied); return Numerics.UnPremultiply(result, Avx.Permute(result, ShuffleAlphaControl)); } @@ -380,7 +380,7 @@ internal static partial class PorterDuffFunctions Vector512 alphaMask = AlphaMask512(); Vector512 backdropPremultiplied = Vector512.ConditionalSelect(alphaMask, backdropAlpha, backdrop * backdropAlpha); Vector512 sourcePremultiplied = Vector512.ConditionalSelect(alphaMask, sourceAlpha, source * sourceAlpha); - Vector512 result = Vector512_.MultiplyAdd(backdropPremultiplied, sourcePremultiplied - backdropPremultiplied, coverage); + Vector512 result = Vector512_.MultiplyAddEstimate(sourcePremultiplied - backdropPremultiplied, coverage, backdropPremultiplied); return Numerics.UnPremultiply(result, Vector512_.ShuffleNative(result, ShuffleAlphaControl)); } @@ -483,8 +483,8 @@ internal static partial class PorterDuffFunctions // calculate final color Vector256 color = destination * dstW; - color = Vector256_.MultiplyAdd(color, source, srcW); - color = Vector256_.MultiplyAdd(color, blend, blendW); + color = Vector256_.MultiplyAddEstimate(source, srcW, color); + color = Vector256_.MultiplyAddEstimate(blend, blendW, color); // unpremultiply return Numerics.UnPremultiply(color, alpha); @@ -513,8 +513,8 @@ internal static partial class PorterDuffFunctions // calculate final color Vector512 color = destination * dstW; - color = Vector512_.MultiplyAdd(color, source, srcW); - color = Vector512_.MultiplyAdd(color, blend, blendW); + color = Vector512_.MultiplyAddEstimate(source, srcW, color); + color = Vector512_.MultiplyAddEstimate(blend, blendW, color); // unpremultiply return Numerics.UnPremultiply(color, alpha); @@ -567,7 +567,7 @@ internal static partial class PorterDuffFunctions Vector256 dstW = alpha - blendW; // calculate final color - Vector256 color = Vector256_.MultiplyAdd(Avx.Multiply(blend, blendW), destination, dstW); + Vector256 color = Vector256_.MultiplyAddEstimate(destination, dstW, Avx.Multiply(blend, blendW)); // unpremultiply return Numerics.UnPremultiply(color, alpha); @@ -592,7 +592,7 @@ internal static partial class PorterDuffFunctions Vector512 dstW = alpha - blendW; // calculate final color - Vector512 color = Vector512_.MultiplyAdd(blend * blendW, destination, dstW); + Vector512 color = Vector512_.MultiplyAddEstimate(destination, dstW, blend * blendW); // unpremultiply return Numerics.UnPremultiply(color, alpha); @@ -751,8 +751,8 @@ internal static partial class PorterDuffFunctions Vector256 dstW = vOne - sW; // calculate alpha - Vector256 alpha = Vector256_.MultiplyAdd(Avx.Multiply(dW, dstW), sW, srcW); - Vector256 color = Vector256_.MultiplyAdd(Avx.Multiply(Avx.Multiply(dW, destination), dstW), Avx.Multiply(sW, source), srcW); + Vector256 alpha = Vector256_.MultiplyAddEstimate(sW, srcW, Avx.Multiply(dW, dstW)); + Vector256 color = Vector256_.MultiplyAddEstimate(Avx.Multiply(sW, source), srcW, Avx.Multiply(Avx.Multiply(dW, destination), dstW)); // unpremultiply return Numerics.UnPremultiply(color, alpha); @@ -776,8 +776,8 @@ internal static partial class PorterDuffFunctions Vector512 dstW = vOne - sW; // calculate alpha - Vector512 alpha = Vector512_.MultiplyAdd(dW * dstW, sW, srcW); - Vector512 color = Vector512_.MultiplyAdd((dW * destination) * dstW, sW * source, srcW); + Vector512 alpha = Vector512_.MultiplyAddEstimate(sW, srcW, dW * dstW); + Vector512 color = Vector512_.MultiplyAddEstimate(sW * source, srcW, (dW * destination) * dstW); // unpremultiply return Numerics.UnPremultiply(color, alpha); diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/Abgr32.cs b/src/ImageSharp/PixelFormats/PixelImplementations/Abgr32.cs index 1190d307a7..ff95e714ca 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/Abgr32.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/Abgr32.cs @@ -182,7 +182,7 @@ public partial struct Abgr32 : IPixel, IPackedVector /// [MethodImpl(MethodImplOptions.AggressiveInlining)] - public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / MaxBytes; + public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue; /// public static PixelTypeInfo GetPixelTypeInfo() diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/Abgr32P.cs b/src/ImageSharp/PixelFormats/PixelImplementations/Abgr32P.cs index 780dbd2f8f..0d1730c801 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/Abgr32P.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/Abgr32P.cs @@ -19,8 +19,6 @@ namespace SixLabors.ImageSharp.PixelFormats; [StructLayout(LayoutKind.Sequential)] public partial struct Abgr32P : IPixel, IPackedVector { - private const float ByteScale = 1F / byte.MaxValue; - private static readonly Vector4 Half = new(0.5F); private static readonly Vector4 MaxBytes = new(byte.MaxValue); @@ -147,7 +145,7 @@ public partial struct Abgr32P : IPixel, IPackedVector => Rgba32.FromScaledVector4(Vector4Converters.AssociatedRgbaCompatible.ToUnassociatedVector4(this.R, this.G, this.B, this.A)); /// - public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) * ByteScale; + public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue; /// public readonly Vector4 ToVector4() => this.ToScaledVector4(); diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/Argb32.cs b/src/ImageSharp/PixelFormats/PixelImplementations/Argb32.cs index b74f5db718..7e40c27539 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/Argb32.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/Argb32.cs @@ -175,7 +175,7 @@ public partial struct Argb32 : IPixel, IPackedVector /// [MethodImpl(MethodImplOptions.AggressiveInlining)] - public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / MaxBytes; + public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue; /// public static PixelTypeInfo GetPixelTypeInfo() diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/Argb32P.cs b/src/ImageSharp/PixelFormats/PixelImplementations/Argb32P.cs index fac4f77018..d5069638cf 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/Argb32P.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/Argb32P.cs @@ -19,8 +19,6 @@ namespace SixLabors.ImageSharp.PixelFormats; [StructLayout(LayoutKind.Sequential)] public partial struct Argb32P : IPixel, IPackedVector { - private const float ByteScale = 1F / byte.MaxValue; - private static readonly Vector4 Half = new(0.5F); private static readonly Vector4 MaxBytes = new(byte.MaxValue); @@ -147,7 +145,7 @@ public partial struct Argb32P : IPixel, IPackedVector => Rgba32.FromScaledVector4(Vector4Converters.AssociatedRgbaCompatible.ToUnassociatedVector4(this.R, this.G, this.B, this.A)); /// - public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) * ByteScale; + public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue; /// public readonly Vector4 ToVector4() => this.ToScaledVector4(); diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/Bgra32.cs b/src/ImageSharp/PixelFormats/PixelImplementations/Bgra32.cs index 903d9dc8cb..8c47032e7d 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/Bgra32.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/Bgra32.cs @@ -124,7 +124,7 @@ public partial struct Bgra32 : IPixel, IPackedVector /// [MethodImpl(MethodImplOptions.AggressiveInlining)] - public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / MaxBytes; + public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue; /// public static PixelTypeInfo GetPixelTypeInfo() diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/Bgra32P.cs b/src/ImageSharp/PixelFormats/PixelImplementations/Bgra32P.cs index a69dfeb46f..bc050d1484 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/Bgra32P.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/Bgra32P.cs @@ -19,8 +19,6 @@ namespace SixLabors.ImageSharp.PixelFormats; [StructLayout(LayoutKind.Sequential)] public partial struct Bgra32P : IPixel, IPackedVector { - private const float ByteScale = 1F / byte.MaxValue; - private static readonly Vector4 Half = new(0.5F); private static readonly Vector4 MaxBytes = new(byte.MaxValue); @@ -147,7 +145,7 @@ public partial struct Bgra32P : IPixel, IPackedVector => Rgba32.FromScaledVector4(Vector4Converters.AssociatedRgbaCompatible.ToUnassociatedVector4(this.R, this.G, this.B, this.A)); /// - public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) * ByteScale; + public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue; /// public readonly Vector4 ToVector4() => this.ToScaledVector4(); diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/NormalizedByte4P.cs b/src/ImageSharp/PixelFormats/PixelImplementations/NormalizedByte4P.cs index 6a31827778..b6549b7dd2 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/NormalizedByte4P.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/NormalizedByte4P.cs @@ -158,10 +158,7 @@ public partial struct NormalizedByte4P : IPixel, IPackedVector /// /// The unassociated scaled vector. /// The associated pixel. - private static NormalizedByte4P FromUnassociatedScaledVector4(Vector4 source) - { - return FromScaledVector4(Associate(source)); - } + private static NormalizedByte4P FromUnassociatedScaledVector4(Vector4 source) => FromScaledVector4(Associate(source)); /// /// Converts an unassociated scaled vector to the associated representation of a signed-normalized-byte destination. diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/Rgba32.cs b/src/ImageSharp/PixelFormats/PixelImplementations/Rgba32.cs index 1eafab854f..0cec0b30dc 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/Rgba32.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/Rgba32.cs @@ -220,7 +220,7 @@ public partial struct Rgba32 : IPixel, IPackedVector /// [MethodImpl(MethodImplOptions.AggressiveInlining)] - public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / MaxBytes; + public readonly Vector4 ToVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue; /// public static PixelTypeInfo GetPixelTypeInfo() diff --git a/src/ImageSharp/PixelFormats/PixelImplementations/Rgba32P.cs b/src/ImageSharp/PixelFormats/PixelImplementations/Rgba32P.cs index 46b6a4f65d..6bf2e1db04 100644 --- a/src/ImageSharp/PixelFormats/PixelImplementations/Rgba32P.cs +++ b/src/ImageSharp/PixelFormats/PixelImplementations/Rgba32P.cs @@ -19,8 +19,6 @@ namespace SixLabors.ImageSharp.PixelFormats; [StructLayout(LayoutKind.Sequential)] public partial struct Rgba32P : IPixel, IPackedVector { - private const float ByteScale = 1F / byte.MaxValue; - private static readonly Vector4 Half = new(0.5F); private static readonly Vector4 MaxBytes = new(byte.MaxValue); @@ -147,7 +145,7 @@ public partial struct Rgba32P : IPixel, IPackedVector => Rgba32.FromScaledVector4(Vector4Converters.AssociatedRgbaCompatible.ToUnassociatedVector4(this.R, this.G, this.B, this.A)); /// - public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) * ByteScale; + public readonly Vector4 ToScaledVector4() => new Vector4(this.R, this.G, this.B, this.A) / byte.MaxValue; /// public readonly Vector4 ToVector4() => this.ToScaledVector4(); diff --git a/src/ImageSharp/Processing/Processors/Transforms/Resize/ResizeKernel.cs b/src/ImageSharp/Processing/Processors/Transforms/Resize/ResizeKernel.cs index a85487d1c1..c0380224c4 100644 --- a/src/ImageSharp/Processing/Processors/Transforms/Resize/ResizeKernel.cs +++ b/src/ImageSharp/Processing/Processors/Transforms/Resize/ResizeKernel.cs @@ -104,8 +104,8 @@ internal readonly unsafe struct ResizeKernel Vector256 pixels256_0 = Unsafe.As>(ref rowStartRef); Vector256 pixels256_1 = Unsafe.As>(ref Unsafe.Add(ref rowStartRef, (nuint)2)); - result256_0 = Vector256_.MultiplyAdd(result256_0, Vector256.Load(bufferStart), pixels256_0); - result256_1 = Vector256_.MultiplyAdd(result256_1, Vector256.Load(bufferStart + 8), pixels256_1); + result256_0 = Vector256_.MultiplyAddEstimate(Vector256.Load(bufferStart), pixels256_0, result256_0); + result256_1 = Vector256_.MultiplyAddEstimate(Vector256.Load(bufferStart + 8), pixels256_1, result256_1); bufferStart += 16; rowStartRef = ref Unsafe.Add(ref rowStartRef, (nuint)4); @@ -116,7 +116,7 @@ internal readonly unsafe struct ResizeKernel if ((this.Length & 3) >= 2) { Vector256 pixels256_0 = Unsafe.As>(ref rowStartRef); - result256_0 = Vector256_.MultiplyAdd(result256_0, Vector256.Load(bufferStart), pixels256_0); + result256_0 = Vector256_.MultiplyAddEstimate(Vector256.Load(bufferStart), pixels256_0, result256_0); bufferStart += 8; rowStartRef = ref Unsafe.Add(ref rowStartRef, (nuint)2); @@ -127,7 +127,7 @@ internal readonly unsafe struct ResizeKernel if ((this.Length & 1) != 0) { Vector128 pixels128 = Unsafe.As>(ref rowStartRef); - result128 = Vector128_.MultiplyAdd(result128, Vector128.Load(bufferStart), pixels128); + result128 = Vector128_.MultiplyAddEstimate(Vector128.Load(bufferStart), pixels128, result128); } return result128.AsVector4(); diff --git a/tests/ImageSharp.Benchmarks/Bulk/FromVector4.cs b/tests/ImageSharp.Benchmarks/Bulk/FromVector4.cs index 13ab9527fa..dcd26c328d 100644 --- a/tests/ImageSharp.Benchmarks/Bulk/FromVector4.cs +++ b/tests/ImageSharp.Benchmarks/Bulk/FromVector4.cs @@ -159,13 +159,13 @@ public class FromVector4Bgra32P : FromVector4 AMD RYZEN AI MAX+ 395 w/ Radeon 8060S 3.00GHz, 1 CPU, 32 logical and 16 physical cores .NET 8.0.28, X64 RyuJIT x86-64-v4 - | Method | Count | Mean | Error | StdDev | Ratio | RatioSD | Code Size | Allocated | - |---------------------------- |------ |------------:|-----------:|----------:|------:|--------:|----------:|----------:| - | PixelOperations_Base | 64 | 59.73 ns | 19.644 ns | 1.077 ns | 1.00 | 0.02 | 1,152 B | - | - | PixelOperations_Specialized | 64 | 50.86 ns | 7.372 ns | 0.404 ns | 0.85 | 0.01 | 2,975 B | - | - | PixelOperations_Base | 256 | 214.03 ns | 70.798 ns | 3.881 ns | 1.00 | 0.02 | 1,152 B | - | - | PixelOperations_Specialized | 256 | 70.21 ns | 17.210 ns | 0.943 ns | 0.33 | 0.01 | 2,992 B | - | - | PixelOperations_Base | 2048 | 1,855.02 ns | 443.677 ns | 24.319 ns | 1.00 | 0.02 | 1,152 B | - | - | PixelOperations_Specialized | 2048 | 272.82 ns | 39.302 ns | 2.154 ns | 0.15 | 0.00 | 2,992 B | - | + | Method | Count | Mean | Error | StdDev | Ratio | Code Size | Allocated | + |---------------------------- |------ |------------:|---------:|----------:|------:|----------:|----------:| + | PixelOperations_Base | 64 | 53.35 ns | 0.208 ns | 0.311 ns | 1.00 | 1,152 B | - | + | PixelOperations_Specialized | 64 | 46.61 ns | 0.195 ns | 0.280 ns | 0.87 | 2,975 B | - | + | PixelOperations_Base | 256 | 212.11 ns | 0.719 ns | 1.076 ns | 1.00 | 1,152 B | - | + | PixelOperations_Specialized | 256 | 77.12 ns | 0.824 ns | 1.233 ns | 0.36 | 2,992 B | - | + | PixelOperations_Base | 2048 | 1,435.87 ns | 9.402 ns | 13.781 ns | 1.00 | 1,152 B | - | + | PixelOperations_Specialized | 2048 | 250.89 ns | 9.354 ns | 13.416 ns | 0.17 | 2,992 B | - | */ } diff --git a/tests/ImageSharp.Benchmarks/Bulk/ToVector4_Bgra32.cs b/tests/ImageSharp.Benchmarks/Bulk/ToVector4_Bgra32.cs index 9777a2f3fc..20c38d61f5 100644 --- a/tests/ImageSharp.Benchmarks/Bulk/ToVector4_Bgra32.cs +++ b/tests/ImageSharp.Benchmarks/Bulk/ToVector4_Bgra32.cs @@ -56,12 +56,12 @@ public class ToVector4_Bgra32P : ToVector4 // AMD RYZEN AI MAX+ 395 w/ Radeon 8060S 3.00GHz, 1 CPU, 32 logical and 16 physical cores // .NET 8.0.28, X64 RyuJIT x86-64-v4 // - // | Method | Count | Mean | Error | StdDev | Ratio | Code Size | Allocated | - // |---------------------------- |------ |------------:|-----------:|---------:|------:|----------:|----------:| - // | PixelOperations_Base | 64 | 58.08 ns | 10.122 ns | 0.555 ns | 1.00 | 932 B | - | - // | PixelOperations_Specialized | 64 | 79.63 ns | 9.080 ns | 0.498 ns | 1.37 | 3,688 B | - | - // | PixelOperations_Base | 256 | 224.99 ns | 46.742 ns | 2.562 ns | 1.00 | 958 B | - | - // | PixelOperations_Specialized | 256 | 99.41 ns | 9.728 ns | 0.533 ns | 0.44 | 3,688 B | - | - // | PixelOperations_Base | 2048 | 1,745.77 ns | 143.244 ns | 7.852 ns | 1.00 | 958 B | - | - // | PixelOperations_Specialized | 2048 | 293.00 ns | 44.137 ns | 2.419 ns | 0.17 | 3,704 B | - | + // | Method | Count | Mean | Error | StdDev | Ratio | RatioSD | Code Size | Allocated | + // |---------------------------- |------ |------------:|---------:|---------:|------:|--------:|----------:|----------:| + // | PixelOperations_Base | 64 | 58.05 ns | 3.204 ns | 4.795 ns | 1.01 | 0.11 | 932 B | - | + // | PixelOperations_Specialized | 64 | 84.78 ns | 0.467 ns | 0.669 ns | 1.47 | 0.12 | 3,731 B | - | + // | PixelOperations_Base | 256 | 208.93 ns | 0.345 ns | 0.484 ns | 1.00 | 0.00 | 958 B | - | + // | PixelOperations_Specialized | 256 | 106.50 ns | 0.280 ns | 0.410 ns | 0.51 | 0.00 | 3,738 B | - | + // | PixelOperations_Base | 2048 | 1,617.39 ns | 3.380 ns | 4.848 ns | 1.00 | 0.00 | 958 B | - | + // | PixelOperations_Specialized | 2048 | 269.90 ns | 0.666 ns | 0.976 ns | 0.17 | 0.00 | 3,735 B | - | } diff --git a/tests/ImageSharp.Tests/Color/ColorTests.cs b/tests/ImageSharp.Tests/Color/ColorTests.cs index 2594caeb5a..6c205888af 100644 --- a/tests/ImageSharp.Tests/Color/ColorTests.cs +++ b/tests/ImageSharp.Tests/Color/ColorTests.cs @@ -1,6 +1,7 @@ // Copyright (c) Six Labors. // Licensed under the Six Labors Split License. +using System.Numerics; using SixLabors.ImageSharp.PixelFormats; namespace SixLabors.ImageSharp.Tests; @@ -40,6 +41,36 @@ public partial class ColorTests Assert.Equal("80402080", color.ToHex()); } + [Fact] + public void ToScaledVector4ConvertsUnassociatedToAssociated() + { + Color color = Color.FromScaledVector(new Vector4(1F, 0.5F, 0.25F, 0.5F)); + + Vector4 actual = color.ToScaledVector4(PixelAlphaRepresentation.Associated); + + Assert.Equal(new Vector4(0.5F, 0.25F, 0.125F, 0.5F), actual); + } + + [Fact] + public void ToScaledVector4ConvertsAssociatedToUnassociated() + { + Color color = Color.FromScaledVector(new Vector4(0.5F, 0.25F, 0.125F, 0.5F), PixelAlphaRepresentation.Associated); + + Vector4 actual = color.ToScaledVector4(PixelAlphaRepresentation.Unassociated); + + Assert.Equal(new Vector4(1F, 0.5F, 0.25F, 0.5F), actual); + } + + [Fact] + public void EqualityDoesNotDependOnInternalAssociationStorage() + { + Color fromPixel = Color.FromPixel(new Rgba32P(64, 32, 16, 128)); + Color fromVector = Color.FromScaledVector(fromPixel.ToScaledVector4(), PixelAlphaRepresentation.Associated); + + Assert.Equal(fromPixel, fromVector); + Assert.Equal(fromPixel.GetHashCode(), fromVector.GetHashCode()); + } + [Theory] [InlineData(false)] [InlineData(true)] diff --git a/tests/ImageSharp.Tests/Common/SimdUtilsTests.cs b/tests/ImageSharp.Tests/Common/SimdUtilsTests.cs index aec8b6c43a..71335032c8 100644 --- a/tests/ImageSharp.Tests/Common/SimdUtilsTests.cs +++ b/tests/ImageSharp.Tests/Common/SimdUtilsTests.cs @@ -3,8 +3,10 @@ using System.Numerics; using System.Runtime.CompilerServices; +using System.Runtime.Intrinsics; using System.Runtime.Intrinsics.Arm; using System.Runtime.Intrinsics.X86; +using SixLabors.ImageSharp.Common.Helpers; using SixLabors.ImageSharp.PixelFormats; using SixLabors.ImageSharp.Tests.TestUtilities; using Xunit.Abstractions; @@ -133,7 +135,7 @@ public partial class SimdUtilsTests FeatureTestRunner.RunWithHwIntrinsicsFeature( RunTest, count, - HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX2); + HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX); } [Theory] @@ -142,13 +144,37 @@ public partial class SimdUtilsTests count, (s, d) => SimdUtils.ByteToNormalizedFloat(s.Span, d.Span)); + [Fact] + public void VectorMultiplyAddUsesRuntimeOrderAndFusedContract() => + FeatureTestRunner.RunWithHwIntrinsicsFeature( + RunVectorMultiplyAddUsesRuntimeOrderAndFusedContract, + HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic); + + private static void RunVectorMultiplyAddUsesRuntimeOrderAndFusedContract() + { + Assert.Equal(Vector128.Create(11F), Vector128_.MultiplyAddEstimate(Vector128.Create(2F), Vector128.Create(3F), Vector128.Create(5F))); + Assert.Equal(Vector256.Create(11F), Vector256_.MultiplyAddEstimate(Vector256.Create(2F), Vector256.Create(3F), Vector256.Create(5F))); + Assert.Equal(Vector512.Create(11F), Vector512_.MultiplyAddEstimate(Vector512.Create(2F), Vector512.Create(3F), Vector512.Create(5F))); + + // These associated-alpha components produce an exact midpoint that a separate multiply and add rounds incorrectly. + float left = (68F / byte.MaxValue) * .625F; + float right = 1F - (160F / byte.MaxValue); + float addend = ((50F / byte.MaxValue) + (50F / byte.MaxValue)) * left; + float expected = MathF.FusedMultiplyAdd(left, right, addend); + + Assert.Equal(Vector128.Create(expected), Vector128_.FusedMultiplyAdd(Vector128.Create(left), Vector128.Create(right), Vector128.Create(addend))); + Assert.Equal(Vector256.Create(expected), Vector256_.FusedMultiplyAdd(Vector256.Create(left), Vector256.Create(right), Vector256.Create(addend))); + Assert.Equal(Vector512.Create(expected), Vector512_.FusedMultiplyAdd(Vector512.Create(left), Vector512.Create(right), Vector512.Create(addend))); + } + private static void TestImpl_BulkConvertByteToNormalizedFloat( int count, Action, Memory> convert) { - byte[] source = new Random(count).GenerateRandomByteArray(count); + byte[] source = Enumerable.Range(0, count).Select(i => (byte)i).ToArray(); float[] result = new float[count]; - float[] expected = source.Select(b => b * (1F / byte.MaxValue)).ToArray(); + // A double-precision oracle keeps this expectation independent from either production implementation. + float[] expected = source.Select(b => (float)(b / (double)byte.MaxValue)).ToArray(); convert(source, result); diff --git a/tests/ImageSharp.Tests/Formats/Icon/Cur/CurEncoderTests.cs b/tests/ImageSharp.Tests/Formats/Icon/Cur/CurEncoderTests.cs index f895afbd51..c41f455cd1 100644 --- a/tests/ImageSharp.Tests/Formats/Icon/Cur/CurEncoderTests.cs +++ b/tests/ImageSharp.Tests/Formats/Icon/Cur/CurEncoderTests.cs @@ -4,7 +4,6 @@ using SixLabors.ImageSharp.Formats; using SixLabors.ImageSharp.Formats.Cur; using SixLabors.ImageSharp.Formats.Ico; -using SixLabors.ImageSharp.Formats.Png; using SixLabors.ImageSharp.PixelFormats; using SixLabors.ImageSharp.Tests.TestUtilities.ImageComparison; using static SixLabors.ImageSharp.Tests.TestImages.Cur; @@ -120,12 +119,6 @@ public class CurEncoderTests for (int x = 0; x < accessor.Width; x++) { - if (expectedColor != rowSpan[x]) - { - int xx = 0; - } - - Assert.Equal(expectedColor, rowSpan[x]); } } diff --git a/tests/ImageSharp.Tests/Formats/Qoi/ImageExtensionsTest.cs b/tests/ImageSharp.Tests/Formats/Qoi/ImageExtensionsTest.cs index 31ec27da0c..5ec23b1464 100644 --- a/tests/ImageSharp.Tests/Formats/Qoi/ImageExtensionsTest.cs +++ b/tests/ImageSharp.Tests/Formats/Qoi/ImageExtensionsTest.cs @@ -1,8 +1,8 @@ // Copyright (c) Six Labors. // Licensed under the Six Labors Split License. -using SixLabors.ImageSharp.Formats.Qoi; using SixLabors.ImageSharp.Formats; +using SixLabors.ImageSharp.Formats.Qoi; using SixLabors.ImageSharp.PixelFormats; namespace SixLabors.ImageSharp.Tests.Formats.Qoi; diff --git a/tests/ImageSharp.Tests/PixelFormats/AssociatedAlphaPixelTests.cs b/tests/ImageSharp.Tests/PixelFormats/AssociatedAlphaPixelTests.cs index dc4c108fcb..f34ca4e05e 100644 --- a/tests/ImageSharp.Tests/PixelFormats/AssociatedAlphaPixelTests.cs +++ b/tests/ImageSharp.Tests/PixelFormats/AssociatedAlphaPixelTests.cs @@ -5,7 +5,6 @@ using System.Numerics; using System.Runtime.CompilerServices; using System.Runtime.InteropServices; using SixLabors.ImageSharp.PixelFormats; -using SixLabors.ImageSharp.PixelFormats.PixelBlenders; namespace SixLabors.ImageSharp.Tests.PixelFormats; @@ -17,8 +16,6 @@ namespace SixLabors.ImageSharp.Tests.PixelFormats; public abstract class AssociatedAlphaPixelTests where TPixel : unmanaged, IPixel { - private static readonly ApproximateFloatComparer VectorComparer = new(.005F); - /// /// Gets the color channels described by the pixel format. /// @@ -42,75 +39,6 @@ public abstract class AssociatedAlphaPixelTests Assert.Equal(expectedComponentPrecision, componentInfo.GetComponentPrecision(i)); } } - - [Fact] - public void ScaledVectorConversionsUseAssociatedComponents() - { - Vector4 associated = new(.25F, .125F, .0625F, .5F); - - TPixel pixel = TPixel.FromScaledVector4(associated); - - Assert.Equal(associated, pixel.ToScaledVector4(), VectorComparer); - } - - [Fact] - public void FromRgba32AssociatesColorComponents() - { - Rgba32 source = new(192, 128, 64, 128); - Vector4 expected = source.ToScaledVector4(); - Numerics.Premultiply(ref expected); - - TPixel pixel = TPixel.FromRgba32(source); - - Assert.Equal(expected, pixel.ToScaledVector4(), VectorComparer); - } - - [Fact] - public void ToRgba32ReturnsUnassociatedColorComponents() - { - Rgba32 expected = new(192, 128, 64, 128); - TPixel pixel = TPixel.FromRgba32(expected); - - Rgba32 actual = pixel.ToRgba32(); - - AssertRgba32Equal(expected, actual, 3); - } - - [Fact] - public void ColorConversionsPreserveUnassociatedColor() - { - Rgba32 expected = new(192, 128, 64, 128); - TPixel source = TPixel.FromRgba32(expected); - - Color color = Color.FromPixel(source); - Rgba32 actual = color.ToPixel(); - TPixel roundTrip = color.ToPixel(); - - AssertRgba32Equal(expected, actual, 3); - Assert.Equal(source.ToScaledVector4(), roundTrip.ToScaledVector4(), VectorComparer); - } - - [Fact] - public void ScalarBlendingUsesUnassociatedColorValues() - { - TPixel background = TPixel.FromRgba32(new Rgba32(200, 40, 80, 160)); - TPixel source = TPixel.FromRgba32(new Rgba32(20, 180, 100, 96)); - PixelBlender associatedBlender = PixelOperations.Instance.GetPixelBlender(PixelColorBlendingMode.Normal, PixelAlphaCompositionMode.SrcOver); - PixelBlender unassociatedBlender = new DefaultPixelBlenders.NormalSrcOver(); - - Rgba32 expected = unassociatedBlender.Blend(background.ToRgba32(), source.ToRgba32(), .75F); - Rgba32 actual = associatedBlender.Blend(background, source, .75F).ToRgba32(); - - AssertRgba32Equal(expected, actual, 4); - } - - private static void AssertRgba32Equal(Rgba32 expected, Rgba32 actual, int tolerance) - { - Assert.InRange(Math.Abs(expected.R - actual.R), 0, tolerance); - Assert.InRange(Math.Abs(expected.G - actual.G), 0, tolerance); - Assert.InRange(Math.Abs(expected.B - actual.B), 0, tolerance); - Assert.InRange(Math.Abs(expected.A - actual.A), 0, tolerance); - } } /// @@ -205,6 +133,29 @@ public class HalfVector4PTests : AssociatedAlphaPixelTests Assert.Equal(new HalfVector4(associated).PackedValue, new HalfVector4P(associated).PackedValue); } + + [Fact] + public void ColorRoundTripDoesNotIntroduceAdditionalAssociationLoss() + { + const ulong zeroComponent = 0xBC00; + + // This is the first valid Half component/alpha pair for which an unassociate/reassociate round trip + // selects the adjacent component value instead of the direct scaled-vector conversion result. + HalfVector4P source = new() + { + PackedValue = 0x8BF5 | (zeroComponent << 16) | (zeroComponent << 32) | (0x05DFUL << 48) + }; + HalfVector4P expected = HalfVector4P.FromScaledVector4(source.ToScaledVector4()); + Color color = Color.FromPixel(source); + Color[] bulkColors = new Color[1]; + + Color.FromPixel(new[] { source }, bulkColors); + + Assert.Equal(source.ToScaledVector4(), color.ToScaledVector4()); + Assert.Equal(expected, color.ToPixel()); + Assert.Equal(source.ToScaledVector4(), bulkColors[0].ToScaledVector4()); + Assert.Equal(expected, bulkColors[0].ToPixel()); + } } /// @@ -248,17 +199,19 @@ public class AssociatedAlphaPackedPixelConversionTests private static void AssertLosslessRoundTrip() where TIntermediate : unmanaged, IPixel { - Rgba32P[] expected = - [ - new(0, 0, 0, 0), - new(1, 2, 3, 4), - new(31, 63, 95, 127), - new(64, 128, 192, 255), - new(255, 255, 255, 255), - ]; + Rgba32P[] expected = new Rgba32P[byte.MaxValue * 4 + 4]; TIntermediate[] intermediate = new TIntermediate[expected.Length]; Rgba32P[] actual = new Rgba32P[expected.Length]; + for (int component = 0; component <= byte.MaxValue; component++) + { + int index = component * 4; + expected[index] = new Rgba32P((byte)component, 0, 0, byte.MaxValue); + expected[index + 1] = new Rgba32P(0, (byte)component, 0, byte.MaxValue); + expected[index + 2] = new Rgba32P(0, 0, (byte)component, byte.MaxValue); + expected[index + 3] = new Rgba32P(0, 0, 0, (byte)component); + } + PixelOperations.Instance.From(Configuration.Default, expected, intermediate); PixelOperations.Instance.From(Configuration.Default, intermediate, actual); @@ -273,13 +226,16 @@ public class AssociatedAlphaPackedPixelConversionTests private static void AssertScalarAndBulkAssociatedVectorsAreEqual(Func createPixel) where TPixel : unmanaged, IPixel { - TPixel[] pixels = new TPixel[64]; + TPixel[] pixels = new TPixel[byte.MaxValue * 4 + 4]; Vector4[] actual = new Vector4[pixels.Length]; - for (int i = 0; i < pixels.Length; i++) + for (int component = 0; component <= byte.MaxValue; component++) { - int component = i * 4; - pixels[i] = createPixel((byte)component, (byte)(component + 1), (byte)(component + 2), (byte)(component + 3)); + int index = component * 4; + pixels[index] = createPixel((byte)component, 0, 0, byte.MaxValue); + pixels[index + 1] = createPixel(0, (byte)component, 0, byte.MaxValue); + pixels[index + 2] = createPixel(0, 0, (byte)component, byte.MaxValue); + pixels[index + 3] = createPixel(0, 0, 0, (byte)component); } PixelOperations.Instance.ToAssociatedScaledVector4(Configuration.Default, pixels, actual); @@ -319,6 +275,110 @@ public class AssociatedAlphaPackedPixelConversionTests /// public class AssociatedToUnassociatedPackedPixelConversionTests { + [Fact] + public void Rgba32ToRgba32PMatchesExactAssociationForEveryComponentAndAlpha() + => AssertUnsignedByteAssociation((red, green, blue, alpha) => new Rgba32P(red, green, blue, alpha)); + + [Fact] + public void Rgba32ToBgra32PMatchesExactAssociationForEveryComponentAndAlpha() + => AssertUnsignedByteAssociation((red, green, blue, alpha) => new Bgra32P(red, green, blue, alpha)); + + [Fact] + public void Rgba32ToArgb32PMatchesExactAssociationForEveryComponentAndAlpha() + => AssertUnsignedByteAssociation((red, green, blue, alpha) => new Argb32P(red, green, blue, alpha)); + + [Fact] + public void Rgba32ToAbgr32PMatchesExactAssociationForEveryComponentAndAlpha() + => AssertUnsignedByteAssociation((red, green, blue, alpha) => new Abgr32P(red, green, blue, alpha)); + + [Fact] + public void Rgba32ToNormalizedByte4PMatchesExactAssociationForEveryComponentAndAlpha() + { + const int pairCount = 65536; + const int channelCount = 3; + Rgba32[] source = new Rgba32[pairCount * channelCount]; + NormalizedByte4P[] expected = new NormalizedByte4P[source.Length]; + NormalizedByte4P[] actualBulk = new NormalizedByte4P[source.Length]; + int index = 0; + + for (int alpha = 0; alpha <= byte.MaxValue; alpha++) + { + // NormalizedByte4 has 254 intervals from -1 through 1, so alpha is first rounded to that destination grid. + int destinationAlpha = ((alpha * 254) + 127) / byte.MaxValue; + + for (int unassociated = 0; unassociated <= byte.MaxValue; unassociated++) + { + // Association uses the alpha representable by the destination, then rounds the result to the same 254-interval grid. + int associated = ((unassociated * destinationAlpha) + 127) / byte.MaxValue; + + source[index] = new Rgba32((byte)unassociated, 0, 0, (byte)alpha); + expected[index++] = CreateNormalizedByte4P(associated, 0, 0, destinationAlpha); + source[index] = new Rgba32(0, (byte)unassociated, 0, (byte)alpha); + expected[index++] = CreateNormalizedByte4P(0, associated, 0, destinationAlpha); + source[index] = new Rgba32(0, 0, (byte)unassociated, (byte)alpha); + expected[index++] = CreateNormalizedByte4P(0, 0, associated, destinationAlpha); + } + } + + PixelOperations.Instance.From(Configuration.Default, source, actualBulk); + + Assert.Equal(expected, actualBulk); + + for (int i = 0; i < source.Length; i++) + { + Assert.Equal(expected[i], NormalizedByte4P.FromRgba32(source[i])); + Assert.Equal(expected[i], Color.FromPixel(source[i]).ToPixel()); + } + } + + [Fact] + public void Rgba32ToHalfVector4PMatchesExactAssociationForEveryComponentAndAlpha() + { + const int pairCount = 65536; + const int channelCount = 3; + + // HalfVector4 maps scaled zero to native -1, whose IEEE 754 binary16 representation is 0xBC00. + const ulong zeroComponentBits = 0xBC00; + Rgba32[] source = new Rgba32[pairCount * channelCount]; + HalfVector4P[] expected = new HalfVector4P[source.Length]; + HalfVector4P[] actualBulk = new HalfVector4P[source.Length]; + int index = 0; + + for (int alpha = 0; alpha <= byte.MaxValue; alpha++) + { + // HalfVector4 stores normalized values over -1 through 1. Association must use the alpha value + // recovered from the destination half, rather than the higher-precision source alpha. + float normalizedAlpha = (float)(alpha / (double)byte.MaxValue); + float nativeAlpha = (normalizedAlpha * 2F) - 1F; + ushort alphaBits = BitConverter.HalfToUInt16Bits((Half)nativeAlpha); + float destinationAlpha = ((float)BitConverter.UInt16BitsToHalf(alphaBits) + 1F) * .5F; + + for (int unassociated = 0; unassociated <= byte.MaxValue; unassociated++) + { + float normalizedComponent = (float)(unassociated / (double)byte.MaxValue); + float associated = normalizedComponent * destinationAlpha; + ushort associatedBits = BitConverter.HalfToUInt16Bits((Half)((associated * 2F) - 1F)); + ulong alphaPacked = (ulong)alphaBits << 48; + + source[index] = new Rgba32((byte)unassociated, 0, 0, (byte)alpha); + expected[index++] = new HalfVector4P { PackedValue = associatedBits | (zeroComponentBits << 16) | (zeroComponentBits << 32) | alphaPacked }; + source[index] = new Rgba32(0, (byte)unassociated, 0, (byte)alpha); + expected[index++] = new HalfVector4P { PackedValue = zeroComponentBits | ((ulong)associatedBits << 16) | (zeroComponentBits << 32) | alphaPacked }; + source[index] = new Rgba32(0, 0, (byte)unassociated, (byte)alpha); + expected[index++] = new HalfVector4P { PackedValue = zeroComponentBits | (zeroComponentBits << 16) | ((ulong)associatedBits << 32) | alphaPacked }; + } + } + + PixelOperations.Instance.From(Configuration.Default, source, actualBulk); + + for (int i = 0; i < source.Length; i++) + { + Assert.Equal(expected[i], HalfVector4P.FromRgba32(source[i])); + Assert.Equal(expected[i], Color.FromPixel(source[i]).ToPixel()); + Assert.Equal(expected[i], actualBulk[i]); + } + } + [Fact] public void Rgba32PToRgba32ScalarRoundTripPreservesEveryValidAssociatedComponent() { @@ -344,6 +404,11 @@ public class AssociatedToUnassociatedPackedPixelConversionTests [Fact] public void ColorFromRgba32PPreservesEveryValidAssociatedComponent() { + const int pairCount = 32896; + Rgba32P[] sourcePixels = new Rgba32P[pairCount]; + Color[] bulkColors = new Color[pairCount]; + int index = 0; + for (int alpha = 0; alpha <= byte.MaxValue; alpha++) { for (int associated = 0; associated <= alpha; associated++) @@ -351,11 +416,19 @@ public class AssociatedToUnassociatedPackedPixelConversionTests byte unassociated = alpha == 0 ? (byte)0 : (byte)(((associated * byte.MaxValue) + (alpha / 2)) / alpha); Rgba32P source = new((byte)associated, 0, 0, (byte)alpha); Color color = Color.FromPixel(source); + sourcePixels[index++] = source; Assert.Equal(new Rgba32(unassociated, 0, 0, (byte)alpha), color.ToPixel()); Assert.Equal(source, color.ToPixel()); } } + + Color.FromPixel(sourcePixels, bulkColors); + + for (int i = 0; i < sourcePixels.Length; i++) + { + Assert.Equal(sourcePixels[i], bulkColors[i].ToPixel()); + } } [Fact] @@ -397,6 +470,11 @@ public class AssociatedToUnassociatedPackedPixelConversionTests [Fact] public void ColorFromNormalizedByte4PPreservesEveryValidAssociatedComponent() { + const int pairCount = 32640; + NormalizedByte4P[] sourcePixels = new NormalizedByte4P[pairCount]; + Color[] bulkColors = new Color[pairCount]; + int index = 0; + for (int alpha = 0; alpha < byte.MaxValue; alpha++) { for (int associated = 0; associated <= alpha; associated++) @@ -405,11 +483,19 @@ public class AssociatedToUnassociatedPackedPixelConversionTests byte unassociatedAlpha = (byte)(((alpha * byte.MaxValue) + 127) / 254); NormalizedByte4P source = CreateNormalizedByte4P(associated, 0, 0, alpha); Color color = Color.FromPixel(source); + sourcePixels[index++] = source; Assert.Equal(new Rgba32(unassociated, 0, 0, unassociatedAlpha), color.ToPixel()); Assert.Equal(source, color.ToPixel()); } } + + Color.FromPixel(sourcePixels, bulkColors); + + for (int i = 0; i < sourcePixels.Length; i++) + { + Assert.Equal(sourcePixels[i], bulkColors[i].ToPixel()); + } } [Fact] @@ -444,6 +530,15 @@ public class AssociatedToUnassociatedPackedPixelConversionTests Assert.Equal(expectedUnassociated, actualUnassociated); Assert.Equal(source, actualRoundTrip); + + for (int i = 0; i < source.Length; i++) + { + Bgra32 expected = expectedUnassociated[i]; + + // Scalar conversion and Color must preserve the same exact canonical value proven for the bulk path. + Assert.Equal(new Rgba32(expected.R, expected.G, expected.B, expected.A), source[i].ToRgba32()); + Assert.Equal(expected, Color.FromPixel(source[i]).ToPixel()); + } } private static void AssertUnsignedByteBulkRoundTrip(Func createPixel) @@ -479,6 +574,53 @@ public class AssociatedToUnassociatedPackedPixelConversionTests Assert.Equal(expectedUnassociated, actualUnassociated); Assert.Equal(source, actualRoundTrip); + + for (int i = 0; i < source.Length; i++) + { + Bgra32 expected = expectedUnassociated[i]; + + // Scalar conversion and Color must use the same exact canonical byte as the independently calculated bulk oracle. + Assert.Equal(new Rgba32(expected.R, expected.G, expected.B, expected.A), source[i].ToRgba32()); + Assert.Equal(expected, Color.FromPixel(source[i]).ToPixel()); + } + } + + private static void AssertUnsignedByteAssociation(Func createPixel) + where TPixel : unmanaged, IPixel + { + const int pairCount = 65536; + const int channelCount = 3; + Rgba32[] source = new Rgba32[pairCount * channelCount]; + TPixel[] expected = new TPixel[source.Length]; + TPixel[] actualBulk = new TPixel[source.Length]; + int index = 0; + + for (int alpha = 0; alpha <= byte.MaxValue; alpha++) + { + for (int unassociated = 0; unassociated <= byte.MaxValue; unassociated++) + { + // The destination stores round(unassociated * alpha / 255). The integer numerator adds half + // the divisor so the expected value is independent of the floating-point implementation. + byte associated = (byte)(((unassociated * alpha) + 127) / byte.MaxValue); + + source[index] = new Rgba32((byte)unassociated, 0, 0, (byte)alpha); + expected[index++] = createPixel(associated, 0, 0, (byte)alpha); + source[index] = new Rgba32(0, (byte)unassociated, 0, (byte)alpha); + expected[index++] = createPixel(0, associated, 0, (byte)alpha); + source[index] = new Rgba32(0, 0, (byte)unassociated, (byte)alpha); + expected[index++] = createPixel(0, 0, associated, (byte)alpha); + } + } + + PixelOperations.Instance.From(Configuration.Default, source, actualBulk); + + Assert.Equal(expected, actualBulk); + + for (int i = 0; i < source.Length; i++) + { + Assert.Equal(expected[i], TPixel.FromRgba32(source[i])); + Assert.Equal(expected[i], Color.FromPixel(source[i]).ToPixel()); + } } private static NormalizedByte4P CreateNormalizedByte4P(int red, int green, int blue, int alpha) @@ -638,7 +780,7 @@ public class AssociatedDestinationAlphaQuantizationTests foreach (byte component in components) { int expectedRed = ((component * expectedAlpha) + 127) / byte.MaxValue; - Vector4 expected = new Vector4(expectedRed, 0, 0, expectedAlpha) * (1F / byte.MaxValue); + Vector4 expected = new Vector4(expectedRed, 0, 0, expectedAlpha) / byte.MaxValue; Assert.Equal(expected, TPixel.FromRgba64(source[index]).ToScaledVector4()); Assert.Equal(expected, actualBulk[index].ToScaledVector4()); diff --git a/tests/ImageSharp.Tests/PixelFormats/PixelBlenderTests.cs b/tests/ImageSharp.Tests/PixelFormats/PixelBlenderTests.cs index d61ab88a63..a851e2fcf5 100644 --- a/tests/ImageSharp.Tests/PixelFormats/PixelBlenderTests.cs +++ b/tests/ImageSharp.Tests/PixelFormats/PixelBlenderTests.cs @@ -229,7 +229,12 @@ public class PixelBlenderTests HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic); [Fact] - public void AssociatedHardLightDestAtopRoundsExactMidpointAwayFromZero() + public void AssociatedHardLightDestAtopRoundsExactMidpointAwayFromZero() => + FeatureTestRunner.RunWithHwIntrinsicsFeature( + RunAssociatedHardLightDestAtopRoundsExactMidpointAwayFromZero, + HwIntrinsics.AllowAll | HwIntrinsics.DisableAVX512F | HwIntrinsics.DisableAVX | HwIntrinsics.DisableHWIntrinsic); + + private static void RunAssociatedHardLightDestAtopRoundsExactMidpointAwayFromZero() { Rgba32P background = Rgba32P.FromRgba32(new Rgba32(220, 80, 40, 160)); Rgba32P source = Rgba32P.FromRgba32(new Rgba32(20, 180, 120, 96)); diff --git a/tests/Images/External/ReferenceOutput/Filters/BlackWhiteTest/ApplyBlackWhiteFilter_Rgba32_TestPattern48x48.png b/tests/Images/External/ReferenceOutput/Filters/BlackWhiteTest/ApplyBlackWhiteFilter_Rgba32_TestPattern48x48.png index 4f57775dc0..aa561ace81 100644 --- a/tests/Images/External/ReferenceOutput/Filters/BlackWhiteTest/ApplyBlackWhiteFilter_Rgba32_TestPattern48x48.png +++ b/tests/Images/External/ReferenceOutput/Filters/BlackWhiteTest/ApplyBlackWhiteFilter_Rgba32_TestPattern48x48.png @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:65d3a099dd24fcee2fc86c63b8664ebd81d7b3c530168da5f4e50ee0ca925496 -size 759 +oid sha256:d76ae5ac1a0aebb9d76b2fa5c6b1fc109e5fcbdc6c41d9681a76ade9c86fab79 +size 748