mirror of https://github.com/SixLabors/ImageSharp
committed by
GitHub
1504 changed files with 58888 additions and 8065 deletions
@ -1 +1 @@ |
|||||
Subproject commit 06a733983486638b9e38197c7c6eb197ecac43e6 |
Subproject commit 33cb12ca77f919b44de56f344d2627cc2a108c3a |
||||
@ -0,0 +1,23 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
namespace SixLabors.ImageSharp |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// The byte order of the data stream.
|
||||
|
/// </summary>
|
||||
|
public enum ByteOrder |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// The big-endian byte order (Motorola).
|
||||
|
/// Most-significant byte comes first, and ends with the least-significant byte.
|
||||
|
/// </summary>
|
||||
|
BigEndian, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The little-endian byte order (Intel).
|
||||
|
/// Least-significant byte comes first and ends with the most-significant byte.
|
||||
|
/// </summary>
|
||||
|
LittleEndian |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,21 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Common.Helpers |
||||
|
{ |
||||
|
internal readonly struct ExifResolutionValues |
||||
|
{ |
||||
|
public ExifResolutionValues(ushort resolutionUnit, double? horizontalResolution, double? verticalResolution) |
||||
|
{ |
||||
|
this.ResolutionUnit = resolutionUnit; |
||||
|
this.HorizontalResolution = horizontalResolution; |
||||
|
this.VerticalResolution = verticalResolution; |
||||
|
} |
||||
|
|
||||
|
public ushort ResolutionUnit { get; } |
||||
|
|
||||
|
public double? HorizontalResolution { get; } |
||||
|
|
||||
|
public double? VerticalResolution { get; } |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,32 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using System; |
||||
|
using System.Runtime.InteropServices; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Provides information about the .NET runtime installation.
|
||||
|
/// Many methods defer to <see cref="RuntimeInformation"/> when available.
|
||||
|
/// </summary>
|
||||
|
internal static class RuntimeEnvironment |
||||
|
{ |
||||
|
private static readonly Lazy<bool> IsNetCoreLazy = new Lazy<bool>(() => FrameworkDescription.StartsWith(".NET Core", StringComparison.OrdinalIgnoreCase)); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets a value indicating whether the .NET installation is .NET Core 3.1 or lower.
|
||||
|
/// </summary>
|
||||
|
public static bool IsNetCore => IsNetCoreLazy.Value; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets the name of the .NET installation on which an app is running.
|
||||
|
/// </summary>
|
||||
|
public static string FrameworkDescription => RuntimeInformation.FrameworkDescription; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Indicates whether the current application is running on the specified platform.
|
||||
|
/// </summary>
|
||||
|
public static bool IsOSPlatform(OSPlatform osPlatform) => RuntimeInformation.IsOSPlatform(osPlatform); |
||||
|
} |
||||
|
} |
||||
@ -1,7 +1,7 @@ |
|||||
// Copyright (c) Six Labors.
|
// Copyright (c) Six Labors.
|
||||
// Licensed under the Apache License, Version 2.0.
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
namespace SixLabors.ImageSharp.Formats.Png.Zlib |
namespace SixLabors.ImageSharp.Compression.Zlib |
||||
{ |
{ |
||||
/// <content>
|
/// <content>
|
||||
/// Contains precalulated tables for scalar calculations.
|
/// Contains precalulated tables for scalar calculations.
|
||||
@ -0,0 +1,217 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using System; |
||||
|
using System.Runtime.CompilerServices; |
||||
|
using System.Runtime.InteropServices; |
||||
|
#if SUPPORTS_RUNTIME_INTRINSICS
|
||||
|
using System.Runtime.Intrinsics; |
||||
|
using System.Runtime.Intrinsics.X86; |
||||
|
#endif
|
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Compression.Zlib |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Calculates the 32 bit Cyclic Redundancy Check (CRC) checksum of a given buffer
|
||||
|
/// according to the IEEE 802.3 specification.
|
||||
|
/// </summary>
|
||||
|
internal static partial class Crc32 |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// The default initial seed value of a Crc32 checksum calculation.
|
||||
|
/// </summary>
|
||||
|
public const uint SeedValue = 0U; |
||||
|
|
||||
|
#if SUPPORTS_RUNTIME_INTRINSICS
|
||||
|
private const int MinBufferSize = 64; |
||||
|
private const int ChunksizeMask = 15; |
||||
|
|
||||
|
// Definitions of the bit-reflected domain constants k1, k2, k3, etc and
|
||||
|
// the CRC32+Barrett polynomials given at the end of the paper.
|
||||
|
private static readonly ulong[] K05Poly = |
||||
|
{ |
||||
|
0x0154442bd4, 0x01c6e41596, // k1, k2
|
||||
|
0x01751997d0, 0x00ccaa009e, // k3, k4
|
||||
|
0x0163cd6124, 0x0000000000, // k5, k0
|
||||
|
0x01db710641, 0x01f7011641 // polynomial
|
||||
|
}; |
||||
|
#endif
|
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Calculates the CRC checksum with the bytes taken from the span.
|
||||
|
/// </summary>
|
||||
|
/// <param name="buffer">The readonly span of bytes.</param>
|
||||
|
/// <returns>The <see cref="uint"/>.</returns>
|
||||
|
[MethodImpl(InliningOptions.ShortMethod)] |
||||
|
public static uint Calculate(ReadOnlySpan<byte> buffer) |
||||
|
=> Calculate(SeedValue, buffer); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Calculates the CRC checksum with the bytes taken from the span and seed.
|
||||
|
/// </summary>
|
||||
|
/// <param name="crc">The input CRC value.</param>
|
||||
|
/// <param name="buffer">The readonly span of bytes.</param>
|
||||
|
/// <returns>The <see cref="uint"/>.</returns>
|
||||
|
[MethodImpl(InliningOptions.ShortMethod)] |
||||
|
public static uint Calculate(uint crc, ReadOnlySpan<byte> buffer) |
||||
|
{ |
||||
|
if (buffer.IsEmpty) |
||||
|
{ |
||||
|
return crc; |
||||
|
} |
||||
|
|
||||
|
#if SUPPORTS_RUNTIME_INTRINSICS
|
||||
|
if (Sse41.IsSupported && Pclmulqdq.IsSupported && buffer.Length >= MinBufferSize) |
||||
|
{ |
||||
|
return ~CalculateSse(~crc, buffer); |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
return ~CalculateScalar(~crc, buffer); |
||||
|
} |
||||
|
#else
|
||||
|
return ~CalculateScalar(~crc, buffer); |
||||
|
#endif
|
||||
|
} |
||||
|
|
||||
|
#if SUPPORTS_RUNTIME_INTRINSICS
|
||||
|
// Based on https://github.com/chromium/chromium/blob/master/third_party/zlib/crc32_simd.c
|
||||
|
[MethodImpl(InliningOptions.HotPath | InliningOptions.ShortMethod)] |
||||
|
private static unsafe uint CalculateSse(uint crc, ReadOnlySpan<byte> buffer) |
||||
|
{ |
||||
|
int chunksize = buffer.Length & ~ChunksizeMask; |
||||
|
int length = chunksize; |
||||
|
|
||||
|
fixed (byte* bufferPtr = buffer) |
||||
|
{ |
||||
|
fixed (ulong* k05PolyPtr = K05Poly) |
||||
|
{ |
||||
|
byte* localBufferPtr = bufferPtr; |
||||
|
ulong* localK05PolyPtr = k05PolyPtr; |
||||
|
|
||||
|
// There's at least one block of 64.
|
||||
|
Vector128<ulong> x1 = Sse2.LoadVector128((ulong*)(localBufferPtr + 0x00)); |
||||
|
Vector128<ulong> x2 = Sse2.LoadVector128((ulong*)(localBufferPtr + 0x10)); |
||||
|
Vector128<ulong> x3 = Sse2.LoadVector128((ulong*)(localBufferPtr + 0x20)); |
||||
|
Vector128<ulong> x4 = Sse2.LoadVector128((ulong*)(localBufferPtr + 0x30)); |
||||
|
Vector128<ulong> x5; |
||||
|
|
||||
|
x1 = Sse2.Xor(x1, Sse2.ConvertScalarToVector128UInt32(crc).AsUInt64()); |
||||
|
|
||||
|
// k1, k2
|
||||
|
Vector128<ulong> x0 = Sse2.LoadVector128(localK05PolyPtr + 0x0); |
||||
|
|
||||
|
localBufferPtr += 64; |
||||
|
length -= 64; |
||||
|
|
||||
|
// Parallel fold blocks of 64, if any.
|
||||
|
while (length >= 64) |
||||
|
{ |
||||
|
x5 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x00); |
||||
|
Vector128<ulong> x6 = Pclmulqdq.CarrylessMultiply(x2, x0, 0x00); |
||||
|
Vector128<ulong> x7 = Pclmulqdq.CarrylessMultiply(x3, x0, 0x00); |
||||
|
Vector128<ulong> x8 = Pclmulqdq.CarrylessMultiply(x4, x0, 0x00); |
||||
|
|
||||
|
x1 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x11); |
||||
|
x2 = Pclmulqdq.CarrylessMultiply(x2, x0, 0x11); |
||||
|
x3 = Pclmulqdq.CarrylessMultiply(x3, x0, 0x11); |
||||
|
x4 = Pclmulqdq.CarrylessMultiply(x4, x0, 0x11); |
||||
|
|
||||
|
Vector128<ulong> y5 = Sse2.LoadVector128((ulong*)(localBufferPtr + 0x00)); |
||||
|
Vector128<ulong> y6 = Sse2.LoadVector128((ulong*)(localBufferPtr + 0x10)); |
||||
|
Vector128<ulong> y7 = Sse2.LoadVector128((ulong*)(localBufferPtr + 0x20)); |
||||
|
Vector128<ulong> y8 = Sse2.LoadVector128((ulong*)(localBufferPtr + 0x30)); |
||||
|
|
||||
|
x1 = Sse2.Xor(x1, x5); |
||||
|
x2 = Sse2.Xor(x2, x6); |
||||
|
x3 = Sse2.Xor(x3, x7); |
||||
|
x4 = Sse2.Xor(x4, x8); |
||||
|
|
||||
|
x1 = Sse2.Xor(x1, y5); |
||||
|
x2 = Sse2.Xor(x2, y6); |
||||
|
x3 = Sse2.Xor(x3, y7); |
||||
|
x4 = Sse2.Xor(x4, y8); |
||||
|
|
||||
|
localBufferPtr += 64; |
||||
|
length -= 64; |
||||
|
} |
||||
|
|
||||
|
// Fold into 128-bits.
|
||||
|
// k3, k4
|
||||
|
x0 = Sse2.LoadVector128(k05PolyPtr + 0x2); |
||||
|
|
||||
|
x5 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x00); |
||||
|
x1 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x11); |
||||
|
x1 = Sse2.Xor(x1, x2); |
||||
|
x1 = Sse2.Xor(x1, x5); |
||||
|
|
||||
|
x5 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x00); |
||||
|
x1 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x11); |
||||
|
x1 = Sse2.Xor(x1, x3); |
||||
|
x1 = Sse2.Xor(x1, x5); |
||||
|
|
||||
|
x5 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x00); |
||||
|
x1 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x11); |
||||
|
x1 = Sse2.Xor(x1, x4); |
||||
|
x1 = Sse2.Xor(x1, x5); |
||||
|
|
||||
|
// Single fold blocks of 16, if any.
|
||||
|
while (length >= 16) |
||||
|
{ |
||||
|
x2 = Sse2.LoadVector128((ulong*)localBufferPtr); |
||||
|
|
||||
|
x5 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x00); |
||||
|
x1 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x11); |
||||
|
x1 = Sse2.Xor(x1, x2); |
||||
|
x1 = Sse2.Xor(x1, x5); |
||||
|
|
||||
|
localBufferPtr += 16; |
||||
|
length -= 16; |
||||
|
} |
||||
|
|
||||
|
// Fold 128 - bits to 64 - bits.
|
||||
|
x2 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x10); |
||||
|
x3 = Vector128.Create(~0, 0, ~0, 0).AsUInt64(); // _mm_setr_epi32 on x86
|
||||
|
x1 = Sse2.ShiftRightLogical128BitLane(x1, 8); |
||||
|
x1 = Sse2.Xor(x1, x2); |
||||
|
|
||||
|
// k5, k0
|
||||
|
x0 = Sse2.LoadScalarVector128(localK05PolyPtr + 0x4); |
||||
|
|
||||
|
x2 = Sse2.ShiftRightLogical128BitLane(x1, 4); |
||||
|
x1 = Sse2.And(x1, x3); |
||||
|
x1 = Pclmulqdq.CarrylessMultiply(x1, x0, 0x00); |
||||
|
x1 = Sse2.Xor(x1, x2); |
||||
|
|
||||
|
// Barret reduce to 32-bits.
|
||||
|
// polynomial
|
||||
|
x0 = Sse2.LoadVector128(localK05PolyPtr + 0x6); |
||||
|
|
||||
|
x2 = Sse2.And(x1, x3); |
||||
|
x2 = Pclmulqdq.CarrylessMultiply(x2, x0, 0x10); |
||||
|
x2 = Sse2.And(x2, x3); |
||||
|
x2 = Pclmulqdq.CarrylessMultiply(x2, x0, 0x00); |
||||
|
x1 = Sse2.Xor(x1, x2); |
||||
|
|
||||
|
crc = (uint)Sse41.Extract(x1.AsInt32(), 1); |
||||
|
return buffer.Length - chunksize == 0 ? crc : CalculateScalar(crc, buffer.Slice(chunksize)); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
#endif
|
||||
|
|
||||
|
[MethodImpl(InliningOptions.HotPath | InliningOptions.ShortMethod)] |
||||
|
private static uint CalculateScalar(uint crc, ReadOnlySpan<byte> buffer) |
||||
|
{ |
||||
|
ref uint crcTableRef = ref MemoryMarshal.GetReference(CrcTable.AsSpan()); |
||||
|
ref byte bufferRef = ref MemoryMarshal.GetReference(buffer); |
||||
|
|
||||
|
for (int i = 0; i < buffer.Length; i++) |
||||
|
{ |
||||
|
crc = Unsafe.Add(ref crcTableRef, (int)((crc ^ Unsafe.Add(ref bufferRef, i)) & 0xFF)) ^ (crc >> 8); |
||||
|
} |
||||
|
|
||||
|
return crc; |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,81 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Compression.Zlib |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Provides enumeration of available deflate compression levels.
|
||||
|
/// </summary>
|
||||
|
public enum DeflateCompressionLevel |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Level 0. Equivalent to <see cref="NoCompression"/>.
|
||||
|
/// </summary>
|
||||
|
Level0 = 0, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// No compression. Equivalent to <see cref="Level0"/>.
|
||||
|
/// </summary>
|
||||
|
NoCompression = Level0, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 1. Equivalent to <see cref="BestSpeed"/>.
|
||||
|
/// </summary>
|
||||
|
Level1 = 1, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Best speed compression level.
|
||||
|
/// </summary>
|
||||
|
BestSpeed = Level1, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 2.
|
||||
|
/// </summary>
|
||||
|
Level2 = 2, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 3.
|
||||
|
/// </summary>
|
||||
|
Level3 = 3, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 4.
|
||||
|
/// </summary>
|
||||
|
Level4 = 4, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 5.
|
||||
|
/// </summary>
|
||||
|
Level5 = 5, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 6. Equivalent to <see cref="DefaultCompression"/>.
|
||||
|
/// </summary>
|
||||
|
Level6 = 6, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The default compression level. Equivalent to <see cref="Level6"/>.
|
||||
|
/// </summary>
|
||||
|
DefaultCompression = Level6, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 7.
|
||||
|
/// </summary>
|
||||
|
Level7 = 7, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 8.
|
||||
|
/// </summary>
|
||||
|
Level8 = 8, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Level 9. Equivalent to <see cref="BestCompression"/>.
|
||||
|
/// </summary>
|
||||
|
Level9 = 9, |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Best compression level. Equivalent to <see cref="Level9"/>.
|
||||
|
/// </summary>
|
||||
|
BestCompression = Level9, |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,149 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
#if SUPPORTS_RUNTIME_INTRINSICS
|
||||
|
using System; |
||||
|
using System.Numerics; |
||||
|
using System.Runtime.CompilerServices; |
||||
|
using System.Runtime.InteropServices; |
||||
|
using System.Runtime.Intrinsics; |
||||
|
using System.Runtime.Intrinsics.X86; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components |
||||
|
{ |
||||
|
internal partial struct Block8x8F |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// A number of rows of 8 scalar coefficients each in <see cref="Block8x8F"/>
|
||||
|
/// </summary>
|
||||
|
public const int RowCount = 8; |
||||
|
|
||||
|
[FieldOffset(0)] |
||||
|
public Vector256<float> V0; |
||||
|
[FieldOffset(32)] |
||||
|
public Vector256<float> V1; |
||||
|
[FieldOffset(64)] |
||||
|
public Vector256<float> V2; |
||||
|
[FieldOffset(96)] |
||||
|
public Vector256<float> V3; |
||||
|
[FieldOffset(128)] |
||||
|
public Vector256<float> V4; |
||||
|
[FieldOffset(160)] |
||||
|
public Vector256<float> V5; |
||||
|
[FieldOffset(192)] |
||||
|
public Vector256<float> V6; |
||||
|
[FieldOffset(224)] |
||||
|
public Vector256<float> V7; |
||||
|
|
||||
|
private static readonly Vector256<int> MultiplyIntoInt16ShuffleMask = Vector256.Create(0, 1, 4, 5, 2, 3, 6, 7); |
||||
|
|
||||
|
private static unsafe void MultiplyIntoInt16_Avx2(ref Block8x8F a, ref Block8x8F b, ref Block8x8 dest) |
||||
|
{ |
||||
|
DebugGuard.IsTrue(Avx2.IsSupported, "Avx2 support is required to run this operation!"); |
||||
|
|
||||
|
ref Vector256<float> aBase = ref a.V0; |
||||
|
ref Vector256<float> bBase = ref b.V0; |
||||
|
|
||||
|
ref Vector256<short> destRef = ref dest.V01; |
||||
|
|
||||
|
for (nint i = 0; i < 8; i += 2) |
||||
|
{ |
||||
|
Vector256<int> row0 = Avx.ConvertToVector256Int32(Avx.Multiply(Unsafe.Add(ref aBase, i + 0), Unsafe.Add(ref bBase, i + 0))); |
||||
|
Vector256<int> row1 = Avx.ConvertToVector256Int32(Avx.Multiply(Unsafe.Add(ref aBase, i + 1), Unsafe.Add(ref bBase, i + 1))); |
||||
|
|
||||
|
Vector256<short> row = Avx2.PackSignedSaturate(row0, row1); |
||||
|
row = Avx2.PermuteVar8x32(row.AsInt32(), MultiplyIntoInt16ShuffleMask).AsInt16(); |
||||
|
|
||||
|
Unsafe.Add(ref destRef, (IntPtr)((uint)i / 2)) = row; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
private static void MultiplyIntoInt16_Sse2(ref Block8x8F a, ref Block8x8F b, ref Block8x8 dest) |
||||
|
{ |
||||
|
DebugGuard.IsTrue(Sse2.IsSupported, "Sse2 support is required to run this operation!"); |
||||
|
|
||||
|
ref Vector128<float> aBase = ref Unsafe.As<Block8x8F, Vector128<float>>(ref a); |
||||
|
ref Vector128<float> bBase = ref Unsafe.As<Block8x8F, Vector128<float>>(ref b); |
||||
|
|
||||
|
ref Vector128<short> destBase = ref Unsafe.As<Block8x8, Vector128<short>>(ref dest); |
||||
|
|
||||
|
for (int i = 0; i < 16; i += 2) |
||||
|
{ |
||||
|
Vector128<int> left = Sse2.ConvertToVector128Int32(Sse.Multiply(Unsafe.Add(ref aBase, i + 0), Unsafe.Add(ref bBase, i + 0))); |
||||
|
Vector128<int> right = Sse2.ConvertToVector128Int32(Sse.Multiply(Unsafe.Add(ref aBase, i + 1), Unsafe.Add(ref bBase, i + 1))); |
||||
|
|
||||
|
Vector128<short> row = Sse2.PackSignedSaturate(left, right); |
||||
|
Unsafe.Add(ref destBase, (IntPtr)((uint)i / 2)) = row; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
private void TransposeInplace_Avx() |
||||
|
{ |
||||
|
// https://stackoverflow.com/questions/25622745/transpose-an-8x8-float-using-avx-avx2/25627536#25627536
|
||||
|
Vector256<float> r0 = Avx.InsertVector128( |
||||
|
this.V0, |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V4L), |
||||
|
1); |
||||
|
|
||||
|
Vector256<float> r1 = Avx.InsertVector128( |
||||
|
this.V1, |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V5L), |
||||
|
1); |
||||
|
|
||||
|
Vector256<float> r2 = Avx.InsertVector128( |
||||
|
this.V2, |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V6L), |
||||
|
1); |
||||
|
|
||||
|
Vector256<float> r3 = Avx.InsertVector128( |
||||
|
this.V3, |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V7L), |
||||
|
1); |
||||
|
|
||||
|
Vector256<float> r4 = Avx.InsertVector128( |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V0R).ToVector256(), |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V4R), |
||||
|
1); |
||||
|
|
||||
|
Vector256<float> r5 = Avx.InsertVector128( |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V1R).ToVector256(), |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V5R), |
||||
|
1); |
||||
|
|
||||
|
Vector256<float> r6 = Avx.InsertVector128( |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V2R).ToVector256(), |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V6R), |
||||
|
1); |
||||
|
|
||||
|
Vector256<float> r7 = Avx.InsertVector128( |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V3R).ToVector256(), |
||||
|
Unsafe.As<Vector4, Vector128<float>>(ref this.V7R), |
||||
|
1); |
||||
|
|
||||
|
Vector256<float> t0 = Avx.UnpackLow(r0, r1); |
||||
|
Vector256<float> t2 = Avx.UnpackLow(r2, r3); |
||||
|
Vector256<float> v = Avx.Shuffle(t0, t2, 0x4E); |
||||
|
this.V0 = Avx.Blend(t0, v, 0xCC); |
||||
|
this.V1 = Avx.Blend(t2, v, 0x33); |
||||
|
|
||||
|
Vector256<float> t4 = Avx.UnpackLow(r4, r5); |
||||
|
Vector256<float> t6 = Avx.UnpackLow(r6, r7); |
||||
|
v = Avx.Shuffle(t4, t6, 0x4E); |
||||
|
this.V4 = Avx.Blend(t4, v, 0xCC); |
||||
|
this.V5 = Avx.Blend(t6, v, 0x33); |
||||
|
|
||||
|
Vector256<float> t1 = Avx.UnpackHigh(r0, r1); |
||||
|
Vector256<float> t3 = Avx.UnpackHigh(r2, r3); |
||||
|
v = Avx.Shuffle(t1, t3, 0x4E); |
||||
|
this.V2 = Avx.Blend(t1, v, 0xCC); |
||||
|
this.V3 = Avx.Blend(t3, v, 0x33); |
||||
|
|
||||
|
Vector256<float> t5 = Avx.UnpackHigh(r4, r5); |
||||
|
Vector256<float> t7 = Avx.UnpackHigh(r6, r7); |
||||
|
v = Avx.Shuffle(t5, t7, 0x4E); |
||||
|
this.V6 = Avx.Blend(t5, v, 0xCC); |
||||
|
this.V7 = Avx.Blend(t7, v, 0x33); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
#endif
|
||||
@ -1,181 +0,0 @@ |
|||||
// Copyright (c) Six Labors.
|
|
||||
// Licensed under the Apache License, Version 2.0.
|
|
||||
|
|
||||
using System; |
|
||||
using System.Buffers; |
|
||||
using System.Numerics; |
|
||||
using System.Threading; |
|
||||
using SixLabors.ImageSharp.Advanced; |
|
||||
using SixLabors.ImageSharp.Memory; |
|
||||
using SixLabors.ImageSharp.PixelFormats; |
|
||||
using JpegColorConverter = SixLabors.ImageSharp.Formats.Jpeg.Components.Decoder.ColorConverters.JpegColorConverter; |
|
||||
|
|
||||
namespace SixLabors.ImageSharp.Formats.Jpeg.Components.Decoder |
|
||||
{ |
|
||||
/// <summary>
|
|
||||
/// Encapsulates the execution od post-processing algorithms to be applied on a <see cref="IRawJpegData"/> to produce a valid <see cref="Image{TPixel}"/>: <br/>
|
|
||||
/// (1) Dequantization <br/>
|
|
||||
/// (2) IDCT <br/>
|
|
||||
/// (3) Color conversion form one of the <see cref="JpegColorSpace"/>-s into a <see cref="Vector4"/> buffer of RGBA values <br/>
|
|
||||
/// (4) Packing <see cref="Image{TPixel}"/> pixels from the <see cref="Vector4"/> buffer. <br/>
|
|
||||
/// These operations are executed in <see cref="NumberOfPostProcessorSteps"/> steps.
|
|
||||
/// <see cref="PixelRowsPerStep"/> image rows are converted in one step,
|
|
||||
/// which means that size of the allocated memory is limited (does not depend on <see cref="ImageFrame.Height"/>).
|
|
||||
/// </summary>
|
|
||||
internal class JpegImagePostProcessor : IDisposable |
|
||||
{ |
|
||||
private readonly Configuration configuration; |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// The number of block rows to be processed in one Step.
|
|
||||
/// </summary>
|
|
||||
public const int BlockRowsPerStep = 4; |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// The number of image pixel rows to be processed in one step.
|
|
||||
/// </summary>
|
|
||||
public const int PixelRowsPerStep = 4 * 8; |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Temporal buffer to store a row of colors.
|
|
||||
/// </summary>
|
|
||||
private readonly IMemoryOwner<Vector4> rgbaBuffer; |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// The <see cref="JpegColorConverter"/> corresponding to the current <see cref="JpegColorSpace"/> determined by <see cref="IRawJpegData.ColorSpace"/>.
|
|
||||
/// </summary>
|
|
||||
private readonly JpegColorConverter colorConverter; |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Initializes a new instance of the <see cref="JpegImagePostProcessor"/> class.
|
|
||||
/// </summary>
|
|
||||
/// <param name="configuration">The <see cref="Configuration"/> to configure internal operations.</param>
|
|
||||
/// <param name="rawJpeg">The <see cref="IRawJpegData"/> representing the uncompressed spectral Jpeg data</param>
|
|
||||
public JpegImagePostProcessor(Configuration configuration, IRawJpegData rawJpeg) |
|
||||
{ |
|
||||
this.configuration = configuration; |
|
||||
this.RawJpeg = rawJpeg; |
|
||||
IJpegComponent c0 = rawJpeg.Components[0]; |
|
||||
this.NumberOfPostProcessorSteps = c0.SizeInBlocks.Height / BlockRowsPerStep; |
|
||||
this.PostProcessorBufferSize = new Size(c0.SizeInBlocks.Width * 8, PixelRowsPerStep); |
|
||||
|
|
||||
MemoryAllocator memoryAllocator = configuration.MemoryAllocator; |
|
||||
|
|
||||
this.ComponentProcessors = new JpegComponentPostProcessor[rawJpeg.Components.Length]; |
|
||||
for (int i = 0; i < rawJpeg.Components.Length; i++) |
|
||||
{ |
|
||||
this.ComponentProcessors[i] = new JpegComponentPostProcessor(memoryAllocator, this, rawJpeg.Components[i]); |
|
||||
} |
|
||||
|
|
||||
this.rgbaBuffer = memoryAllocator.Allocate<Vector4>(rawJpeg.ImageSizeInPixels.Width); |
|
||||
this.colorConverter = JpegColorConverter.GetConverter(rawJpeg.ColorSpace, rawJpeg.Precision); |
|
||||
} |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Gets the <see cref="JpegComponentPostProcessor"/> instances.
|
|
||||
/// </summary>
|
|
||||
public JpegComponentPostProcessor[] ComponentProcessors { get; } |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Gets the <see cref="IRawJpegData"/> to be processed.
|
|
||||
/// </summary>
|
|
||||
public IRawJpegData RawJpeg { get; } |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Gets the total number of post processor steps deduced from the height of the image and <see cref="PixelRowsPerStep"/>.
|
|
||||
/// </summary>
|
|
||||
public int NumberOfPostProcessorSteps { get; } |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Gets the size of the temporary buffers we need to allocate into <see cref="JpegComponentPostProcessor.ColorBuffer"/>.
|
|
||||
/// </summary>
|
|
||||
public Size PostProcessorBufferSize { get; } |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Gets the value of the counter that grows by each step by <see cref="PixelRowsPerStep"/>.
|
|
||||
/// </summary>
|
|
||||
public int PixelRowCounter { get; private set; } |
|
||||
|
|
||||
/// <inheritdoc />
|
|
||||
public void Dispose() |
|
||||
{ |
|
||||
foreach (JpegComponentPostProcessor cpp in this.ComponentProcessors) |
|
||||
{ |
|
||||
cpp.Dispose(); |
|
||||
} |
|
||||
|
|
||||
this.rgbaBuffer.Dispose(); |
|
||||
} |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Process all pixels into 'destination'. The image dimensions should match <see cref="RawJpeg"/>.
|
|
||||
/// </summary>
|
|
||||
/// <typeparam name="TPixel">The pixel type</typeparam>
|
|
||||
/// <param name="destination">The destination image</param>
|
|
||||
/// <param name="cancellationToken">The token to request cancellation.</param>
|
|
||||
public void PostProcess<TPixel>(ImageFrame<TPixel> destination, CancellationToken cancellationToken) |
|
||||
where TPixel : unmanaged, IPixel<TPixel> |
|
||||
{ |
|
||||
this.PixelRowCounter = 0; |
|
||||
|
|
||||
if (this.RawJpeg.ImageSizeInPixels != destination.Size()) |
|
||||
{ |
|
||||
throw new ArgumentException("Input image is not of the size of the processed one!"); |
|
||||
} |
|
||||
|
|
||||
while (this.PixelRowCounter < this.RawJpeg.ImageSizeInPixels.Height) |
|
||||
{ |
|
||||
cancellationToken.ThrowIfCancellationRequested(); |
|
||||
this.DoPostProcessorStep(destination); |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Execute one step processing <see cref="PixelRowsPerStep"/> pixel rows into 'destination'.
|
|
||||
/// </summary>
|
|
||||
/// <typeparam name="TPixel">The pixel type</typeparam>
|
|
||||
/// <param name="destination">The destination image.</param>
|
|
||||
public void DoPostProcessorStep<TPixel>(ImageFrame<TPixel> destination) |
|
||||
where TPixel : unmanaged, IPixel<TPixel> |
|
||||
{ |
|
||||
foreach (JpegComponentPostProcessor cpp in this.ComponentProcessors) |
|
||||
{ |
|
||||
cpp.CopyBlocksToColorBuffer(); |
|
||||
} |
|
||||
|
|
||||
this.ConvertColorsInto(destination); |
|
||||
|
|
||||
this.PixelRowCounter += PixelRowsPerStep; |
|
||||
} |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Convert and copy <see cref="PixelRowsPerStep"/> row of colors into 'destination' starting at row <see cref="PixelRowCounter"/>.
|
|
||||
/// </summary>
|
|
||||
/// <typeparam name="TPixel">The pixel type</typeparam>
|
|
||||
/// <param name="destination">The destination image</param>
|
|
||||
private void ConvertColorsInto<TPixel>(ImageFrame<TPixel> destination) |
|
||||
where TPixel : unmanaged, IPixel<TPixel> |
|
||||
{ |
|
||||
int maxY = Math.Min(destination.Height, this.PixelRowCounter + PixelRowsPerStep); |
|
||||
|
|
||||
var buffers = new Buffer2D<float>[this.ComponentProcessors.Length]; |
|
||||
for (int i = 0; i < this.ComponentProcessors.Length; i++) |
|
||||
{ |
|
||||
buffers[i] = this.ComponentProcessors[i].ColorBuffer; |
|
||||
} |
|
||||
|
|
||||
for (int yy = this.PixelRowCounter; yy < maxY; yy++) |
|
||||
{ |
|
||||
int y = yy - this.PixelRowCounter; |
|
||||
|
|
||||
var values = new JpegColorConverter.ComponentValues(buffers, y); |
|
||||
this.colorConverter.ConvertToRgba(values, this.rgbaBuffer.GetSpan()); |
|
||||
|
|
||||
Span<TPixel> destRow = destination.GetPixelRowSpan(yy); |
|
||||
|
|
||||
// TODO: Investigate if slicing is actually necessary
|
|
||||
PixelOperations<TPixel>.Instance.FromVector4Destructive(this.configuration, this.rgbaBuffer.GetSpan().Slice(0, destRow.Length), destRow); |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
@ -1,144 +0,0 @@ |
|||||
// Copyright (c) Six Labors.
|
|
||||
// Licensed under the Apache License, Version 2.0.
|
|
||||
|
|
||||
namespace SixLabors.ImageSharp.Formats.Jpeg.Components.Decoder |
|
||||
{ |
|
||||
/// <summary>
|
|
||||
/// Provides methods to evaluate the quality of an image.
|
|
||||
/// Ported from <see href="https://github.com/ImageMagick/ImageMagick/blob/f362c02083d27211b913c6e44794f0ac6edaf2bd/coders/jpeg.c#L855"/>
|
|
||||
/// </summary>
|
|
||||
internal static class QualityEvaluator |
|
||||
{ |
|
||||
private static readonly int[] Hash = new int[101] |
|
||||
{ |
|
||||
1020, 1015, 932, 848, 780, 735, 702, 679, 660, 645, |
|
||||
632, 623, 613, 607, 600, 594, 589, 585, 581, 571, |
|
||||
555, 542, 529, 514, 494, 474, 457, 439, 424, 410, |
|
||||
397, 386, 373, 364, 351, 341, 334, 324, 317, 309, |
|
||||
299, 294, 287, 279, 274, 267, 262, 257, 251, 247, |
|
||||
243, 237, 232, 227, 222, 217, 213, 207, 202, 198, |
|
||||
192, 188, 183, 177, 173, 168, 163, 157, 153, 148, |
|
||||
143, 139, 132, 128, 125, 119, 115, 108, 104, 99, |
|
||||
94, 90, 84, 79, 74, 70, 64, 59, 55, 49, |
|
||||
45, 40, 34, 30, 25, 20, 15, 11, 6, 4, |
|
||||
0 |
|
||||
}; |
|
||||
|
|
||||
private static readonly int[] Sums = new int[101] |
|
||||
{ |
|
||||
32640, 32635, 32266, 31495, 30665, 29804, 29146, 28599, 28104, |
|
||||
27670, 27225, 26725, 26210, 25716, 25240, 24789, 24373, 23946, |
|
||||
23572, 22846, 21801, 20842, 19949, 19121, 18386, 17651, 16998, |
|
||||
16349, 15800, 15247, 14783, 14321, 13859, 13535, 13081, 12702, |
|
||||
12423, 12056, 11779, 11513, 11135, 10955, 10676, 10392, 10208, |
|
||||
9928, 9747, 9564, 9369, 9193, 9017, 8822, 8639, 8458, |
|
||||
8270, 8084, 7896, 7710, 7527, 7347, 7156, 6977, 6788, |
|
||||
6607, 6422, 6236, 6054, 5867, 5684, 5495, 5305, 5128, |
|
||||
4945, 4751, 4638, 4442, 4248, 4065, 3888, 3698, 3509, |
|
||||
3326, 3139, 2957, 2775, 2586, 2405, 2216, 2037, 1846, |
|
||||
1666, 1483, 1297, 1109, 927, 735, 554, 375, 201, |
|
||||
128, 0 |
|
||||
}; |
|
||||
|
|
||||
private static readonly int[] Hash1 = new int[101] |
|
||||
{ |
|
||||
510, 505, 422, 380, 355, 338, 326, 318, 311, 305, |
|
||||
300, 297, 293, 291, 288, 286, 284, 283, 281, 280, |
|
||||
279, 278, 277, 273, 262, 251, 243, 233, 225, 218, |
|
||||
211, 205, 198, 193, 186, 181, 177, 172, 168, 164, |
|
||||
158, 156, 152, 148, 145, 142, 139, 136, 133, 131, |
|
||||
129, 126, 123, 120, 118, 115, 113, 110, 107, 105, |
|
||||
102, 100, 97, 94, 92, 89, 87, 83, 81, 79, |
|
||||
76, 74, 70, 68, 66, 63, 61, 57, 55, 52, |
|
||||
50, 48, 44, 42, 39, 37, 34, 31, 29, 26, |
|
||||
24, 21, 18, 16, 13, 11, 8, 6, 3, 2, |
|
||||
0 |
|
||||
}; |
|
||||
|
|
||||
private static readonly int[] Sums1 = new int[101] |
|
||||
{ |
|
||||
16320, 16315, 15946, 15277, 14655, 14073, 13623, 13230, 12859, |
|
||||
12560, 12240, 11861, 11456, 11081, 10714, 10360, 10027, 9679, |
|
||||
9368, 9056, 8680, 8331, 7995, 7668, 7376, 7084, 6823, |
|
||||
6562, 6345, 6125, 5939, 5756, 5571, 5421, 5240, 5086, |
|
||||
4976, 4829, 4719, 4616, 4463, 4393, 4280, 4166, 4092, |
|
||||
3980, 3909, 3835, 3755, 3688, 3621, 3541, 3467, 3396, |
|
||||
3323, 3247, 3170, 3096, 3021, 2952, 2874, 2804, 2727, |
|
||||
2657, 2583, 2509, 2437, 2362, 2290, 2211, 2136, 2068, |
|
||||
1996, 1915, 1858, 1773, 1692, 1620, 1552, 1477, 1398, |
|
||||
1326, 1251, 1179, 1109, 1031, 961, 884, 814, 736, |
|
||||
667, 592, 518, 441, 369, 292, 221, 151, 86, |
|
||||
64, 0 |
|
||||
}; |
|
||||
|
|
||||
/// <summary>
|
|
||||
/// Returns an estimated quality of the image based on the quantization tables.
|
|
||||
/// </summary>
|
|
||||
/// <param name="quantizationTables">The quantization tables.</param>
|
|
||||
/// <returns>The <see cref="int"/>.</returns>
|
|
||||
public static int EstimateQuality(Block8x8F[] quantizationTables) |
|
||||
{ |
|
||||
int quality = 75; |
|
||||
float sum = 0; |
|
||||
|
|
||||
for (int i = 0; i < quantizationTables.Length; i++) |
|
||||
{ |
|
||||
ref Block8x8F qTable = ref quantizationTables[i]; |
|
||||
|
|
||||
if (!qTable.Equals(default)) |
|
||||
{ |
|
||||
for (int j = 0; j < Block8x8F.Size; j++) |
|
||||
{ |
|
||||
sum += qTable[j]; |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
ref Block8x8F qTable0 = ref quantizationTables[0]; |
|
||||
ref Block8x8F qTable1 = ref quantizationTables[1]; |
|
||||
|
|
||||
if (!qTable0.Equals(default)) |
|
||||
{ |
|
||||
if (!qTable1.Equals(default)) |
|
||||
{ |
|
||||
quality = (int)(qTable0[2] |
|
||||
+ qTable0[53] |
|
||||
+ qTable1[0] |
|
||||
+ qTable1[Block8x8F.Size - 1]); |
|
||||
|
|
||||
for (int i = 0; i < 100; i++) |
|
||||
{ |
|
||||
if (quality < Hash[i] && sum < Sums[i]) |
|
||||
{ |
|
||||
continue; |
|
||||
} |
|
||||
|
|
||||
if (((quality <= Hash[i]) && (sum <= Sums[i])) || (i >= 50)) |
|
||||
{ |
|
||||
return i + 1; |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
else |
|
||||
{ |
|
||||
quality = (int)(qTable0[2] + qTable0[53]); |
|
||||
|
|
||||
for (int i = 0; i < 100; i++) |
|
||||
{ |
|
||||
if (quality < Hash1[i] && sum < Sums1[i]) |
|
||||
{ |
|
||||
continue; |
|
||||
} |
|
||||
|
|
||||
if (((quality <= Hash1[i]) && (sum <= Sums1[i])) || (i >= 50)) |
|
||||
{ |
|
||||
return i + 1; |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
return quality; |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
@ -0,0 +1,44 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using SixLabors.ImageSharp.Formats.Jpeg.Components.Decoder.ColorConverters; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components.Decoder |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Converter used to convert jpeg spectral data.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// This is tightly coupled with <see cref="HuffmanScanDecoder"/> and <see cref="JpegDecoderCore"/>.
|
||||
|
/// </remarks>
|
||||
|
internal abstract class SpectralConverter |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Injects jpeg image decoding metadata.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// This is guaranteed to be called only once at SOF marker by <see cref="HuffmanScanDecoder"/>.
|
||||
|
/// </remarks>
|
||||
|
/// <param name="frame"><see cref="JpegFrame"/> instance containing decoder-specific parameters.</param>
|
||||
|
/// <param name="jpegData"><see cref="IRawJpegData"/> instance containing decoder-specific parameters.</param>
|
||||
|
public abstract void InjectFrameData(JpegFrame frame, IRawJpegData jpegData); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Called once per spectral stride for each component in <see cref="HuffmanScanDecoder"/>.
|
||||
|
/// This is called only for baseline interleaved jpegs.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// Spectral 'stride' doesn't particularly mean 'single stride'.
|
||||
|
/// Actual stride height depends on the subsampling factor of the given component.
|
||||
|
/// </remarks>
|
||||
|
public abstract void ConvertStrideBaseline(); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets the color converter.
|
||||
|
/// </summary>
|
||||
|
/// <param name="frame">The jpeg frame with the color space to convert to.</param>
|
||||
|
/// <param name="jpegData">The raw JPEG data.</param>
|
||||
|
/// <returns>The color converter.</returns>
|
||||
|
protected virtual JpegColorConverter GetColorConverter(JpegFrame frame, IRawJpegData jpegData) => JpegColorConverter.GetConverter(jpegData.ColorSpace, frame.Precision); |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,172 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using System; |
||||
|
using System.Buffers; |
||||
|
using System.Numerics; |
||||
|
using System.Threading; |
||||
|
using SixLabors.ImageSharp.Formats.Jpeg.Components.Decoder.ColorConverters; |
||||
|
using SixLabors.ImageSharp.Memory; |
||||
|
using SixLabors.ImageSharp.PixelFormats; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components.Decoder |
||||
|
{ |
||||
|
internal class SpectralConverter<TPixel> : SpectralConverter, IDisposable |
||||
|
where TPixel : unmanaged, IPixel<TPixel> |
||||
|
{ |
||||
|
private readonly Configuration configuration; |
||||
|
|
||||
|
private readonly CancellationToken cancellationToken; |
||||
|
|
||||
|
private JpegComponentPostProcessor[] componentProcessors; |
||||
|
|
||||
|
private JpegColorConverter colorConverter; |
||||
|
|
||||
|
// private IMemoryOwner<Vector4> rgbaBuffer;
|
||||
|
private IMemoryOwner<byte> rgbBuffer; |
||||
|
|
||||
|
private IMemoryOwner<TPixel> paddedProxyPixelRow; |
||||
|
|
||||
|
private Buffer2D<TPixel> pixelBuffer; |
||||
|
|
||||
|
private int blockRowsPerStep; |
||||
|
|
||||
|
private int pixelRowsPerStep; |
||||
|
|
||||
|
private int pixelRowCounter; |
||||
|
|
||||
|
public SpectralConverter(Configuration configuration, CancellationToken cancellationToken) |
||||
|
{ |
||||
|
this.configuration = configuration; |
||||
|
this.cancellationToken = cancellationToken; |
||||
|
} |
||||
|
|
||||
|
private bool Converted => this.pixelRowCounter >= this.pixelBuffer.Height; |
||||
|
|
||||
|
public Buffer2D<TPixel> GetPixelBuffer() |
||||
|
{ |
||||
|
if (!this.Converted) |
||||
|
{ |
||||
|
int steps = (int)Math.Ceiling(this.pixelBuffer.Height / (float)this.pixelRowsPerStep); |
||||
|
|
||||
|
for (int step = 0; step < steps; step++) |
||||
|
{ |
||||
|
this.cancellationToken.ThrowIfCancellationRequested(); |
||||
|
this.ConvertNextStride(step); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
return this.pixelBuffer; |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
public override void InjectFrameData(JpegFrame frame, IRawJpegData jpegData) |
||||
|
{ |
||||
|
MemoryAllocator allocator = this.configuration.MemoryAllocator; |
||||
|
|
||||
|
// iteration data
|
||||
|
IJpegComponent c0 = frame.Components[0]; |
||||
|
|
||||
|
const int blockPixelHeight = 8; |
||||
|
this.blockRowsPerStep = c0.SamplingFactors.Height; |
||||
|
this.pixelRowsPerStep = this.blockRowsPerStep * blockPixelHeight; |
||||
|
|
||||
|
// pixel buffer for resulting image
|
||||
|
this.pixelBuffer = allocator.Allocate2D<TPixel>(frame.PixelWidth, frame.PixelHeight); |
||||
|
this.paddedProxyPixelRow = allocator.Allocate<TPixel>(frame.PixelWidth + 3); |
||||
|
|
||||
|
// component processors from spectral to Rgba32
|
||||
|
var postProcessorBufferSize = new Size(c0.SizeInBlocks.Width * 8, this.pixelRowsPerStep); |
||||
|
this.componentProcessors = new JpegComponentPostProcessor[frame.Components.Length]; |
||||
|
for (int i = 0; i < this.componentProcessors.Length; i++) |
||||
|
{ |
||||
|
this.componentProcessors[i] = new JpegComponentPostProcessor(allocator, frame, jpegData, postProcessorBufferSize, frame.Components[i]); |
||||
|
} |
||||
|
|
||||
|
// single 'stride' rgba32 buffer for conversion between spectral and TPixel
|
||||
|
// this.rgbaBuffer = allocator.Allocate<Vector4>(frame.PixelWidth);
|
||||
|
this.rgbBuffer = allocator.Allocate<byte>(frame.PixelWidth * 3); |
||||
|
|
||||
|
// color converter from Rgba32 to TPixel
|
||||
|
this.colorConverter = this.GetColorConverter(frame, jpegData); |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
public override void ConvertStrideBaseline() |
||||
|
{ |
||||
|
// Convert next pixel stride using single spectral `stride'
|
||||
|
// Note that zero passing eliminates the need of virtual call from JpegComponentPostProcessor
|
||||
|
this.ConvertNextStride(spectralStep: 0); |
||||
|
|
||||
|
// Clear spectral stride - this is VERY important as jpeg possibly won't fill entire buffer each stride
|
||||
|
// Which leads to decoding artifacts
|
||||
|
// Note that this code clears all buffers of the post processors, it's their responsibility to allocate only single stride
|
||||
|
foreach (JpegComponentPostProcessor cpp in this.componentProcessors) |
||||
|
{ |
||||
|
cpp.ClearSpectralBuffers(); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
public void Dispose() |
||||
|
{ |
||||
|
if (this.componentProcessors != null) |
||||
|
{ |
||||
|
foreach (JpegComponentPostProcessor cpp in this.componentProcessors) |
||||
|
{ |
||||
|
cpp.Dispose(); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
this.rgbBuffer?.Dispose(); |
||||
|
this.paddedProxyPixelRow?.Dispose(); |
||||
|
} |
||||
|
|
||||
|
private void ConvertNextStride(int spectralStep) |
||||
|
{ |
||||
|
int maxY = Math.Min(this.pixelBuffer.Height, this.pixelRowCounter + this.pixelRowsPerStep); |
||||
|
|
||||
|
var buffers = new Buffer2D<float>[this.componentProcessors.Length]; |
||||
|
for (int i = 0; i < this.componentProcessors.Length; i++) |
||||
|
{ |
||||
|
this.componentProcessors[i].CopyBlocksToColorBuffer(spectralStep); |
||||
|
buffers[i] = this.componentProcessors[i].ColorBuffer; |
||||
|
} |
||||
|
|
||||
|
int width = this.pixelBuffer.Width; |
||||
|
|
||||
|
for (int yy = this.pixelRowCounter; yy < maxY; yy++) |
||||
|
{ |
||||
|
int y = yy - this.pixelRowCounter; |
||||
|
|
||||
|
var values = new JpegColorConverter.ComponentValues(buffers, y); |
||||
|
|
||||
|
this.colorConverter.ConvertToRgbInplace(values); |
||||
|
values = values.Slice(0, width); // slice away Jpeg padding
|
||||
|
|
||||
|
Span<byte> r = this.rgbBuffer.Slice(0, width); |
||||
|
Span<byte> g = this.rgbBuffer.Slice(width, width); |
||||
|
Span<byte> b = this.rgbBuffer.Slice(width * 2, width); |
||||
|
|
||||
|
SimdUtils.NormalizedFloatToByteSaturate(values.Component0, r); |
||||
|
SimdUtils.NormalizedFloatToByteSaturate(values.Component1, g); |
||||
|
SimdUtils.NormalizedFloatToByteSaturate(values.Component2, b); |
||||
|
|
||||
|
// PackFromRgbPlanes expects the destination to be padded, so try to get padded span containing extra elements from the next row.
|
||||
|
// If we can't get such a padded row because we are on a MemoryGroup boundary or at the last row,
|
||||
|
// pack pixels to a temporary, padded proxy buffer, then copy the relevant values to the destination row.
|
||||
|
if (this.pixelBuffer.TryGetPaddedRowSpan(yy, 3, out Span<TPixel> destRow)) |
||||
|
{ |
||||
|
PixelOperations<TPixel>.Instance.PackFromRgbPlanes(this.configuration, r, g, b, destRow); |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
Span<TPixel> proxyRow = this.paddedProxyPixelRow.GetSpan(); |
||||
|
PixelOperations<TPixel>.Instance.PackFromRgbPlanes(this.configuration, r, g, b, proxyRow); |
||||
|
proxyRow.Slice(0, width).CopyTo(this.pixelBuffer.GetRowSpan(yy)); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
this.pixelRowCounter += this.pixelRowsPerStep; |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,689 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using System; |
||||
|
using System.IO; |
||||
|
using System.Numerics; |
||||
|
using System.Runtime.CompilerServices; |
||||
|
using System.Runtime.InteropServices; |
||||
|
using System.Threading; |
||||
|
using SixLabors.ImageSharp.Memory; |
||||
|
using SixLabors.ImageSharp.PixelFormats; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components.Encoder |
||||
|
{ |
||||
|
internal class HuffmanScanEncoder |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Maximum number of bytes encoded jpeg 8x8 block can occupy.
|
||||
|
/// It's highly unlikely for block to occupy this much space - it's a theoretical limit.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// Where 16 is maximum huffman code binary length according to itu
|
||||
|
/// specs. 10 is maximum value binary length, value comes from discrete
|
||||
|
/// cosine tranform with value range: [-1024..1023]. Block stores
|
||||
|
/// 8x8 = 64 values thus multiplication by 64. Then divided by 8 to get
|
||||
|
/// the number of bytes. This value is then multiplied by
|
||||
|
/// <see cref="MaxBytesPerBlockMultiplier"/> for performance reasons.
|
||||
|
/// </remarks>
|
||||
|
private const int MaxBytesPerBlock = (16 + 10) * 64 / 8 * MaxBytesPerBlockMultiplier; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Multiplier used within cache buffers size calculation.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// <para>
|
||||
|
/// Theoretically, <see cref="MaxBytesPerBlock"/> bytes buffer can fit
|
||||
|
/// exactly one minimal coding unit. In reality, coding blocks occupy much
|
||||
|
/// less space than the theoretical maximum - this can be exploited.
|
||||
|
/// If temporal buffer size is multiplied by at least 2, second half of
|
||||
|
/// the resulting buffer will be used as an overflow 'guard' if next
|
||||
|
/// block would occupy maximum number of bytes. While first half may fit
|
||||
|
/// many blocks before needing to flush.
|
||||
|
/// </para>
|
||||
|
/// <para>
|
||||
|
/// This is subject to change. This can be equal to 1 but recomended
|
||||
|
/// value is 2 or even greater - futher benchmarking needed.
|
||||
|
/// </para>
|
||||
|
/// </remarks>
|
||||
|
private const int MaxBytesPerBlockMultiplier = 2; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// <see cref="streamWriteBuffer"/> size multiplier.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// Jpeg specification requiers to insert 'stuff' bytes after each
|
||||
|
/// 0xff byte value. Worst case scenarion is when all bytes are 0xff.
|
||||
|
/// While it's highly unlikely (if not impossible) to get such
|
||||
|
/// combination, it's theoretically possible so buffer size must be guarded.
|
||||
|
/// </remarks>
|
||||
|
private const int OutputBufferLengthMultiplier = 2; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Compiled huffman tree to encode given values.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>Yields codewords by index consisting of [run length | bitsize].</remarks>
|
||||
|
private HuffmanLut[] huffmanTables; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Emitted bits 'micro buffer' before being transferred to the <see cref="emitBuffer"/>.
|
||||
|
/// </summary>
|
||||
|
private uint accumulatedBits; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Buffer for temporal storage of huffman rle encoding bit data.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// Encoding bits are assembled to 4 byte unsigned integers and then copied to this buffer.
|
||||
|
/// This process does NOT include inserting stuff bytes.
|
||||
|
/// </remarks>
|
||||
|
private readonly uint[] emitBuffer; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Buffer for temporal storage which is then written to the output stream.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// Encoding bits from <see cref="emitBuffer"/> are copied to this byte buffer including stuff bytes.
|
||||
|
/// </remarks>
|
||||
|
private readonly byte[] streamWriteBuffer; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Number of jagged bits stored in <see cref="accumulatedBits"/>
|
||||
|
/// </summary>
|
||||
|
private int bitCount; |
||||
|
|
||||
|
private int emitWriteIndex; |
||||
|
|
||||
|
private Block8x8 tempBlock; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The output stream. All attempted writes after the first error become no-ops.
|
||||
|
/// </summary>
|
||||
|
private readonly Stream target; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Initializes a new instance of the <see cref="HuffmanScanEncoder"/> class.
|
||||
|
/// </summary>
|
||||
|
/// <param name="blocksPerCodingUnit">Amount of encoded 8x8 blocks per single jpeg macroblock.</param>
|
||||
|
/// <param name="outputStream">Output stream for saving encoded data.</param>
|
||||
|
public HuffmanScanEncoder(int blocksPerCodingUnit, Stream outputStream) |
||||
|
{ |
||||
|
int emitBufferByteLength = MaxBytesPerBlock * blocksPerCodingUnit; |
||||
|
this.emitBuffer = new uint[emitBufferByteLength / sizeof(uint)]; |
||||
|
this.emitWriteIndex = this.emitBuffer.Length; |
||||
|
|
||||
|
this.streamWriteBuffer = new byte[emitBufferByteLength * OutputBufferLengthMultiplier]; |
||||
|
|
||||
|
this.target = outputStream; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets a value indicating whether <see cref="emitBuffer"/> is full
|
||||
|
/// and must be flushed using <see cref="FlushToStream()"/>
|
||||
|
/// before encoding next 8x8 coding block.
|
||||
|
/// </summary>
|
||||
|
private bool IsStreamFlushNeeded |
||||
|
{ |
||||
|
[MethodImpl(MethodImplOptions.AggressiveInlining)] |
||||
|
get => this.emitWriteIndex < (uint)this.emitBuffer.Length / 2; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Encodes the image with no subsampling.
|
||||
|
/// </summary>
|
||||
|
/// <typeparam name="TPixel">The pixel format.</typeparam>
|
||||
|
/// <param name="pixels">The pixel accessor providing access to the image pixels.</param>
|
||||
|
/// <param name="luminanceQuantTable">Luminance quantization table provided by the callee.</param>
|
||||
|
/// <param name="chrominanceQuantTable">Chrominance quantization table provided by the callee.</param>
|
||||
|
/// <param name="cancellationToken">The token to monitor for cancellation.</param>
|
||||
|
public void Encode444<TPixel>(Image<TPixel> pixels, ref Block8x8F luminanceQuantTable, ref Block8x8F chrominanceQuantTable, CancellationToken cancellationToken) |
||||
|
where TPixel : unmanaged, IPixel<TPixel> |
||||
|
{ |
||||
|
FastFloatingPointDCT.AdjustToFDCT(ref luminanceQuantTable); |
||||
|
FastFloatingPointDCT.AdjustToFDCT(ref chrominanceQuantTable); |
||||
|
|
||||
|
this.huffmanTables = HuffmanLut.TheHuffmanLut; |
||||
|
|
||||
|
// ReSharper disable once InconsistentNaming
|
||||
|
int prevDCY = 0, prevDCCb = 0, prevDCCr = 0; |
||||
|
|
||||
|
ImageFrame<TPixel> frame = pixels.Frames.RootFrame; |
||||
|
Buffer2D<TPixel> pixelBuffer = frame.PixelBuffer; |
||||
|
RowOctet<TPixel> currentRows = default; |
||||
|
|
||||
|
var pixelConverter = new YCbCrForwardConverter444<TPixel>(frame); |
||||
|
|
||||
|
for (int y = 0; y < pixels.Height; y += 8) |
||||
|
{ |
||||
|
cancellationToken.ThrowIfCancellationRequested(); |
||||
|
currentRows.Update(pixelBuffer, y); |
||||
|
|
||||
|
for (int x = 0; x < pixels.Width; x += 8) |
||||
|
{ |
||||
|
pixelConverter.Convert(x, y, ref currentRows); |
||||
|
|
||||
|
prevDCY = this.WriteBlock( |
||||
|
QuantIndex.Luminance, |
||||
|
prevDCY, |
||||
|
ref pixelConverter.Y, |
||||
|
ref luminanceQuantTable); |
||||
|
|
||||
|
prevDCCb = this.WriteBlock( |
||||
|
QuantIndex.Chrominance, |
||||
|
prevDCCb, |
||||
|
ref pixelConverter.Cb, |
||||
|
ref chrominanceQuantTable); |
||||
|
|
||||
|
prevDCCr = this.WriteBlock( |
||||
|
QuantIndex.Chrominance, |
||||
|
prevDCCr, |
||||
|
ref pixelConverter.Cr, |
||||
|
ref chrominanceQuantTable); |
||||
|
|
||||
|
if (this.IsStreamFlushNeeded) |
||||
|
{ |
||||
|
this.FlushToStream(); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
this.FlushRemainingBytes(); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Encodes the image with subsampling. The Cb and Cr components are each subsampled
|
||||
|
/// at a factor of 2 both horizontally and vertically.
|
||||
|
/// </summary>
|
||||
|
/// <typeparam name="TPixel">The pixel format.</typeparam>
|
||||
|
/// <param name="pixels">The pixel accessor providing access to the image pixels.</param>
|
||||
|
/// <param name="luminanceQuantTable">Luminance quantization table provided by the callee.</param>
|
||||
|
/// <param name="chrominanceQuantTable">Chrominance quantization table provided by the callee.</param>
|
||||
|
/// <param name="cancellationToken">The token to monitor for cancellation.</param>
|
||||
|
public void Encode420<TPixel>(Image<TPixel> pixels, ref Block8x8F luminanceQuantTable, ref Block8x8F chrominanceQuantTable, CancellationToken cancellationToken) |
||||
|
where TPixel : unmanaged, IPixel<TPixel> |
||||
|
{ |
||||
|
FastFloatingPointDCT.AdjustToFDCT(ref luminanceQuantTable); |
||||
|
FastFloatingPointDCT.AdjustToFDCT(ref chrominanceQuantTable); |
||||
|
|
||||
|
this.huffmanTables = HuffmanLut.TheHuffmanLut; |
||||
|
|
||||
|
// ReSharper disable once InconsistentNaming
|
||||
|
int prevDCY = 0, prevDCCb = 0, prevDCCr = 0; |
||||
|
ImageFrame<TPixel> frame = pixels.Frames.RootFrame; |
||||
|
Buffer2D<TPixel> pixelBuffer = frame.PixelBuffer; |
||||
|
RowOctet<TPixel> currentRows = default; |
||||
|
|
||||
|
var pixelConverter = new YCbCrForwardConverter420<TPixel>(frame); |
||||
|
|
||||
|
for (int y = 0; y < pixels.Height; y += 16) |
||||
|
{ |
||||
|
cancellationToken.ThrowIfCancellationRequested(); |
||||
|
for (int x = 0; x < pixels.Width; x += 16) |
||||
|
{ |
||||
|
for (int i = 0; i < 2; i++) |
||||
|
{ |
||||
|
int yOff = i * 8; |
||||
|
currentRows.Update(pixelBuffer, y + yOff); |
||||
|
pixelConverter.Convert(x, y, ref currentRows, i); |
||||
|
|
||||
|
prevDCY = this.WriteBlock( |
||||
|
QuantIndex.Luminance, |
||||
|
prevDCY, |
||||
|
ref pixelConverter.YLeft, |
||||
|
ref luminanceQuantTable); |
||||
|
|
||||
|
prevDCY = this.WriteBlock( |
||||
|
QuantIndex.Luminance, |
||||
|
prevDCY, |
||||
|
ref pixelConverter.YRight, |
||||
|
ref luminanceQuantTable); |
||||
|
} |
||||
|
|
||||
|
prevDCCb = this.WriteBlock( |
||||
|
QuantIndex.Chrominance, |
||||
|
prevDCCb, |
||||
|
ref pixelConverter.Cb, |
||||
|
ref chrominanceQuantTable); |
||||
|
|
||||
|
prevDCCr = this.WriteBlock( |
||||
|
QuantIndex.Chrominance, |
||||
|
prevDCCr, |
||||
|
ref pixelConverter.Cr, |
||||
|
ref chrominanceQuantTable); |
||||
|
|
||||
|
if (this.IsStreamFlushNeeded) |
||||
|
{ |
||||
|
this.FlushToStream(); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
this.FlushRemainingBytes(); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Encodes the image with no chroma, just luminance.
|
||||
|
/// </summary>
|
||||
|
/// <typeparam name="TPixel">The pixel format.</typeparam>
|
||||
|
/// <param name="pixels">The pixel accessor providing access to the image pixels.</param>
|
||||
|
/// <param name="luminanceQuantTable">Luminance quantization table provided by the callee.</param>
|
||||
|
/// <param name="cancellationToken">The token to monitor for cancellation.</param>
|
||||
|
public void EncodeGrayscale<TPixel>(Image<TPixel> pixels, ref Block8x8F luminanceQuantTable, CancellationToken cancellationToken) |
||||
|
where TPixel : unmanaged, IPixel<TPixel> |
||||
|
{ |
||||
|
FastFloatingPointDCT.AdjustToFDCT(ref luminanceQuantTable); |
||||
|
|
||||
|
this.huffmanTables = HuffmanLut.TheHuffmanLut; |
||||
|
|
||||
|
// ReSharper disable once InconsistentNaming
|
||||
|
int prevDCY = 0; |
||||
|
|
||||
|
var pixelConverter = LuminanceForwardConverter<TPixel>.Create(); |
||||
|
ImageFrame<TPixel> frame = pixels.Frames.RootFrame; |
||||
|
Buffer2D<TPixel> pixelBuffer = frame.PixelBuffer; |
||||
|
RowOctet<TPixel> currentRows = default; |
||||
|
|
||||
|
for (int y = 0; y < pixels.Height; y += 8) |
||||
|
{ |
||||
|
cancellationToken.ThrowIfCancellationRequested(); |
||||
|
currentRows.Update(pixelBuffer, y); |
||||
|
|
||||
|
for (int x = 0; x < pixels.Width; x += 8) |
||||
|
{ |
||||
|
pixelConverter.Convert(frame, x, y, ref currentRows); |
||||
|
|
||||
|
prevDCY = this.WriteBlock( |
||||
|
QuantIndex.Luminance, |
||||
|
prevDCY, |
||||
|
ref pixelConverter.Y, |
||||
|
ref luminanceQuantTable); |
||||
|
|
||||
|
if (this.IsStreamFlushNeeded) |
||||
|
{ |
||||
|
this.FlushToStream(); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
this.FlushRemainingBytes(); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Encodes the image with no subsampling and keeps the pixel data as Rgb24.
|
||||
|
/// </summary>
|
||||
|
/// <typeparam name="TPixel">The pixel format.</typeparam>
|
||||
|
/// <param name="pixels">The pixel accessor providing access to the image pixels.</param>
|
||||
|
/// <param name="quantTable">Quantization table provided by the callee.</param>
|
||||
|
/// <param name="cancellationToken">The token to monitor for cancellation.</param>
|
||||
|
public void EncodeRgb<TPixel>(Image<TPixel> pixels, ref Block8x8F quantTable, CancellationToken cancellationToken) |
||||
|
where TPixel : unmanaged, IPixel<TPixel> |
||||
|
{ |
||||
|
FastFloatingPointDCT.AdjustToFDCT(ref quantTable); |
||||
|
|
||||
|
this.huffmanTables = HuffmanLut.TheHuffmanLut; |
||||
|
|
||||
|
// ReSharper disable once InconsistentNaming
|
||||
|
int prevDCR = 0, prevDCG = 0, prevDCB = 0; |
||||
|
|
||||
|
ImageFrame<TPixel> frame = pixels.Frames.RootFrame; |
||||
|
Buffer2D<TPixel> pixelBuffer = frame.PixelBuffer; |
||||
|
RowOctet<TPixel> currentRows = default; |
||||
|
|
||||
|
var pixelConverter = new RgbForwardConverter<TPixel>(frame); |
||||
|
|
||||
|
for (int y = 0; y < pixels.Height; y += 8) |
||||
|
{ |
||||
|
cancellationToken.ThrowIfCancellationRequested(); |
||||
|
currentRows.Update(pixelBuffer, y); |
||||
|
|
||||
|
for (int x = 0; x < pixels.Width; x += 8) |
||||
|
{ |
||||
|
pixelConverter.Convert(x, y, ref currentRows); |
||||
|
|
||||
|
prevDCR = this.WriteBlock( |
||||
|
QuantIndex.Luminance, |
||||
|
prevDCR, |
||||
|
ref pixelConverter.R, |
||||
|
ref quantTable); |
||||
|
|
||||
|
prevDCG = this.WriteBlock( |
||||
|
QuantIndex.Luminance, |
||||
|
prevDCG, |
||||
|
ref pixelConverter.G, |
||||
|
ref quantTable); |
||||
|
|
||||
|
prevDCB = this.WriteBlock( |
||||
|
QuantIndex.Luminance, |
||||
|
prevDCB, |
||||
|
ref pixelConverter.B, |
||||
|
ref quantTable); |
||||
|
|
||||
|
if (this.IsStreamFlushNeeded) |
||||
|
{ |
||||
|
this.FlushToStream(); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
this.FlushRemainingBytes(); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Writes a block of pixel data using the given quantization table,
|
||||
|
/// returning the post-quantized DC value of the DCT-transformed block.
|
||||
|
/// The block is in natural (not zig-zag) order.
|
||||
|
/// </summary>
|
||||
|
/// <param name="index">The quantization table index.</param>
|
||||
|
/// <param name="prevDC">The previous DC value.</param>
|
||||
|
/// <param name="block">Source block.</param>
|
||||
|
/// <param name="quant">Quantization table.</param>
|
||||
|
/// <returns>The <see cref="int"/>.</returns>
|
||||
|
private int WriteBlock( |
||||
|
QuantIndex index, |
||||
|
int prevDC, |
||||
|
ref Block8x8F block, |
||||
|
ref Block8x8F quant) |
||||
|
{ |
||||
|
ref Block8x8 spectralBlock = ref this.tempBlock; |
||||
|
|
||||
|
// Shifting level from 0..255 to -128..127
|
||||
|
block.AddInPlace(-128f); |
||||
|
|
||||
|
// Discrete cosine transform
|
||||
|
FastFloatingPointDCT.TransformFDCT(ref block); |
||||
|
|
||||
|
// Quantization
|
||||
|
Block8x8F.Quantize(ref block, ref spectralBlock, ref quant); |
||||
|
|
||||
|
// Emit the DC delta.
|
||||
|
int dc = spectralBlock[0]; |
||||
|
this.EmitHuffRLE(this.huffmanTables[2 * (int)index].Values, 0, dc - prevDC); |
||||
|
|
||||
|
// Emit the AC components.
|
||||
|
int[] acHuffTable = this.huffmanTables[(2 * (int)index) + 1].Values; |
||||
|
|
||||
|
nint lastValuableIndex = spectralBlock.GetLastNonZeroIndex(); |
||||
|
|
||||
|
int runLength = 0; |
||||
|
ref short blockRef = ref Unsafe.As<Block8x8, short>(ref spectralBlock); |
||||
|
for (nint zig = 1; zig <= lastValuableIndex; zig++) |
||||
|
{ |
||||
|
const int zeroRun1 = 1 << 4; |
||||
|
const int zeroRun16 = 16 << 4; |
||||
|
|
||||
|
int ac = Unsafe.Add(ref blockRef, zig); |
||||
|
if (ac == 0) |
||||
|
{ |
||||
|
runLength += zeroRun1; |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
while (runLength >= zeroRun16) |
||||
|
{ |
||||
|
this.EmitHuff(acHuffTable, 0xf0); |
||||
|
runLength -= zeroRun16; |
||||
|
} |
||||
|
|
||||
|
this.EmitHuffRLE(acHuffTable, runLength, ac); |
||||
|
runLength = 0; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// if mcu block contains trailing zeros - we must write end of block (EOB) value indicating that current block is over
|
||||
|
// this can be done for any number of trailing zeros, even when all 63 ac values are zero
|
||||
|
// (Block8x8F.Size - 1) == 63 - last index of the mcu elements
|
||||
|
if (lastValuableIndex != Block8x8F.Size - 1) |
||||
|
{ |
||||
|
this.EmitHuff(acHuffTable, 0x00); |
||||
|
} |
||||
|
|
||||
|
return dc; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Emits the most significant count of bits to the buffer.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// <para>
|
||||
|
/// Supports up to 32 count of bits but, generally speaking, jpeg
|
||||
|
/// standard assures that there won't be more than 16 bits per single
|
||||
|
/// value.
|
||||
|
/// </para>
|
||||
|
/// <para>
|
||||
|
/// Emitting algorithm uses 3 intermediate buffers for caching before
|
||||
|
/// writing to the stream:
|
||||
|
/// <list type="number">
|
||||
|
/// <item>
|
||||
|
/// <term>uint32</term>
|
||||
|
/// <description>
|
||||
|
/// Bit buffer. Encoded spectral values can occupy up to 16 bits, bits
|
||||
|
/// are assembled to whole bytes via this intermediate buffer.
|
||||
|
/// </description>
|
||||
|
/// </item>
|
||||
|
/// <item>
|
||||
|
/// <term>uint32[]</term>
|
||||
|
/// <description>
|
||||
|
/// Assembled bytes from uint32 buffer are saved into this buffer.
|
||||
|
/// uint32 buffer values are saved using indices from the last to the first.
|
||||
|
/// As bytes are saved to the memory as 4-byte packages endianness matters:
|
||||
|
/// Jpeg stream is big-endian, indexing buffer bytes from the last index to the
|
||||
|
/// first eliminates all operations to extract separate bytes. This only works for
|
||||
|
/// little-endian machines (there are no known examples of big-endian users atm).
|
||||
|
/// For big-endians this approach is slower due to the separate byte extraction.
|
||||
|
/// </description>
|
||||
|
/// </item>
|
||||
|
/// <item>
|
||||
|
/// <term>byte[]</term>
|
||||
|
/// <description>
|
||||
|
/// Byte buffer used only during <see cref="FlushToStream(int)"/> method.
|
||||
|
/// </description>
|
||||
|
/// </item>
|
||||
|
/// </list>
|
||||
|
/// </para>
|
||||
|
/// </remarks>
|
||||
|
/// <param name="bits">Bits to emit, must be shifted to the left.</param>
|
||||
|
/// <param name="count">Bits count stored in the bits parameter.</param>
|
||||
|
[MethodImpl(InliningOptions.ShortMethod)] |
||||
|
private void Emit(uint bits, int count) |
||||
|
{ |
||||
|
this.accumulatedBits |= bits >> this.bitCount; |
||||
|
|
||||
|
count += this.bitCount; |
||||
|
|
||||
|
if (count >= 32) |
||||
|
{ |
||||
|
this.emitBuffer[--this.emitWriteIndex] = this.accumulatedBits; |
||||
|
this.accumulatedBits = bits << (32 - this.bitCount); |
||||
|
|
||||
|
count -= 32; |
||||
|
} |
||||
|
|
||||
|
this.bitCount = count; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Emits the given value with the given Huffman table.
|
||||
|
/// </summary>
|
||||
|
/// <param name="table">Huffman table.</param>
|
||||
|
/// <param name="value">Value to encode.</param>
|
||||
|
[MethodImpl(InliningOptions.ShortMethod)] |
||||
|
private void EmitHuff(int[] table, int value) |
||||
|
{ |
||||
|
int x = table[value]; |
||||
|
this.Emit((uint)x & 0xffff_ff00u, x & 0xff); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Emits given value via huffman rle encoding.
|
||||
|
/// </summary>
|
||||
|
/// <param name="table">Huffman table.</param>
|
||||
|
/// <param name="runLength">The number of preceding zeroes, preshifted by 4 to the left.</param>
|
||||
|
/// <param name="value">Value to encode.</param>
|
||||
|
[MethodImpl(InliningOptions.ShortMethod)] |
||||
|
private void EmitHuffRLE(int[] table, int runLength, int value) |
||||
|
{ |
||||
|
DebugGuard.IsTrue((runLength & 0xf) == 0, $"{nameof(runLength)} parameter must be shifted to the left by 4 bits"); |
||||
|
|
||||
|
int a = value; |
||||
|
int b = value; |
||||
|
if (a < 0) |
||||
|
{ |
||||
|
a = -value; |
||||
|
b = value - 1; |
||||
|
} |
||||
|
|
||||
|
int valueLen = GetHuffmanEncodingLength((uint)a); |
||||
|
|
||||
|
// Huffman prefix code
|
||||
|
int huffPackage = table[runLength | valueLen]; |
||||
|
int prefixLen = huffPackage & 0xff; |
||||
|
uint prefix = (uint)huffPackage & 0xffff_0000u; |
||||
|
|
||||
|
// Actual encoded value
|
||||
|
uint encodedValue = (uint)b << (32 - valueLen); |
||||
|
|
||||
|
// Doing two binary shifts to get rid of leading 1's in negative value case
|
||||
|
this.Emit(prefix | (encodedValue >> prefixLen), prefixLen + valueLen); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Calculates how many minimum bits needed to store given value for Huffman jpeg encoding.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// This is an internal operation supposed to be used only in <see cref="HuffmanScanEncoder"/> class for jpeg encoding.
|
||||
|
/// </remarks>
|
||||
|
/// <param name="value">The value.</param>
|
||||
|
[MethodImpl(InliningOptions.ShortMethod)] |
||||
|
internal static int GetHuffmanEncodingLength(uint value) |
||||
|
{ |
||||
|
DebugGuard.IsTrue(value <= (1 << 16), "Huffman encoder is supposed to encode a value of 16bit size max"); |
||||
|
#if SUPPORTS_BITOPERATIONS
|
||||
|
// This should have been implemented as (BitOperations.Log2(value) + 1) as in non-intrinsic implementation
|
||||
|
// But internal log2 is implemented like this: (31 - (int)Lzcnt.LeadingZeroCount(value))
|
||||
|
|
||||
|
// BitOperations.Log2 implementation also checks if input value is zero for the convention 0->0
|
||||
|
// Lzcnt would return 32 for input value of 0 - no need to check that with branching
|
||||
|
// Fallback code if Lzcnt is not supported still use if-check
|
||||
|
// But most modern CPUs support this instruction so this should not be a problem
|
||||
|
return 32 - BitOperations.LeadingZeroCount(value); |
||||
|
#else
|
||||
|
// Ideally:
|
||||
|
// if 0 - return 0 in this case
|
||||
|
// else - return log2(value) + 1
|
||||
|
//
|
||||
|
// Hack based on input value constraint:
|
||||
|
// We know that input values are guaranteed to be maximum 16 bit large for huffman encoding
|
||||
|
// We can safely shift input value for one bit -> log2(value << 1)
|
||||
|
// Because of the 16 bit value constraint it won't overflow
|
||||
|
// With that input value change we no longer need to add 1 before returning
|
||||
|
// And this eliminates need to check if input value is zero - it is a standard convention which Log2SoftwareFallback adheres to
|
||||
|
return Numerics.Log2(value << 1); |
||||
|
#endif
|
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// General method for flushing cached spectral data bytes to
|
||||
|
/// the ouput stream respecting stuff bytes.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// Bytes cached via <see cref="Emit"/> are stored in 4-bytes blocks
|
||||
|
/// which makes this method endianness dependent.
|
||||
|
/// </remarks>
|
||||
|
[MethodImpl(InliningOptions.ShortMethod)] |
||||
|
private void FlushToStream(int endIndex) |
||||
|
{ |
||||
|
Span<byte> emitBytes = MemoryMarshal.AsBytes(this.emitBuffer.AsSpan()); |
||||
|
|
||||
|
int writeIdx = 0; |
||||
|
int startIndex = emitBytes.Length - 1; |
||||
|
|
||||
|
// Some platforms may fail to eliminate this if-else branching
|
||||
|
// Even if it happens - buffer is flushed in big packs,
|
||||
|
// branching overhead shouldn't be noticeable
|
||||
|
if (BitConverter.IsLittleEndian) |
||||
|
{ |
||||
|
// For little endian case bytes are ordered and can be
|
||||
|
// safely written to the stream with stuff bytes
|
||||
|
// First byte is cached on the most significant index
|
||||
|
// so we are going from the end of the array to its beginning:
|
||||
|
// ... [ double word #1 ] [ double word #0 ]
|
||||
|
// ... [idx3|idx2|idx1|idx0] [idx3|idx2|idx1|idx0]
|
||||
|
for (int i = startIndex; i >= endIndex; i--) |
||||
|
{ |
||||
|
byte value = emitBytes[i]; |
||||
|
this.streamWriteBuffer[writeIdx++] = value; |
||||
|
|
||||
|
// Inserting stuff byte
|
||||
|
if (value == 0xff) |
||||
|
{ |
||||
|
this.streamWriteBuffer[writeIdx++] = 0x00; |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
// For big endian case bytes are ordered in 4-byte packs
|
||||
|
// which are ordered like bytes in the little endian case by in 4-byte packs:
|
||||
|
// ... [ double word #1 ] [ double word #0 ]
|
||||
|
// ... [idx0|idx1|idx2|idx3] [idx0|idx1|idx2|idx3]
|
||||
|
// So we must write each 4-bytes in 'natural order'
|
||||
|
for (int i = startIndex; i >= endIndex; i -= 4) |
||||
|
{ |
||||
|
// This loop is caused by the nature of underlying byte buffer
|
||||
|
// implementation and indeed causes performace by somewhat 5%
|
||||
|
// compared to little endian scenario
|
||||
|
// Even with this performance drop this cached buffer implementation
|
||||
|
// is faster than individually writing bytes using binary shifts and binary and(s)
|
||||
|
for (int j = i - 3; j <= i; j++) |
||||
|
{ |
||||
|
byte value = emitBytes[j]; |
||||
|
this.streamWriteBuffer[writeIdx++] = value; |
||||
|
|
||||
|
// Inserting stuff byte
|
||||
|
if (value == 0xff) |
||||
|
{ |
||||
|
this.streamWriteBuffer[writeIdx++] = 0x00; |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
this.target.Write(this.streamWriteBuffer, 0, writeIdx); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Flushes spectral data bytes after encoding all channel blocks
|
||||
|
/// in a single jpeg macroblock using <see cref="WriteBlock"/>.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// This must be called only if <see cref="IsStreamFlushNeeded"/> is true
|
||||
|
/// only during the macroblocks encoding routine.
|
||||
|
/// </remarks>
|
||||
|
private void FlushToStream() |
||||
|
{ |
||||
|
this.FlushToStream(this.emitWriteIndex * 4); |
||||
|
this.emitWriteIndex = this.emitBuffer.Length; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Flushes final cached bits to the stream padding 1's to
|
||||
|
/// complement full bytes.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// This must be called only once at the end of the encoding routine.
|
||||
|
/// <see cref="IsStreamFlushNeeded"/> check is not needed.
|
||||
|
/// </remarks>
|
||||
|
[MethodImpl(InliningOptions.ShortMethod)] |
||||
|
private void FlushRemainingBytes() |
||||
|
{ |
||||
|
// Padding all 4 bytes with 1's while not corrupting initial bits stored in accumulatedBits
|
||||
|
// And writing only valuable count of bytes count we want to write to the output stream
|
||||
|
int valuableBytesCount = (int)Numerics.DivideCeil((uint)this.bitCount, 8); |
||||
|
uint packedBytes = this.accumulatedBits | (uint.MaxValue >> this.bitCount); |
||||
|
this.emitBuffer[--this.emitWriteIndex] = packedBytes; |
||||
|
|
||||
|
// Flush cached bytes to the output stream with padding bits
|
||||
|
this.FlushToStream((this.emitWriteIndex * 4) - 4 + valuableBytesCount); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,114 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using System; |
||||
|
using System.Runtime.CompilerServices; |
||||
|
using System.Runtime.InteropServices; |
||||
|
using SixLabors.ImageSharp.Advanced; |
||||
|
using SixLabors.ImageSharp.PixelFormats; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components.Encoder |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// On-stack worker struct to convert TPixel -> Rgb24 of 8x8 pixel blocks.
|
||||
|
/// </summary>
|
||||
|
/// <typeparam name="TPixel">The pixel type to work on.</typeparam>
|
||||
|
internal ref struct RgbForwardConverter<TPixel> |
||||
|
where TPixel : unmanaged, IPixel<TPixel> |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Number of pixels processed per single <see cref="Convert(int, int, ref RowOctet{TPixel})"/> call
|
||||
|
/// </summary>
|
||||
|
private const int PixelsPerSample = 8 * 8; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Total byte size of processed pixels converted from TPixel to <see cref="Rgb24"/>
|
||||
|
/// </summary>
|
||||
|
private const int RgbSpanByteSize = PixelsPerSample * 3; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// <see cref="Size"/> of sampling area from given frame pixel buffer.
|
||||
|
/// </summary>
|
||||
|
private static readonly Size SampleSize = new Size(8, 8); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The Red component.
|
||||
|
/// </summary>
|
||||
|
public Block8x8F R; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The Green component.
|
||||
|
/// </summary>
|
||||
|
public Block8x8F G; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The Blue component.
|
||||
|
/// </summary>
|
||||
|
public Block8x8F B; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Temporal 64-byte span to hold unconverted TPixel data.
|
||||
|
/// </summary>
|
||||
|
private readonly Span<TPixel> pixelSpan; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Temporal 64-byte span to hold converted Rgb24 data.
|
||||
|
/// </summary>
|
||||
|
private readonly Span<Rgb24> rgbSpan; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Sampled pixel buffer size.
|
||||
|
/// </summary>
|
||||
|
private readonly Size samplingAreaSize; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// <see cref="Configuration"/> for internal operations.
|
||||
|
/// </summary>
|
||||
|
private readonly Configuration config; |
||||
|
|
||||
|
public RgbForwardConverter(ImageFrame<TPixel> frame) |
||||
|
{ |
||||
|
this.R = default; |
||||
|
this.G = default; |
||||
|
this.B = default; |
||||
|
|
||||
|
// temporal pixel buffers
|
||||
|
this.pixelSpan = new TPixel[PixelsPerSample].AsSpan(); |
||||
|
this.rgbSpan = MemoryMarshal.Cast<byte, Rgb24>(new byte[RgbSpanByteSize + RgbToYCbCrConverterVectorized.AvxCompatibilityPadding].AsSpan()); |
||||
|
|
||||
|
// frame data
|
||||
|
this.samplingAreaSize = new Size(frame.Width, frame.Height); |
||||
|
this.config = frame.GetConfiguration(); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Converts a 8x8 image area inside 'pixels' at position (x, y) to Rgb24.
|
||||
|
/// </summary>
|
||||
|
public void Convert(int x, int y, ref RowOctet<TPixel> currentRows) |
||||
|
{ |
||||
|
YCbCrForwardConverter<TPixel>.LoadAndStretchEdges(currentRows, this.pixelSpan, new Point(x, y), SampleSize, this.samplingAreaSize); |
||||
|
|
||||
|
PixelOperations<TPixel>.Instance.ToRgb24(this.config, this.pixelSpan, this.rgbSpan); |
||||
|
|
||||
|
ref Block8x8F redBlock = ref this.R; |
||||
|
ref Block8x8F greenBlock = ref this.G; |
||||
|
ref Block8x8F blueBlock = ref this.B; |
||||
|
|
||||
|
CopyToBlock(this.rgbSpan, ref redBlock, ref greenBlock, ref blueBlock); |
||||
|
} |
||||
|
|
||||
|
private static void CopyToBlock(Span<Rgb24> rgbSpan, ref Block8x8F redBlock, ref Block8x8F greenBlock, ref Block8x8F blueBlock) |
||||
|
{ |
||||
|
ref Rgb24 rgbStart = ref MemoryMarshal.GetReference(rgbSpan); |
||||
|
|
||||
|
for (int i = 0; i < Block8x8F.Size; i++) |
||||
|
{ |
||||
|
Rgb24 c = Unsafe.Add(ref rgbStart, (nint)(uint)i); |
||||
|
|
||||
|
redBlock[i] = c.R; |
||||
|
greenBlock[i] = c.G; |
||||
|
blueBlock[i] = c.B; |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,121 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using System; |
||||
|
using System.Runtime.InteropServices; |
||||
|
using SixLabors.ImageSharp.Advanced; |
||||
|
using SixLabors.ImageSharp.PixelFormats; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components.Encoder |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// On-stack worker struct to efficiently encapsulate the TPixel -> Rgb24 -> YCbCr conversion chain of 8x8 pixel blocks.
|
||||
|
/// </summary>
|
||||
|
/// <typeparam name="TPixel">The pixel type to work on</typeparam>
|
||||
|
internal ref struct YCbCrForwardConverter420<TPixel> |
||||
|
where TPixel : unmanaged, IPixel<TPixel> |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Number of pixels processed per single <see cref="Convert(int, int, ref RowOctet{TPixel}, int)"/> call
|
||||
|
/// </summary>
|
||||
|
private const int PixelsPerSample = 16 * 8; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Total byte size of processed pixels converted from TPixel to <see cref="Rgb24"/>
|
||||
|
/// </summary>
|
||||
|
private const int RgbSpanByteSize = PixelsPerSample * 3; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// <see cref="Size"/> of sampling area from given frame pixel buffer
|
||||
|
/// </summary>
|
||||
|
private static readonly Size SampleSize = new Size(16, 8); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The left Y component
|
||||
|
/// </summary>
|
||||
|
public Block8x8F YLeft; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The left Y component
|
||||
|
/// </summary>
|
||||
|
public Block8x8F YRight; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The Cb component
|
||||
|
/// </summary>
|
||||
|
public Block8x8F Cb; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The Cr component
|
||||
|
/// </summary>
|
||||
|
public Block8x8F Cr; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The color conversion tables
|
||||
|
/// </summary>
|
||||
|
private RgbToYCbCrConverterLut colorTables; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Temporal 16x8 block to hold TPixel data
|
||||
|
/// </summary>
|
||||
|
private readonly Span<TPixel> pixelSpan; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Temporal RGB block
|
||||
|
/// </summary>
|
||||
|
private readonly Span<Rgb24> rgbSpan; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Sampled pixel buffer size
|
||||
|
/// </summary>
|
||||
|
private readonly Size samplingAreaSize; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// <see cref="Configuration"/> for internal operations
|
||||
|
/// </summary>
|
||||
|
private readonly Configuration config; |
||||
|
|
||||
|
public YCbCrForwardConverter420(ImageFrame<TPixel> frame) |
||||
|
{ |
||||
|
// matrices would be filled during convert calls
|
||||
|
this.YLeft = default; |
||||
|
this.YRight = default; |
||||
|
this.Cb = default; |
||||
|
this.Cr = default; |
||||
|
|
||||
|
// temporal pixel buffers
|
||||
|
this.pixelSpan = new TPixel[PixelsPerSample].AsSpan(); |
||||
|
this.rgbSpan = MemoryMarshal.Cast<byte, Rgb24>(new byte[RgbSpanByteSize + RgbToYCbCrConverterVectorized.AvxCompatibilityPadding].AsSpan()); |
||||
|
|
||||
|
// frame data
|
||||
|
this.samplingAreaSize = new Size(frame.Width, frame.Height); |
||||
|
this.config = frame.GetConfiguration(); |
||||
|
|
||||
|
// conversion vector fallback data
|
||||
|
if (!RgbToYCbCrConverterVectorized.IsSupported) |
||||
|
{ |
||||
|
this.colorTables = RgbToYCbCrConverterLut.Create(); |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
this.colorTables = default; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
public void Convert(int x, int y, ref RowOctet<TPixel> currentRows, int idx) |
||||
|
{ |
||||
|
YCbCrForwardConverter<TPixel>.LoadAndStretchEdges(currentRows, this.pixelSpan, new Point(x, y), SampleSize, this.samplingAreaSize); |
||||
|
|
||||
|
PixelOperations<TPixel>.Instance.ToRgb24(this.config, this.pixelSpan, this.rgbSpan); |
||||
|
|
||||
|
if (RgbToYCbCrConverterVectorized.IsSupported) |
||||
|
{ |
||||
|
RgbToYCbCrConverterVectorized.Convert420(this.rgbSpan, ref this.YLeft, ref this.YRight, ref this.Cb, ref this.Cr, idx); |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
this.colorTables.Convert420(this.rgbSpan, ref this.YLeft, ref this.YRight, ref this.Cb, ref this.Cr, idx); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,122 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using System; |
||||
|
using System.Runtime.InteropServices; |
||||
|
using SixLabors.ImageSharp.Advanced; |
||||
|
using SixLabors.ImageSharp.PixelFormats; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components.Encoder |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// On-stack worker struct to efficiently encapsulate the TPixel -> Rgb24 -> YCbCr conversion chain of 8x8 pixel blocks.
|
||||
|
/// </summary>
|
||||
|
/// <typeparam name="TPixel">The pixel type to work on</typeparam>
|
||||
|
internal ref struct YCbCrForwardConverter444<TPixel> |
||||
|
where TPixel : unmanaged, IPixel<TPixel> |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Number of pixels processed per single <see cref="Convert(int, int, ref RowOctet{TPixel})"/> call
|
||||
|
/// </summary>
|
||||
|
private const int PixelsPerSample = 8 * 8; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Total byte size of processed pixels converted from TPixel to <see cref="Rgb24"/>
|
||||
|
/// </summary>
|
||||
|
private const int RgbSpanByteSize = PixelsPerSample * 3; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// <see cref="Size"/> of sampling area from given frame pixel buffer
|
||||
|
/// </summary>
|
||||
|
private static readonly Size SampleSize = new Size(8, 8); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The Y component
|
||||
|
/// </summary>
|
||||
|
public Block8x8F Y; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The Cb component
|
||||
|
/// </summary>
|
||||
|
public Block8x8F Cb; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The Cr component
|
||||
|
/// </summary>
|
||||
|
public Block8x8F Cr; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The color conversion tables
|
||||
|
/// </summary>
|
||||
|
private RgbToYCbCrConverterLut colorTables; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Temporal 64-byte span to hold unconverted TPixel data
|
||||
|
/// </summary>
|
||||
|
private readonly Span<TPixel> pixelSpan; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Temporal 64-byte span to hold converted Rgb24 data
|
||||
|
/// </summary>
|
||||
|
private readonly Span<Rgb24> rgbSpan; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Sampled pixel buffer size
|
||||
|
/// </summary>
|
||||
|
private readonly Size samplingAreaSize; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// <see cref="Configuration"/> for internal operations
|
||||
|
/// </summary>
|
||||
|
private readonly Configuration config; |
||||
|
|
||||
|
public YCbCrForwardConverter444(ImageFrame<TPixel> frame) |
||||
|
{ |
||||
|
// matrices would be filled during convert calls
|
||||
|
this.Y = default; |
||||
|
this.Cb = default; |
||||
|
this.Cr = default; |
||||
|
|
||||
|
// temporal pixel buffers
|
||||
|
this.pixelSpan = new TPixel[PixelsPerSample].AsSpan(); |
||||
|
this.rgbSpan = MemoryMarshal.Cast<byte, Rgb24>(new byte[RgbSpanByteSize + RgbToYCbCrConverterVectorized.AvxCompatibilityPadding].AsSpan()); |
||||
|
|
||||
|
// frame data
|
||||
|
this.samplingAreaSize = new Size(frame.Width, frame.Height); |
||||
|
this.config = frame.GetConfiguration(); |
||||
|
|
||||
|
// conversion vector fallback data
|
||||
|
if (!RgbToYCbCrConverterVectorized.IsSupported) |
||||
|
{ |
||||
|
this.colorTables = RgbToYCbCrConverterLut.Create(); |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
this.colorTables = default; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Converts a 8x8 image area inside 'pixels' at position (x,y) placing the result members of the structure (<see cref="Y"/>, <see cref="Cb"/>, <see cref="Cr"/>)
|
||||
|
/// </summary>
|
||||
|
public void Convert(int x, int y, ref RowOctet<TPixel> currentRows) |
||||
|
{ |
||||
|
YCbCrForwardConverter<TPixel>.LoadAndStretchEdges(currentRows, this.pixelSpan, new Point(x, y), SampleSize, this.samplingAreaSize); |
||||
|
|
||||
|
PixelOperations<TPixel>.Instance.ToRgb24(this.config, this.pixelSpan, this.rgbSpan); |
||||
|
|
||||
|
ref Block8x8F yBlock = ref this.Y; |
||||
|
ref Block8x8F cbBlock = ref this.Cb; |
||||
|
ref Block8x8F crBlock = ref this.Cr; |
||||
|
|
||||
|
if (RgbToYCbCrConverterVectorized.IsSupported) |
||||
|
{ |
||||
|
RgbToYCbCrConverterVectorized.Convert444(this.rgbSpan, ref yBlock, ref cbBlock, ref crBlock); |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
this.colorTables.Convert444(this.rgbSpan, ref yBlock, ref cbBlock, ref crBlock); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,161 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
#if SUPPORTS_RUNTIME_INTRINSICS
|
||||
|
using System.Diagnostics; |
||||
|
using System.Numerics; |
||||
|
using System.Runtime.CompilerServices; |
||||
|
using System.Runtime.Intrinsics; |
||||
|
using System.Runtime.Intrinsics.X86; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components |
||||
|
{ |
||||
|
internal static partial class FastFloatingPointDCT |
||||
|
{ |
||||
|
#pragma warning disable SA1310, SA1311, IDE1006 // naming rules violation warnings
|
||||
|
private static readonly Vector256<float> mm256_F_0_7071 = Vector256.Create(0.707106781f); |
||||
|
private static readonly Vector256<float> mm256_F_0_3826 = Vector256.Create(0.382683433f); |
||||
|
private static readonly Vector256<float> mm256_F_0_5411 = Vector256.Create(0.541196100f); |
||||
|
private static readonly Vector256<float> mm256_F_1_3065 = Vector256.Create(1.306562965f); |
||||
|
|
||||
|
private static readonly Vector256<float> mm256_F_1_1758 = Vector256.Create(1.175876f); |
||||
|
private static readonly Vector256<float> mm256_F_n1_9615 = Vector256.Create(-1.961570560f); |
||||
|
private static readonly Vector256<float> mm256_F_n0_3901 = Vector256.Create(-0.390180644f); |
||||
|
private static readonly Vector256<float> mm256_F_n0_8999 = Vector256.Create(-0.899976223f); |
||||
|
private static readonly Vector256<float> mm256_F_n2_5629 = Vector256.Create(-2.562915447f); |
||||
|
private static readonly Vector256<float> mm256_F_0_2986 = Vector256.Create(0.298631336f); |
||||
|
private static readonly Vector256<float> mm256_F_2_0531 = Vector256.Create(2.053119869f); |
||||
|
private static readonly Vector256<float> mm256_F_3_0727 = Vector256.Create(3.072711026f); |
||||
|
private static readonly Vector256<float> mm256_F_1_5013 = Vector256.Create(1.501321110f); |
||||
|
private static readonly Vector256<float> mm256_F_n1_8477 = Vector256.Create(-1.847759065f); |
||||
|
private static readonly Vector256<float> mm256_F_0_7653 = Vector256.Create(0.765366865f); |
||||
|
#pragma warning restore SA1310, SA1311, IDE1006
|
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Apply floating point FDCT inplace using simd operations.
|
||||
|
/// </summary>
|
||||
|
/// <param name="block">Input matrix.</param>
|
||||
|
private static void ForwardTransform_Avx(ref Block8x8F block) |
||||
|
{ |
||||
|
DebugGuard.IsTrue(Avx.IsSupported, "Avx support is required to execute this operation."); |
||||
|
|
||||
|
// First pass - process rows
|
||||
|
block.TransposeInplace(); |
||||
|
FDCT8x8_Avx(ref block); |
||||
|
|
||||
|
// Second pass - process columns
|
||||
|
block.TransposeInplace(); |
||||
|
FDCT8x8_Avx(ref block); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Apply 1D floating point FDCT inplace using AVX operations on 8x8 matrix.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// Requires Avx support.
|
||||
|
/// </remarks>
|
||||
|
/// <param name="block">Input matrix.</param>
|
||||
|
public static void FDCT8x8_Avx(ref Block8x8F block) |
||||
|
{ |
||||
|
DebugGuard.IsTrue(Avx.IsSupported, "Avx support is required to execute this operation."); |
||||
|
|
||||
|
Vector256<float> tmp0 = Avx.Add(block.V0, block.V7); |
||||
|
Vector256<float> tmp7 = Avx.Subtract(block.V0, block.V7); |
||||
|
Vector256<float> tmp1 = Avx.Add(block.V1, block.V6); |
||||
|
Vector256<float> tmp6 = Avx.Subtract(block.V1, block.V6); |
||||
|
Vector256<float> tmp2 = Avx.Add(block.V2, block.V5); |
||||
|
Vector256<float> tmp5 = Avx.Subtract(block.V2, block.V5); |
||||
|
Vector256<float> tmp3 = Avx.Add(block.V3, block.V4); |
||||
|
Vector256<float> tmp4 = Avx.Subtract(block.V3, block.V4); |
||||
|
|
||||
|
// Even part
|
||||
|
Vector256<float> tmp10 = Avx.Add(tmp0, tmp3); |
||||
|
Vector256<float> tmp13 = Avx.Subtract(tmp0, tmp3); |
||||
|
Vector256<float> tmp11 = Avx.Add(tmp1, tmp2); |
||||
|
Vector256<float> tmp12 = Avx.Subtract(tmp1, tmp2); |
||||
|
|
||||
|
block.V0 = Avx.Add(tmp10, tmp11); |
||||
|
block.V4 = Avx.Subtract(tmp10, tmp11); |
||||
|
|
||||
|
Vector256<float> z1 = Avx.Multiply(Avx.Add(tmp12, tmp13), mm256_F_0_7071); |
||||
|
block.V2 = Avx.Add(tmp13, z1); |
||||
|
block.V6 = Avx.Subtract(tmp13, z1); |
||||
|
|
||||
|
// Odd part
|
||||
|
tmp10 = Avx.Add(tmp4, tmp5); |
||||
|
tmp11 = Avx.Add(tmp5, tmp6); |
||||
|
tmp12 = Avx.Add(tmp6, tmp7); |
||||
|
|
||||
|
Vector256<float> z5 = Avx.Multiply(Avx.Subtract(tmp10, tmp12), mm256_F_0_3826); |
||||
|
Vector256<float> z2 = SimdUtils.HwIntrinsics.MultiplyAdd(z5, mm256_F_0_5411, tmp10); |
||||
|
Vector256<float> z4 = SimdUtils.HwIntrinsics.MultiplyAdd(z5, mm256_F_1_3065, tmp12); |
||||
|
Vector256<float> z3 = Avx.Multiply(tmp11, mm256_F_0_7071); |
||||
|
|
||||
|
Vector256<float> z11 = Avx.Add(tmp7, z3); |
||||
|
Vector256<float> z13 = Avx.Subtract(tmp7, z3); |
||||
|
|
||||
|
block.V5 = Avx.Add(z13, z2); |
||||
|
block.V3 = Avx.Subtract(z13, z2); |
||||
|
block.V1 = Avx.Add(z11, z4); |
||||
|
block.V7 = Avx.Subtract(z11, z4); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Combined operation of <see cref="IDCT8x4_LeftPart(ref Block8x8F, ref Block8x8F)"/> and <see cref="IDCT8x4_RightPart(ref Block8x8F, ref Block8x8F)"/>
|
||||
|
/// using AVX commands.
|
||||
|
/// </summary>
|
||||
|
/// <param name="s">Source</param>
|
||||
|
/// <param name="d">Destination</param>
|
||||
|
public static void IDCT8x8_Avx(ref Block8x8F s, ref Block8x8F d) |
||||
|
{ |
||||
|
Debug.Assert(Avx.IsSupported, "AVX is required to execute this method"); |
||||
|
|
||||
|
Vector256<float> my1 = s.V1; |
||||
|
Vector256<float> my7 = s.V7; |
||||
|
Vector256<float> mz0 = Avx.Add(my1, my7); |
||||
|
|
||||
|
Vector256<float> my3 = s.V3; |
||||
|
Vector256<float> mz2 = Avx.Add(my3, my7); |
||||
|
Vector256<float> my5 = s.V5; |
||||
|
Vector256<float> mz1 = Avx.Add(my3, my5); |
||||
|
Vector256<float> mz3 = Avx.Add(my1, my5); |
||||
|
|
||||
|
Vector256<float> mz4 = Avx.Multiply(Avx.Add(mz0, mz1), mm256_F_1_1758); |
||||
|
|
||||
|
mz2 = SimdUtils.HwIntrinsics.MultiplyAdd(mz4, mz2, mm256_F_n1_9615); |
||||
|
mz3 = SimdUtils.HwIntrinsics.MultiplyAdd(mz4, mz3, mm256_F_n0_3901); |
||||
|
mz0 = Avx.Multiply(mz0, mm256_F_n0_8999); |
||||
|
mz1 = Avx.Multiply(mz1, mm256_F_n2_5629); |
||||
|
|
||||
|
Vector256<float> mb3 = Avx.Add(SimdUtils.HwIntrinsics.MultiplyAdd(mz0, my7, mm256_F_0_2986), mz2); |
||||
|
Vector256<float> mb2 = Avx.Add(SimdUtils.HwIntrinsics.MultiplyAdd(mz1, my5, mm256_F_2_0531), mz3); |
||||
|
Vector256<float> mb1 = Avx.Add(SimdUtils.HwIntrinsics.MultiplyAdd(mz1, my3, mm256_F_3_0727), mz2); |
||||
|
Vector256<float> mb0 = Avx.Add(SimdUtils.HwIntrinsics.MultiplyAdd(mz0, my1, mm256_F_1_5013), mz3); |
||||
|
|
||||
|
Vector256<float> my2 = s.V2; |
||||
|
Vector256<float> my6 = s.V6; |
||||
|
mz4 = Avx.Multiply(Avx.Add(my2, my6), mm256_F_0_5411); |
||||
|
Vector256<float> my0 = s.V0; |
||||
|
Vector256<float> my4 = s.V4; |
||||
|
mz0 = Avx.Add(my0, my4); |
||||
|
mz1 = Avx.Subtract(my0, my4); |
||||
|
mz2 = SimdUtils.HwIntrinsics.MultiplyAdd(mz4, my6, mm256_F_n1_8477); |
||||
|
mz3 = SimdUtils.HwIntrinsics.MultiplyAdd(mz4, my2, mm256_F_0_7653); |
||||
|
|
||||
|
my0 = Avx.Add(mz0, mz3); |
||||
|
my3 = Avx.Subtract(mz0, mz3); |
||||
|
my1 = Avx.Add(mz1, mz2); |
||||
|
my2 = Avx.Subtract(mz1, mz2); |
||||
|
|
||||
|
d.V0 = Avx.Add(my0, mb0); |
||||
|
d.V7 = Avx.Subtract(my0, mb0); |
||||
|
d.V1 = Avx.Add(my1, mb1); |
||||
|
d.V6 = Avx.Subtract(my1, mb1); |
||||
|
d.V2 = Avx.Add(my2, mb2); |
||||
|
d.V5 = Avx.Subtract(my2, mb2); |
||||
|
d.V3 = Avx.Add(my3, mb3); |
||||
|
d.V4 = Avx.Subtract(my3, mb3); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
#endif
|
||||
@ -0,0 +1,199 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
using System; |
||||
|
using System.Runtime.CompilerServices; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Provides methods and properties related to jpeg quantization.
|
||||
|
/// </summary>
|
||||
|
internal static class Quantization |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// Upper bound (inclusive) for jpeg quality setting.
|
||||
|
/// </summary>
|
||||
|
public const int MaxQualityFactor = 100; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Lower bound (inclusive) for jpeg quality setting.
|
||||
|
/// </summary>
|
||||
|
public const int MinQualityFactor = 1; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Default JPEG quality for both luminance and chominance tables.
|
||||
|
/// </summary>
|
||||
|
public const int DefaultQualityFactor = 75; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Represents lowest quality setting which can be estimated with enough confidence.
|
||||
|
/// Any quality below it results in a highly compressed jpeg image
|
||||
|
/// which shouldn't use standard itu quantization tables for re-encoding.
|
||||
|
/// </summary>
|
||||
|
public const int QualityEstimationConfidenceLowerThreshold = 25; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Represents highest quality setting which can be estimated with enough confidence.
|
||||
|
/// </summary>
|
||||
|
public const int QualityEstimationConfidenceUpperThreshold = 98; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets unscaled luminance quantization table.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// The values are derived from ITU section K.1.
|
||||
|
/// </remarks>
|
||||
|
// The C# compiler emits this as a compile-time constant embedded in the PE file.
|
||||
|
// This is effectively compiled down to: return new ReadOnlySpan<byte>(&data, length)
|
||||
|
// More details can be found: https://github.com/dotnet/roslyn/pull/24621
|
||||
|
public static ReadOnlySpan<byte> LuminanceTable => new byte[] |
||||
|
{ |
||||
|
16, 11, 10, 16, 24, 40, 51, 61, |
||||
|
12, 12, 14, 19, 26, 58, 60, 55, |
||||
|
14, 13, 16, 24, 40, 57, 69, 56, |
||||
|
14, 17, 22, 29, 51, 87, 80, 62, |
||||
|
18, 22, 37, 56, 68, 109, 103, 77, |
||||
|
24, 35, 55, 64, 81, 104, 113, 92, |
||||
|
49, 64, 78, 87, 103, 121, 120, 101, |
||||
|
72, 92, 95, 98, 112, 100, 103, 99, |
||||
|
}; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets unscaled chrominance quantization table.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// The values are derived from ITU section K.1.
|
||||
|
/// </remarks>
|
||||
|
// The C# compiler emits this as a compile-time constant embedded in the PE file.
|
||||
|
// This is effectively compiled down to: return new ReadOnlySpan<byte>(&data, length)
|
||||
|
// More details can be found: https://github.com/dotnet/roslyn/pull/24621
|
||||
|
public static ReadOnlySpan<byte> ChrominanceTable => new byte[] |
||||
|
{ |
||||
|
17, 18, 24, 47, 99, 99, 99, 99, |
||||
|
18, 21, 26, 66, 99, 99, 99, 99, |
||||
|
24, 26, 56, 99, 99, 99, 99, 99, |
||||
|
47, 66, 99, 99, 99, 99, 99, 99, |
||||
|
99, 99, 99, 99, 99, 99, 99, 99, |
||||
|
99, 99, 99, 99, 99, 99, 99, 99, |
||||
|
99, 99, 99, 99, 99, 99, 99, 99, |
||||
|
99, 99, 99, 99, 99, 99, 99, 99, |
||||
|
}; |
||||
|
|
||||
|
/// Ported from JPEGsnoop:
|
||||
|
/// https://github.com/ImpulseAdventure/JPEGsnoop/blob/9732ee0961f100eb69bbff4a0c47438d5997abee/source/JfifDecode.cpp#L4570-L4694
|
||||
|
/// <summary>
|
||||
|
/// Estimates jpeg quality based on standard quantization table.
|
||||
|
/// </summary>
|
||||
|
/// <remarks>
|
||||
|
/// Technically, this can be used with any given table but internal decoder code uses ITU spec tables:
|
||||
|
/// <see cref="LuminanceTable"/> and <see cref="ChrominanceTable"/>.
|
||||
|
/// </remarks>
|
||||
|
/// <param name="table">Input quantization table.</param>
|
||||
|
/// <param name="target">Natural order quantization table to estimate against.</param>
|
||||
|
/// <returns>Estimated quality.</returns>
|
||||
|
public static int EstimateQuality(ref Block8x8F table, ReadOnlySpan<byte> target) |
||||
|
{ |
||||
|
// This method can be SIMD'ified if standard table is injected as Block8x8F.
|
||||
|
// Or when we go to full-int16 spectral code implementation and inject both tables as Block8x8.
|
||||
|
double comparePercent; |
||||
|
double sumPercent = 0; |
||||
|
|
||||
|
// Corner case - all 1's => 100 quality
|
||||
|
// It would fail to deduce using algorithm below without this check
|
||||
|
if (table.EqualsToScalar(1)) |
||||
|
{ |
||||
|
// While this is a 100% to be 100 quality, any given table can be scaled to all 1's.
|
||||
|
// According to jpeg creators, top of the line quality is 99, 100 is just a technical 'limit' which will affect result filesize drastically.
|
||||
|
// Quality=100 shouldn't be used in usual use case.
|
||||
|
return 100; |
||||
|
} |
||||
|
|
||||
|
int quality; |
||||
|
for (int i = 0; i < Block8x8F.Size; i++) |
||||
|
{ |
||||
|
int coeff = (int)table[i]; |
||||
|
|
||||
|
// Coefficients are actually int16 casted to float numbers so there's no truncating error.
|
||||
|
if (coeff != 0) |
||||
|
{ |
||||
|
comparePercent = 100.0 * (table[i] / target[i]); |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
// No 'valid' quantization table should contain zero at any position
|
||||
|
// while this is okay to decode with, it will throw DivideByZeroException at encoding proces stage.
|
||||
|
// Not sure what to do here, we can't throw as this technically correct
|
||||
|
// but this will screw up the encoder.
|
||||
|
comparePercent = 999.99; |
||||
|
} |
||||
|
|
||||
|
sumPercent += comparePercent; |
||||
|
} |
||||
|
|
||||
|
// Perform some statistical analysis of the quality factor
|
||||
|
// to determine the likelihood of the current quantization
|
||||
|
// table being a scaled version of the "standard" tables.
|
||||
|
// If the variance is high, it is unlikely to be the case.
|
||||
|
sumPercent /= 64.0; |
||||
|
|
||||
|
// Generate the equivalent IJQ "quality" factor
|
||||
|
if (sumPercent <= 100.0) |
||||
|
{ |
||||
|
quality = (int)Math.Round((200 - sumPercent) / 2); |
||||
|
} |
||||
|
else |
||||
|
{ |
||||
|
quality = (int)Math.Round(5000.0 / sumPercent); |
||||
|
} |
||||
|
|
||||
|
return quality; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Estimates jpeg quality based on quantization table in zig-zag order.
|
||||
|
/// </summary>
|
||||
|
/// <param name="luminanceTable">Luminance quantization table.</param>
|
||||
|
/// <returns>Estimated quality</returns>
|
||||
|
[MethodImpl(MethodImplOptions.AggressiveInlining)] |
||||
|
public static int EstimateLuminanceQuality(ref Block8x8F luminanceTable) |
||||
|
=> EstimateQuality(ref luminanceTable, LuminanceTable); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Estimates jpeg quality based on quantization table in zig-zag order.
|
||||
|
/// </summary>
|
||||
|
/// <param name="chrominanceTable">Chrominance quantization table.</param>
|
||||
|
/// <returns>Estimated quality</returns>
|
||||
|
[MethodImpl(MethodImplOptions.AggressiveInlining)] |
||||
|
public static int EstimateChrominanceQuality(ref Block8x8F chrominanceTable) |
||||
|
=> EstimateQuality(ref chrominanceTable, ChrominanceTable); |
||||
|
|
||||
|
[MethodImpl(MethodImplOptions.AggressiveInlining)] |
||||
|
private static int QualityToScale(int quality) |
||||
|
{ |
||||
|
DebugGuard.MustBeBetweenOrEqualTo(quality, MinQualityFactor, MaxQualityFactor, nameof(quality)); |
||||
|
|
||||
|
return quality < 50 ? (5000 / quality) : (200 - (quality * 2)); |
||||
|
} |
||||
|
|
||||
|
private static Block8x8F ScaleQuantizationTable(int scale, ReadOnlySpan<byte> unscaledTable) |
||||
|
{ |
||||
|
Block8x8F table = default; |
||||
|
for (int j = 0; j < Block8x8F.Size; j++) |
||||
|
{ |
||||
|
int x = ((unscaledTable[j] * scale) + 50) / 100; |
||||
|
table[j] = Numerics.Clamp(x, 1, 255); |
||||
|
} |
||||
|
|
||||
|
return table; |
||||
|
} |
||||
|
|
||||
|
[MethodImpl(MethodImplOptions.AggressiveInlining)] |
||||
|
public static Block8x8F ScaleLuminanceTable(int quality) |
||||
|
=> ScaleQuantizationTable(scale: QualityToScale(quality), LuminanceTable); |
||||
|
|
||||
|
[MethodImpl(MethodImplOptions.AggressiveInlining)] |
||||
|
public static Block8x8F ScaleChrominanceTable(int quality) |
||||
|
=> ScaleQuantizationTable(scale: QualityToScale(quality), ChrominanceTable); |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,300 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Apache License, Version 2.0.
|
||||
|
|
||||
|
#if SUPPORTS_RUNTIME_INTRINSICS
|
||||
|
using System; |
||||
|
using System.Runtime.Intrinsics; |
||||
|
using System.Runtime.Intrinsics.X86; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Formats.Jpeg.Components |
||||
|
{ |
||||
|
internal static partial class ZigZag |
||||
|
{ |
||||
|
#pragma warning disable SA1309 // naming rules violation warnings
|
||||
|
/// <summary>
|
||||
|
/// Special byte value to zero out elements during Sse/Avx shuffle intrinsics.
|
||||
|
/// </summary>
|
||||
|
private const byte _ = 0xff; |
||||
|
#pragma warning restore SA1309
|
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets shuffle vectors for <see cref="ApplyZigZagOrderingSsse3"/>
|
||||
|
/// zig zag implementation.
|
||||
|
/// </summary>
|
||||
|
private static ReadOnlySpan<byte> SseShuffleMasks => new byte[] |
||||
|
{ |
||||
|
// row0
|
||||
|
0, 1, 2, 3, _, _, _, _, _, _, 4, 5, 6, 7, _, _, |
||||
|
_, _, _, _, 0, 1, _, _, 2, 3, _, _, _, _, 4, 5, |
||||
|
_, _, _, _, _, _, 0, 1, _, _, _, _, _, _, _, _, |
||||
|
|
||||
|
// row1
|
||||
|
_, _, _, _, _, _, _, _, _, _, _, _, 8, 9, 10, 11, |
||||
|
2, 3, _, _, _, _, _, _, 4, 5, _, _, _, _, _, _, |
||||
|
_, _, 0, 1, _, _, 2, 3, _, _, _, _, _, _, _, _, |
||||
|
|
||||
|
// row2
|
||||
|
_, _, _, _, _, _, 2, 3, _, _, _, _, _, _, 4, 5, |
||||
|
_, _, _, _, _, _, _, _, 0, 1, _, _, 2, 3, _, _, |
||||
|
|
||||
|
// row3
|
||||
|
_, _, _, _, _, _, 12, 13, 14, 15, _, _, _, _, _, _, |
||||
|
_, _, _, _, 10, 11, _, _, _, _, 12, 13, _, _, _, _, |
||||
|
_, _, 8, 9, _, _, _, _, _, _, _, _, 10, 11, _, _, |
||||
|
6, 7, _, _, _, _, _, _, _, _, _, _, _, _, 8, 9, |
||||
|
|
||||
|
// row4
|
||||
|
_, _, 4, 5, _, _, _, _, _, _, _, _, 6, 7, _, _, |
||||
|
_, _, _, _, 2, 3, _, _, _, _, 4, 5, _, _, _, _, |
||||
|
_, _, _, _, _, _, 0, 1, 2, 3, _, _, _, _, _, _, |
||||
|
|
||||
|
// row5
|
||||
|
_, _, 12, 13, _, _, 14, 15, _, _, _, _, _, _, _, _, |
||||
|
10, 11, _, _, _, _, _, _, 12, 13, _, _, _, _, _, _, |
||||
|
|
||||
|
// row6
|
||||
|
_, _, _, _, _, _, _, _, 12, 13, _, _, 14, 15, _, _, |
||||
|
_, _, _, _, _, _, 10, 11, _, _, _, _, _, _, 12, 13, |
||||
|
4, 5, 6, 7, _, _, _, _, _, _, _, _, _, _, _, _, |
||||
|
|
||||
|
// row7
|
||||
|
10, 11, _, _, _, _, 12, 13, _, _, 14, 15, _, _, _, _, |
||||
|
_, _, 8, 9, 10, 11, _, _, _, _, _, _, 12, 13, 14, 15 |
||||
|
}; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets shuffle vectors for <see cref="ApplyZigZagOrderingAvx2"/>
|
||||
|
/// zig zag implementation.
|
||||
|
/// </summary>
|
||||
|
private static ReadOnlySpan<byte> AvxShuffleMasks => new byte[] |
||||
|
{ |
||||
|
// 01_AB/01_EF/23_CD - cross-lane
|
||||
|
0, 0, 0, 0, 1, 0, 0, 0, 4, 0, 0, 0, 5, 0, 0, 0, 0, 0, 0, 0, 2, 0, 0, 0, 5, 0, 0, 0, 6, 0, 0, 0, |
||||
|
|
||||
|
// 01_AB - inner-lane
|
||||
|
0, 1, 2, 3, 8, 9, _, _, 10, 11, 4, 5, 6, 7, 12, 13, _, _, _, _, _, _, _, _, _, _, 10, 11, 4, 5, 6, 7, |
||||
|
|
||||
|
// 01_CD/23_GH - cross-lane
|
||||
|
0, 0, 0, 0, 1, 0, 0, 0, 4, 0, 0, 0, _, _, _, _, 0, 0, 0, 0, 1, 0, 0, 0, 4, 0, 0, 0, _, _, _, _, |
||||
|
|
||||
|
// 01_CD - inner-lane
|
||||
|
_, _, _, _, _, _, 0, 1, _, _, _, _, _, _, _, _, 2, 3, 8, 9, _, _, 10, 11, 4, 5, _, _, _, _, _, _, |
||||
|
|
||||
|
// 01_EF - inner-lane
|
||||
|
_, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, 0, 1, _, _, _, _, _, _, _, _, _, _, |
||||
|
|
||||
|
// 23_AB/45_CD/67_EF - cross-lane
|
||||
|
3, 0, 0, 0, 6, 0, 0, 0, 7, 0, 0, 0, _, _, _, _, 3, 0, 0, 0, 6, 0, 0, 0, 7, 0, 0, 0, _, _, _, _, |
||||
|
|
||||
|
// 23_AB - inner-lane
|
||||
|
4, 5, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, 6, 7, 0, 1, 2, 3, 8, 9, _, _, _, _, |
||||
|
|
||||
|
// 23_CD - inner-lane
|
||||
|
_, _, 6, 7, 12, 13, _, _, _, _, _, _, _, _, _, _, 10, 11, 4, 5, _, _, _, _, _, _, _, _, 6, 7, 12, 13, |
||||
|
|
||||
|
// 23_EF - inner-lane
|
||||
|
_, _, _, _, _, _, 2, 3, 8, 9, _, _, 10, 11, 4, 5, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, |
||||
|
|
||||
|
// 23_GH - inner-lane
|
||||
|
_, _, _, _, _, _, _, _, _, _, 0, 1, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, |
||||
|
|
||||
|
// 45_AB - inner-lane
|
||||
|
_, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, 10, 11, _, _, _, _, _, _, _, _, _, _, |
||||
|
|
||||
|
// 45_CD - inner-lane
|
||||
|
_, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, 6, 7, 0, 1, _, _, 2, 3, 8, 9, _, _, _, _, _, _, |
||||
|
|
||||
|
// 45_EF - cross-lane
|
||||
|
1, 0, 0, 0, 2, 0, 0, 0, 5, 0, 0, 0, _, _, _, _, 2, 0, 0, 0, 3, 0, 0, 0, 6, 0, 0, 0, 7, 0, 0, 0, |
||||
|
|
||||
|
// 45_EF - inner-lane
|
||||
|
2, 3, 8, 9, _, _, _, _, _, _, _, _, 10, 11, 4, 5, _, _, _, _, _, _, _, _, _, _, 2, 3, 8, 9, _, _, |
||||
|
|
||||
|
// 45_GH - inner-lane
|
||||
|
_, _, _, _, 2, 3, 8, 9, 10, 11, 4, 5, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, 6, 7, |
||||
|
|
||||
|
// 67_CD - inner-lane
|
||||
|
_, _, _, _, _, _, _, _, _, _, 10, 11, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, _, |
||||
|
|
||||
|
// 67_EF - inner-lane
|
||||
|
_, _, _, _, _, _, 6, 7, 0, 1, _, _, 2, 3, 8, 9, _, _, _, _, _, _, _, _, 10, 11, _, _, _, _, _, _, |
||||
|
|
||||
|
// 67_GH - inner-lane
|
||||
|
8, 9, 10, 11, 4, 5, _, _, _, _, _, _, _, _, _, _, 2, 3, 8, 9, 10, 11, 4, 5, _, _, 6, 7, 12, 13, 14, 15 |
||||
|
}; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Applies zig zag ordering for given 8x8 matrix using SSE cpu intrinsics.
|
||||
|
/// </summary>
|
||||
|
/// <param name="block">Input matrix.</param>
|
||||
|
public static unsafe void ApplyZigZagOrderingSsse3(ref Block8x8 block) |
||||
|
{ |
||||
|
DebugGuard.IsTrue(Ssse3.IsSupported, "Ssse3 support is required to run this operation!"); |
||||
|
|
||||
|
fixed (byte* maskPtr = SseShuffleMasks) |
||||
|
{ |
||||
|
Vector128<byte> rowA = block.V0.AsByte(); |
||||
|
Vector128<byte> rowB = block.V1.AsByte(); |
||||
|
Vector128<byte> rowC = block.V2.AsByte(); |
||||
|
Vector128<byte> rowD = block.V3.AsByte(); |
||||
|
Vector128<byte> rowE = block.V4.AsByte(); |
||||
|
Vector128<byte> rowF = block.V5.AsByte(); |
||||
|
Vector128<byte> rowG = block.V6.AsByte(); |
||||
|
Vector128<byte> rowH = block.V7.AsByte(); |
||||
|
|
||||
|
// row0 - A0 A1 B0 C0 B1 A2 A3 B2
|
||||
|
Vector128<short> rowA0 = Ssse3.Shuffle(rowA, Sse2.LoadVector128(maskPtr + (16 * 0))).AsInt16(); |
||||
|
Vector128<short> rowB0 = Ssse3.Shuffle(rowB, Sse2.LoadVector128(maskPtr + (16 * 1))).AsInt16(); |
||||
|
Vector128<short> row0 = Sse2.Or(rowA0, rowB0); |
||||
|
Vector128<short> rowC0 = Ssse3.Shuffle(rowC, Sse2.LoadVector128(maskPtr + (16 * 2))).AsInt16(); |
||||
|
row0 = Sse2.Or(row0, rowC0); |
||||
|
|
||||
|
// row1 - C1 D0 E0 D1 C2 B3 A4 A5
|
||||
|
Vector128<short> rowA1 = Ssse3.Shuffle(rowA, Sse2.LoadVector128(maskPtr + (16 * 3))).AsInt16(); |
||||
|
Vector128<short> rowC1 = Ssse3.Shuffle(rowC, Sse2.LoadVector128(maskPtr + (16 * 4))).AsInt16(); |
||||
|
Vector128<short> row1 = Sse2.Or(rowA1, rowC1); |
||||
|
Vector128<short> rowD1 = Ssse3.Shuffle(rowD, Sse2.LoadVector128(maskPtr + (16 * 5))).AsInt16(); |
||||
|
row1 = Sse2.Or(row1, rowD1); |
||||
|
row1 = Sse2.Insert(row1.AsUInt16(), Sse2.Extract(rowB.AsUInt16(), 3), 5).AsInt16(); |
||||
|
row1 = Sse2.Insert(row1.AsUInt16(), Sse2.Extract(rowE.AsUInt16(), 0), 2).AsInt16(); |
||||
|
|
||||
|
// row2
|
||||
|
Vector128<short> rowE2 = Ssse3.Shuffle(rowE, Sse2.LoadVector128(maskPtr + (16 * 6))).AsInt16(); |
||||
|
Vector128<short> rowF2 = Ssse3.Shuffle(rowF, Sse2.LoadVector128(maskPtr + (16 * 7))).AsInt16(); |
||||
|
Vector128<short> row2 = Sse2.Or(rowE2, rowF2); |
||||
|
row2 = Sse2.Insert(row2.AsUInt16(), Sse2.Extract(rowB.AsUInt16(), 4), 0).AsInt16(); |
||||
|
row2 = Sse2.Insert(row2.AsUInt16(), Sse2.Extract(rowC.AsUInt16(), 3), 1).AsInt16(); |
||||
|
row2 = Sse2.Insert(row2.AsUInt16(), Sse2.Extract(rowD.AsUInt16(), 2), 2).AsInt16(); |
||||
|
row2 = Sse2.Insert(row2.AsUInt16(), Sse2.Extract(rowG.AsUInt16(), 0), 5).AsInt16(); |
||||
|
|
||||
|
// row3
|
||||
|
Vector128<short> rowA3 = Ssse3.Shuffle(rowA, Sse2.LoadVector128(maskPtr + (16 * 8))).AsInt16().AsInt16(); |
||||
|
Vector128<short> rowB3 = Ssse3.Shuffle(rowB, Sse2.LoadVector128(maskPtr + (16 * 9))).AsInt16().AsInt16(); |
||||
|
Vector128<short> row3 = Sse2.Or(rowA3, rowB3); |
||||
|
Vector128<short> rowC3 = Ssse3.Shuffle(rowC, Sse2.LoadVector128(maskPtr + (16 * 10))).AsInt16(); |
||||
|
row3 = Sse2.Or(row3, rowC3); |
||||
|
Vector128<byte> shuffleRowD3EF = Sse2.LoadVector128(maskPtr + (16 * 11)); |
||||
|
Vector128<short> rowD3 = Ssse3.Shuffle(rowD, shuffleRowD3EF).AsInt16(); |
||||
|
row3 = Sse2.Or(row3, rowD3); |
||||
|
|
||||
|
// row4
|
||||
|
Vector128<short> rowE4 = Ssse3.Shuffle(rowE, shuffleRowD3EF).AsInt16(); |
||||
|
Vector128<short> rowF4 = Ssse3.Shuffle(rowF, Sse2.LoadVector128(maskPtr + (16 * 12))).AsInt16(); |
||||
|
Vector128<short> row4 = Sse2.Or(rowE4, rowF4); |
||||
|
Vector128<short> rowG4 = Ssse3.Shuffle(rowG, Sse2.LoadVector128(maskPtr + (16 * 13))).AsInt16(); |
||||
|
row4 = Sse2.Or(row4, rowG4); |
||||
|
Vector128<short> rowH4 = Ssse3.Shuffle(rowH, Sse2.LoadVector128(maskPtr + (16 * 14))).AsInt16(); |
||||
|
row4 = Sse2.Or(row4, rowH4); |
||||
|
|
||||
|
// row5
|
||||
|
Vector128<short> rowC5 = Ssse3.Shuffle(rowC, Sse2.LoadVector128(maskPtr + (16 * 15))).AsInt16(); |
||||
|
Vector128<short> rowD5 = Ssse3.Shuffle(rowD, Sse2.LoadVector128(maskPtr + (16 * 16))).AsInt16(); |
||||
|
Vector128<short> row5 = Sse2.Or(rowC5, rowD5); |
||||
|
row5 = Sse2.Insert(row5.AsUInt16(), Sse2.Extract(rowB.AsUInt16(), 7), 2).AsInt16(); |
||||
|
row5 = Sse2.Insert(row5.AsUInt16(), Sse2.Extract(rowE.AsUInt16(), 5), 5).AsInt16(); |
||||
|
row5 = Sse2.Insert(row5.AsUInt16(), Sse2.Extract(rowF.AsUInt16(), 4), 6).AsInt16(); |
||||
|
row5 = Sse2.Insert(row5.AsUInt16(), Sse2.Extract(rowG.AsUInt16(), 3), 7).AsInt16(); |
||||
|
|
||||
|
// row6
|
||||
|
Vector128<short> rowE6 = Ssse3.Shuffle(rowE, Sse2.LoadVector128(maskPtr + (16 * 17))).AsInt16(); |
||||
|
Vector128<short> rowF6 = Ssse3.Shuffle(rowF, Sse2.LoadVector128(maskPtr + (16 * 18))).AsInt16(); |
||||
|
Vector128<short> row6 = Sse2.Or(rowE6, rowF6); |
||||
|
Vector128<short> rowH6 = Ssse3.Shuffle(rowH, Sse2.LoadVector128(maskPtr + (16 * 19))).AsInt16(); |
||||
|
row6 = Sse2.Or(row6, rowH6); |
||||
|
row6 = Sse2.Insert(row6.AsUInt16(), Sse2.Extract(rowD.AsUInt16(), 7), 5).AsInt16(); |
||||
|
row6 = Sse2.Insert(row6.AsUInt16(), Sse2.Extract(rowG.AsUInt16(), 4), 2).AsInt16(); |
||||
|
|
||||
|
// row7
|
||||
|
Vector128<short> rowG7 = Ssse3.Shuffle(rowG, Sse2.LoadVector128(maskPtr + (16 * 20))).AsInt16(); |
||||
|
Vector128<short> rowH7 = Ssse3.Shuffle(rowH, Sse2.LoadVector128(maskPtr + (16 * 21))).AsInt16(); |
||||
|
Vector128<short> row7 = Sse2.Or(rowG7, rowH7); |
||||
|
row7 = Sse2.Insert(row7.AsUInt16(), Sse2.Extract(rowF.AsUInt16(), 7), 4).AsInt16(); |
||||
|
|
||||
|
block.V0 = row0; |
||||
|
block.V1 = row1; |
||||
|
block.V2 = row2; |
||||
|
block.V3 = row3; |
||||
|
block.V4 = row4; |
||||
|
block.V5 = row5; |
||||
|
block.V6 = row6; |
||||
|
block.V7 = row7; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Applies zig zag ordering for given 8x8 matrix using AVX cpu intrinsics.
|
||||
|
/// </summary>
|
||||
|
/// <param name="block">Input matrix.</param>
|
||||
|
public static unsafe void ApplyZigZagOrderingAvx2(ref Block8x8 block) |
||||
|
{ |
||||
|
DebugGuard.IsTrue(Avx2.IsSupported, "Avx2 support is required to run this operation!"); |
||||
|
|
||||
|
fixed (byte* shuffleVectorsPtr = AvxShuffleMasks) |
||||
|
{ |
||||
|
Vector256<byte> rowsAB = block.V01.AsByte(); |
||||
|
Vector256<byte> rowsCD = block.V23.AsByte(); |
||||
|
Vector256<byte> rowsEF = block.V45.AsByte(); |
||||
|
Vector256<byte> rowsGH = block.V67.AsByte(); |
||||
|
|
||||
|
// rows 0 1
|
||||
|
Vector256<int> rows_AB01_EF01_CD23_shuffleMask = Avx.LoadVector256(shuffleVectorsPtr + (0 * 32)).AsInt32(); |
||||
|
Vector256<byte> row01_AB = Avx2.PermuteVar8x32(rowsAB.AsInt32(), rows_AB01_EF01_CD23_shuffleMask).AsByte(); |
||||
|
row01_AB = Avx2.Shuffle(row01_AB, Avx.LoadVector256(shuffleVectorsPtr + (1 * 32))).AsByte(); |
||||
|
|
||||
|
Vector256<int> rows_CD01_GH23_shuffleMask = Avx.LoadVector256(shuffleVectorsPtr + (2 * 32)).AsInt32(); |
||||
|
Vector256<byte> row01_CD = Avx2.PermuteVar8x32(rowsCD.AsInt32(), rows_CD01_GH23_shuffleMask).AsByte(); |
||||
|
row01_CD = Avx2.Shuffle(row01_CD, Avx.LoadVector256(shuffleVectorsPtr + (3 * 32))).AsByte(); |
||||
|
|
||||
|
Vector256<byte> row0123_EF = Avx2.PermuteVar8x32(rowsEF.AsInt32(), rows_AB01_EF01_CD23_shuffleMask).AsByte(); |
||||
|
Vector256<byte> row01_EF = Avx2.Shuffle(row0123_EF, Avx.LoadVector256(shuffleVectorsPtr + (4 * 32))).AsByte(); |
||||
|
|
||||
|
Vector256<byte> row01 = Avx2.Or(Avx2.Or(row01_AB, row01_CD), row01_EF); |
||||
|
|
||||
|
// rows 2 3
|
||||
|
Vector256<int> rows_AB23_CD45_EF67_shuffleMask = Avx.LoadVector256(shuffleVectorsPtr + (5 * 32)).AsInt32(); |
||||
|
Vector256<byte> row2345_AB = Avx2.PermuteVar8x32(rowsAB.AsInt32(), rows_AB23_CD45_EF67_shuffleMask).AsByte(); |
||||
|
Vector256<byte> row23_AB = Avx2.Shuffle(row2345_AB, Avx.LoadVector256(shuffleVectorsPtr + (6 * 32))).AsByte(); |
||||
|
|
||||
|
Vector256<byte> row23_CD = Avx2.PermuteVar8x32(rowsCD.AsInt32(), rows_AB01_EF01_CD23_shuffleMask).AsByte(); |
||||
|
row23_CD = Avx2.Shuffle(row23_CD, Avx.LoadVector256(shuffleVectorsPtr + (7 * 32))).AsByte(); |
||||
|
|
||||
|
Vector256<byte> row23_EF = Avx2.Shuffle(row0123_EF, Avx.LoadVector256(shuffleVectorsPtr + (8 * 32))).AsByte(); |
||||
|
|
||||
|
Vector256<byte> row2345_GH = Avx2.PermuteVar8x32(rowsGH.AsInt32(), rows_CD01_GH23_shuffleMask).AsByte(); |
||||
|
Vector256<byte> row23_GH = Avx2.Shuffle(row2345_GH, Avx.LoadVector256(shuffleVectorsPtr + (9 * 32)).AsByte()); |
||||
|
|
||||
|
Vector256<byte> row23 = Avx2.Or(Avx2.Or(row23_AB, row23_CD), Avx2.Or(row23_EF, row23_GH)); |
||||
|
|
||||
|
// rows 4 5
|
||||
|
Vector256<byte> row45_AB = Avx2.Shuffle(row2345_AB, Avx.LoadVector256(shuffleVectorsPtr + (10 * 32)).AsByte()); |
||||
|
Vector256<byte> row4567_CD = Avx2.PermuteVar8x32(rowsCD.AsInt32(), rows_AB23_CD45_EF67_shuffleMask).AsByte(); |
||||
|
Vector256<byte> row45_CD = Avx2.Shuffle(row4567_CD, Avx.LoadVector256(shuffleVectorsPtr + (11 * 32)).AsByte()); |
||||
|
|
||||
|
Vector256<int> rows_EF45_GH67_shuffleMask = Avx.LoadVector256(shuffleVectorsPtr + (12 * 32)).AsInt32(); |
||||
|
Vector256<byte> row45_EF = Avx2.PermuteVar8x32(rowsEF.AsInt32(), rows_EF45_GH67_shuffleMask).AsByte(); |
||||
|
row45_EF = Avx2.Shuffle(row45_EF, Avx.LoadVector256(shuffleVectorsPtr + (13 * 32)).AsByte()); |
||||
|
|
||||
|
Vector256<byte> row45_GH = Avx2.Shuffle(row2345_GH, Avx.LoadVector256(shuffleVectorsPtr + (14 * 32)).AsByte()); |
||||
|
|
||||
|
Vector256<byte> row45 = Avx2.Or(Avx2.Or(row45_AB, row45_CD), Avx2.Or(row45_EF, row45_GH)); |
||||
|
|
||||
|
// rows 6 7
|
||||
|
Vector256<byte> row67_CD = Avx2.Shuffle(row4567_CD, Avx.LoadVector256(shuffleVectorsPtr + (15 * 32)).AsByte()); |
||||
|
|
||||
|
Vector256<byte> row67_EF = Avx2.PermuteVar8x32(rowsEF.AsInt32(), rows_AB23_CD45_EF67_shuffleMask).AsByte(); |
||||
|
row67_EF = Avx2.Shuffle(row67_EF, Avx.LoadVector256(shuffleVectorsPtr + (16 * 32)).AsByte()); |
||||
|
|
||||
|
Vector256<byte> row67_GH = Avx2.PermuteVar8x32(rowsGH.AsInt32(), rows_EF45_GH67_shuffleMask).AsByte(); |
||||
|
row67_GH = Avx2.Shuffle(row67_GH, Avx.LoadVector256(shuffleVectorsPtr + (17 * 32)).AsByte()); |
||||
|
|
||||
|
Vector256<byte> row67 = Avx2.Or(Avx2.Or(row67_CD, row67_EF), row67_GH); |
||||
|
|
||||
|
block.V01 = row01.AsInt16(); |
||||
|
block.V23 = row23.AsInt16(); |
||||
|
block.V45 = row45.AsInt16(); |
||||
|
block.V67 = row67.AsInt16(); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
#endif
|
||||
Some files were not shown because too many files changed in this diff
Loading…
Reference in new issue