mirror of https://github.com/SixLabors/ImageSharp
6 changed files with 618 additions and 0 deletions
@ -0,0 +1,201 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Six Labors Split License.
|
||||
|
|
||||
|
using System.Numerics; |
||||
|
using BenchmarkDotNet.Attributes; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Components; |
||||
|
using SixLabors.ImageSharp.Memory; |
||||
|
using SixLabors.ImageSharp.PixelFormats; |
||||
|
using SixLabors.ImageSharp.Processing; |
||||
|
using SixLabors.ImageSharp.Tests; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Benchmarks.Codecs.Heif; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Measures encoding a photographic sequence with fractional motion at the interpolation-search effort boundaries.
|
||||
|
/// </summary>
|
||||
|
[MemoryDiagnoser] |
||||
|
public class Av1SequenceEncoderBenchmarks |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// The number of displayed pictures in each independently encoded sequence.
|
||||
|
/// </summary>
|
||||
|
private const int FrameCount = 3; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The native AV1 quantizer index corresponding to libaom's public constant-quality level 30.
|
||||
|
/// </summary>
|
||||
|
private const int QIndex = 120; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The fixed native speed baseline, independent of ImageSharp's effort scale.
|
||||
|
/// </summary>
|
||||
|
private const int NativeCpuUsed = 6; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// The native public quantizer corresponding to <see cref="QIndex"/>, also used for both rate-control bounds.
|
||||
|
/// </summary>
|
||||
|
private const int NativeQuality = 30; |
||||
|
|
||||
|
private Image<Rgb24> sequence; |
||||
|
private Configuration configuration; |
||||
|
private ObuColorConfig colorConfig; |
||||
|
private MemoryStream output; |
||||
|
private string outputDirectory; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets or sets the square frame dimension.
|
||||
|
/// </summary>
|
||||
|
[Params(256, 512)] |
||||
|
public int Dimension { get; set; } |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Gets or sets the effort controlling fixed, common switchable, or independently switchable filters.
|
||||
|
/// </summary>
|
||||
|
[Params(7, 8, 9)] |
||||
|
public int Effort { get; set; } |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Prepares identical photographic RGB frames and a planar source file for checking reconstructed output quality.
|
||||
|
/// </summary>
|
||||
|
[GlobalSetup] |
||||
|
public void Setup() |
||||
|
{ |
||||
|
this.configuration = Configuration.Default.Clone(); |
||||
|
this.configuration.MaxDegreeOfParallelism = 1; |
||||
|
this.colorConfig = new ObuColorConfig |
||||
|
{ |
||||
|
BitDepth = Av1BitDepth.EightBit, |
||||
|
IsColorDescriptionPresent = true, |
||||
|
ColorPrimaries = ObuColorPrimaries.Bt601, |
||||
|
TransferCharacteristics = ObuTransferCharacteristics.Bt601, |
||||
|
MatrixCoefficients = ObuMatrixCoefficients.Bt601, |
||||
|
ColorRange = true, |
||||
|
SubSamplingX = true, |
||||
|
SubSamplingY = true, |
||||
|
ChromaSamplePosition = ObuChromoSamplePosition.Unknown |
||||
|
}; |
||||
|
|
||||
|
this.output = new MemoryStream(); |
||||
|
this.outputDirectory = TestEnvironment.CreateOutputDirectory("Heif", "Av1", nameof(Av1SequenceEncoderBenchmarks)); |
||||
|
string inputPath = Path.Combine(TestEnvironment.InputImagesDirectoryFullPath, TestImages.Png.Bike); |
||||
|
using Image<Rgb24> photograph = Image.Load<Rgb24>(inputPath); |
||||
|
|
||||
|
// Leave a source margin for the half-pixel translations. Resampling is setup work, not encoder time;
|
||||
|
// every invocation consumes the same three images rather than repeatedly translating a previous result.
|
||||
|
photograph.Mutate(context => context.Resize(new ResizeOptions |
||||
|
{ |
||||
|
Size = new Size(this.Dimension + FrameCount, this.Dimension + FrameCount), |
||||
|
Mode = ResizeMode.Crop |
||||
|
})); |
||||
|
|
||||
|
Rectangle sourceBounds = new(0, 0, photograph.Width, photograph.Height); |
||||
|
Size targetSize = new(this.Dimension, this.Dimension); |
||||
|
this.sequence = photograph.Clone(context => context.Crop(new Rectangle(Point.Empty, targetSize))); |
||||
|
for (int frameIndex = 1; frameIndex < FrameCount; frameIndex++) |
||||
|
{ |
||||
|
Matrix3x2 translation = Matrix3x2.CreateTranslation(-0.5F * frameIndex, -0.5F * frameIndex); |
||||
|
using Image<Rgb24> translated = photograph.Clone(context => |
||||
|
context.Transform(sourceBounds, translation, targetSize, KnownResamplers.Bicubic)); |
||||
|
|
||||
|
this.sequence.Frames.AddFrame(translated.Frames.RootFrame); |
||||
|
} |
||||
|
|
||||
|
// Export the production-converted source planes only for checking reconstructed output quality.
|
||||
|
// Neither timed encoder reads this file: both convert the original RGB frames during each operation.
|
||||
|
using Av1EncoderFrameBuffer<byte> planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0); |
||||
|
using FileStream raw = File.Create(Path.Combine(this.outputDirectory, $"bike-{this.Dimension}-3frames.source.yuv")); |
||||
|
foreach (ImageFrame<Rgb24> frame in this.sequence.Frames) |
||||
|
{ |
||||
|
Av1FrameEncoder.PrepareSource(this.configuration, frame, planar.Frame, this.colorConfig); |
||||
|
for (int planeIndex = 0; planeIndex < this.colorConfig.PlaneCount; planeIndex++) |
||||
|
{ |
||||
|
Buffer2DRegion<byte> plane = planar.Frame.View.GetPlane((Av1Plane)planeIndex); |
||||
|
for (int y = 0; y < plane.Height; y++) |
||||
|
{ |
||||
|
raw.Write(plane.DangerousGetRowSpan(y)); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Encodes one key picture and two dependent pictures, returning the complete OBU payload length.
|
||||
|
/// </summary>
|
||||
|
/// <returns>The encoded sequence length.</returns>
|
||||
|
[Benchmark] |
||||
|
public long ImageSharp() |
||||
|
{ |
||||
|
this.output.SetLength(0); |
||||
|
using Av1FrameEncoder.SequenceEncoder encoder = Av1FrameEncoder.CreateColorSequenceEncoder( |
||||
|
this.configuration, this.Dimension, this.Dimension, this.colorConfig, QIndex, this.Effort); |
||||
|
|
||||
|
// One operation owns the real sequence lifetime: allocation, conversion, key/inter coding, and disposal.
|
||||
|
// The caller's destination is reused, excluding filesystem and MemoryStream growth from steady-state timing.
|
||||
|
encoder.EncodeKeyFrame(this.sequence.Frames.RootFrame, this.output); |
||||
|
for (int frameIndex = 1; frameIndex < FrameCount; frameIndex++) |
||||
|
{ |
||||
|
encoder.EncodeInterFrame(this.sequence.Frames[frameIndex], this.output); |
||||
|
} |
||||
|
|
||||
|
return this.output.Length; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Encodes the same RGB sequence with current-main libaom, including conversion, allocation, output, and disposal.
|
||||
|
/// </summary>
|
||||
|
/// <returns>The encoded sequence length.</returns>
|
||||
|
[Benchmark(Baseline = true)] |
||||
|
public long Libaom() |
||||
|
{ |
||||
|
// cpu-used is a separate speed scale, not an ImageSharp effort mapping. Keep the reference at
|
||||
|
// good-quality speed six while comparing the three managed interpolation-search boundaries.
|
||||
|
this.output.SetLength(0); |
||||
|
using LibaomBenchmarkEncoder encoder = LibaomBenchmarkEncoder.Open(this.Dimension, this.Dimension, NativeQuality, NativeCpuUsed); |
||||
|
using Av1EncoderFrameBuffer<byte> planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0); |
||||
|
using Av1FrameEncoder.Av1EncoderConversionWorkspace conversion = new(this.configuration, this.Dimension, this.colorConfig, false, false); |
||||
|
Rectangle bounds = new(0, 0, this.Dimension, this.Dimension); |
||||
|
for (int frameIndex = 0; frameIndex < FrameCount; frameIndex++) |
||||
|
{ |
||||
|
// Conversion belongs inside both measured paths. Reuse the same row workspace and SIMD converter
|
||||
|
// as the managed sequence encoder, writing directly into the planes passed to native libaom.
|
||||
|
conversion.Convert<Rgb24, Av1EncoderFrame<byte>.PlanarView, byte, HeifByteSampleConverter>( |
||||
|
this.configuration, this.sequence.Frames[frameIndex], bounds, planar.Frame.View); |
||||
|
|
||||
|
encoder.Encode(planar.Frame, frameIndex, this.output); |
||||
|
} |
||||
|
|
||||
|
encoder.Finish(this.output); |
||||
|
return this.output.Length; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Retains the measured managed encoder output and releases the input images and destination stream.
|
||||
|
/// </summary>
|
||||
|
[GlobalCleanup(Target = nameof(ImageSharp))] |
||||
|
public void CleanupImageSharp() => this.Cleanup($"bike-{this.Dimension}-q{QIndex}-effort{this.Effort}.obu"); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Retains the measured reference encoder output and releases the input images and destination stream.
|
||||
|
/// </summary>
|
||||
|
[GlobalCleanup(Target = nameof(Libaom))] |
||||
|
public void CleanupLibaom() => this.Cleanup($"bike-{this.Dimension}-q{QIndex}-libaom-cpu{NativeCpuUsed}.obu"); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Writes the measured payload without another encoding pass and releases the shared benchmark resources.
|
||||
|
/// </summary>
|
||||
|
/// <param name="outputName">The codec-specific payload file name.</param>
|
||||
|
private void Cleanup(string outputName) |
||||
|
{ |
||||
|
// Output validation and quality measurement use the actual measured payload, with no encode or decode
|
||||
|
// hidden inside the timed operation and no file-sized ToArray copy.
|
||||
|
using FileStream encoded = File.Create(Path.Combine(this.outputDirectory, outputName)); |
||||
|
this.output.Position = 0; |
||||
|
this.output.CopyTo(encoded); |
||||
|
this.output.Dispose(); |
||||
|
this.sequence.Dispose(); |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,135 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Six Labors Split License.
|
||||
|
|
||||
|
using System.Runtime.InteropServices; |
||||
|
using Microsoft.Win32.SafeHandles; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1; |
||||
|
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; |
||||
|
using SixLabors.ImageSharp.Memory; |
||||
|
|
||||
|
namespace SixLabors.ImageSharp.Benchmarks.Codecs.Heif; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Owns a benchmark-only libaom encoder through a C adapter compiled against the reference's actual headers.
|
||||
|
/// </summary>
|
||||
|
internal sealed unsafe partial class LibaomBenchmarkEncoder : SafeHandleZeroOrMinusOneIsInvalid |
||||
|
{ |
||||
|
/// <summary>
|
||||
|
/// The benchmark adapter's platform-independent library name.
|
||||
|
/// </summary>
|
||||
|
private const string LibraryName = "imagesharp_aom_benchmark"; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Initializes an empty handle for the native create call's generated marshaller.
|
||||
|
/// </summary>
|
||||
|
public LibaomBenchmarkEncoder() |
||||
|
: base(ownsHandle: true) |
||||
|
{ |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Opens one native sequence with a single coding thread, no lookahead, and the requested quality and speed.
|
||||
|
/// </summary>
|
||||
|
public static LibaomBenchmarkEncoder Open(int width, int height, int quality, int speed) |
||||
|
{ |
||||
|
CheckStatus(Create((uint)width, (uint)height, (uint)quality, speed, out LibaomBenchmarkEncoder encoder)); |
||||
|
return encoder; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Encodes the converted source planes and consumes every output packet before those bytes can be invalidated.
|
||||
|
/// </summary>
|
||||
|
public void Encode(Av1EncoderFrame<byte> frame, long frameIndex, Stream output) |
||||
|
{ |
||||
|
Buffer2DRegion<byte> y = frame.View.GetPlane(Av1Plane.Y); |
||||
|
Buffer2DRegion<byte> u = frame.View.GetPlane(Av1Plane.U); |
||||
|
Buffer2DRegion<byte> v = frame.View.GetPlane(Av1Plane.V); |
||||
|
fixed (byte* yPointer = y.DangerousGetRowSpan(0), uPointer = u.DangerousGetRowSpan(0), vPointer = v.DangerousGetRowSpan(0)) |
||||
|
{ |
||||
|
// The native call is synchronous. Its input descriptors borrow these pinned rows only until
|
||||
|
// EncodeFrame returns; the encoder owns any retained reference and lookahead storage itself.
|
||||
|
CheckStatus(EncodeFrame(this, yPointer, uPointer, vPointer, y.Stride, u.Stride, frameIndex)); |
||||
|
} |
||||
|
|
||||
|
this.WritePackets(output); |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Finishes the sequence and writes any remaining coded packets.
|
||||
|
/// </summary>
|
||||
|
public void Finish(Stream output) |
||||
|
{ |
||||
|
do |
||||
|
{ |
||||
|
CheckStatus(Flush(this)); |
||||
|
} |
||||
|
while (this.WritePackets(output)); |
||||
|
} |
||||
|
|
||||
|
/// <inheritdoc/>
|
||||
|
protected override bool ReleaseHandle() => Destroy(this.handle) == 0; |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Writes borrowed packet memory before the next call into the native codec invalidates it.
|
||||
|
/// </summary>
|
||||
|
private bool WritePackets(Stream output) |
||||
|
{ |
||||
|
bool wrotePacket = false; |
||||
|
while (NextPacket(this, out byte* data, out nuint length) != 0) |
||||
|
{ |
||||
|
// Packet lengths are bounded by the benchmark's image dimensions. Stream.Write consumes the
|
||||
|
// borrowed bytes synchronously, without retaining a native pointer or creating a managed array.
|
||||
|
output.Write(new ReadOnlySpan<byte>(data, (int)length)); |
||||
|
wrotePacket = true; |
||||
|
} |
||||
|
|
||||
|
return wrotePacket; |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Converts a native codec error into a managed benchmark failure instead of accepting invalid timing data.
|
||||
|
/// </summary>
|
||||
|
private static void CheckStatus(int status) |
||||
|
{ |
||||
|
if (status != 0) |
||||
|
{ |
||||
|
throw new InvalidOperationException(Marshal.PtrToStringUTF8(ErrorString(status))); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Creates an owned opaque context; no libaom structure layout crosses the managed boundary.
|
||||
|
/// </summary>
|
||||
|
[LibraryImport(LibraryName, EntryPoint = "benchmark_create")] |
||||
|
private static partial int Create(uint width, uint height, uint quality, int speed, out LibaomBenchmarkEncoder encoder); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Borrows three pinned planes for one synchronous native encode call.
|
||||
|
/// </summary>
|
||||
|
[LibraryImport(LibraryName, EntryPoint = "benchmark_encode")] |
||||
|
private static partial int EncodeFrame(LibaomBenchmarkEncoder encoder, byte* y, byte* u, byte* v, int yStride, int uvStride, long frameIndex); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Signals the end of the native sequence.
|
||||
|
/// </summary>
|
||||
|
[LibraryImport(LibraryName, EntryPoint = "benchmark_flush")] |
||||
|
private static partial int Flush(LibaomBenchmarkEncoder encoder); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Returns borrowed native packet storage and its pointer-sized length.
|
||||
|
/// </summary>
|
||||
|
[LibraryImport(LibraryName, EntryPoint = "benchmark_next_packet")] |
||||
|
private static partial int NextPacket(LibaomBenchmarkEncoder encoder, out byte* data, out nuint length); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Destroys the context through the same native library that allocated it.
|
||||
|
/// </summary>
|
||||
|
[LibraryImport(LibraryName, EntryPoint = "benchmark_destroy")] |
||||
|
private static partial int Destroy(nint encoder); |
||||
|
|
||||
|
/// <summary>
|
||||
|
/// Returns a static UTF-8 error message owned by libaom.
|
||||
|
/// </summary>
|
||||
|
[LibraryImport(LibraryName, EntryPoint = "benchmark_error_string")] |
||||
|
private static partial nint ErrorString(int status); |
||||
|
} |
||||
@ -0,0 +1,12 @@ |
|||||
|
cmake_minimum_required(VERSION 3.20) |
||||
|
project(imagesharp_aom_benchmark LANGUAGES C CXX) |
||||
|
|
||||
|
# Link the explicitly supplied current-main Release reference build. This adapter is benchmark-only; |
||||
|
# it is never referenced by ImageSharp or copied by the production project. |
||||
|
set(AOM_SOURCE_DIRECTORY "" CACHE PATH "Verified official libaom source directory") |
||||
|
set(AOM_BUILD_DIRECTORY "" CACHE PATH "Matching optimized Release libaom build directory") |
||||
|
add_library(imagesharp_aom_benchmark SHARED aom_benchmark.c) |
||||
|
target_include_directories(imagesharp_aom_benchmark PRIVATE "${AOM_SOURCE_DIRECTORY}") |
||||
|
find_library(AOM_LIBRARY NAMES aom PATHS "${AOM_BUILD_DIRECTORY}" NO_DEFAULT_PATH REQUIRED) |
||||
|
target_link_libraries(imagesharp_aom_benchmark PRIVATE "${AOM_LIBRARY}") |
||||
|
set_target_properties(imagesharp_aom_benchmark PROPERTIES C_STANDARD 11 LINKER_LANGUAGE CXX) |
||||
@ -0,0 +1,131 @@ |
|||||
|
// Copyright (c) Six Labors.
|
||||
|
// Licensed under the Six Labors Split License.
|
||||
|
|
||||
|
#include <stdint.h> |
||||
|
#include <stdlib.h> |
||||
|
#include "aom/aom_encoder.h" |
||||
|
#include "aom/aomcx.h" |
||||
|
|
||||
|
#if defined(_WIN32) |
||||
|
#define BENCHMARK_API __declspec(dllexport) |
||||
|
#else |
||||
|
#define BENCHMARK_API __attribute__((visibility("default"))) |
||||
|
#endif |
||||
|
|
||||
|
// Keep libaom's version-dependent structures and variadic controls entirely on the C side.
|
||||
|
// The managed benchmark exchanges only opaque ownership, fixed-width integers, and borrowed buffers.
|
||||
|
typedef struct benchmark_encoder { |
||||
|
aom_codec_ctx_t codec; |
||||
|
aom_codec_iter_t iterator; |
||||
|
unsigned int width; |
||||
|
unsigned int height; |
||||
|
} benchmark_encoder; |
||||
|
|
||||
|
// Creates one single-threaded, unlagged, constant-quality sequence using the reference's normal tools.
|
||||
|
// A successful result belongs to the caller and must be released by benchmark_destroy.
|
||||
|
BENCHMARK_API int benchmark_create(unsigned int width, unsigned int height, |
||||
|
unsigned int quality, int speed, |
||||
|
benchmark_encoder **result) { |
||||
|
aom_codec_enc_cfg_t config; |
||||
|
aom_codec_iface_t *iface = aom_codec_av1_cx(); |
||||
|
aom_codec_err_t status = aom_codec_enc_config_default(iface, &config, AOM_USAGE_GOOD_QUALITY); |
||||
|
*result = NULL; |
||||
|
if (status != AOM_CODEC_OK) return status; |
||||
|
|
||||
|
benchmark_encoder *encoder = calloc(1, sizeof(*encoder)); |
||||
|
if (encoder == NULL) return AOM_CODEC_MEM_ERROR; |
||||
|
|
||||
|
config.g_w = width; |
||||
|
config.g_h = height; |
||||
|
config.g_threads = 1; |
||||
|
config.g_lag_in_frames = 0; |
||||
|
config.g_timebase.num = 1; |
||||
|
config.g_timebase.den = 30; |
||||
|
config.rc_end_usage = AOM_Q; |
||||
|
// Fix both bounds to the requested quantizer. CQ alone permits frame-quality boosts, whereas
|
||||
|
// the managed sequence uses this same base quantizer for every frame in the comparison.
|
||||
|
config.rc_min_quantizer = quality; |
||||
|
config.rc_max_quantizer = quality; |
||||
|
config.kf_mode = AOM_KF_DISABLED; |
||||
|
status = aom_codec_enc_init(&encoder->codec, iface, &config, 0); |
||||
|
if (status != AOM_CODEC_OK) { |
||||
|
free(encoder); |
||||
|
return status; |
||||
|
} |
||||
|
|
||||
|
// These calls remain type-checked against the current libaom headers, including each control's argument type.
|
||||
|
status = aom_codec_control(&encoder->codec, AOME_SET_CPUUSED, speed); |
||||
|
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AOME_SET_CQ_LEVEL, quality); |
||||
|
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_ROW_MT, 0u); |
||||
|
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_COLOR_PRIMARIES, AOM_CICP_CP_BT_601); |
||||
|
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_TRANSFER_CHARACTERISTICS, AOM_CICP_TC_BT_601); |
||||
|
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_MATRIX_COEFFICIENTS, AOM_CICP_MC_BT_601); |
||||
|
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_COLOR_RANGE, AOM_CR_FULL_RANGE); |
||||
|
if (status != AOM_CODEC_OK) { |
||||
|
aom_codec_destroy(&encoder->codec); |
||||
|
free(encoder); |
||||
|
return status; |
||||
|
} |
||||
|
|
||||
|
encoder->width = width; |
||||
|
encoder->height = height; |
||||
|
*result = encoder; |
||||
|
return AOM_CODEC_OK; |
||||
|
} |
||||
|
|
||||
|
// Borrows the already converted planes for this synchronous encode call. The caller pins all three
|
||||
|
// pointers until it returns; libaom retains its own reference pictures, never these managed input pointers.
|
||||
|
BENCHMARK_API int benchmark_encode(benchmark_encoder *encoder, unsigned char *y, |
||||
|
unsigned char *u, unsigned char *v, |
||||
|
int y_stride, int uv_stride, int64_t frame_index) { |
||||
|
aom_image_t image; |
||||
|
if (aom_img_wrap(&image, AOM_IMG_FMT_I420, encoder->width, encoder->height, 1, y) == NULL) { |
||||
|
return AOM_CODEC_INVALID_PARAM; |
||||
|
} |
||||
|
|
||||
|
// ImageSharp's source planes include aligned borders. Describe those existing rows directly instead
|
||||
|
// of flattening them into another contiguous YUV allocation before calling the reference encoder.
|
||||
|
image.planes[AOM_PLANE_Y] = y; |
||||
|
image.planes[AOM_PLANE_U] = u; |
||||
|
image.planes[AOM_PLANE_V] = v; |
||||
|
image.stride[AOM_PLANE_Y] = y_stride; |
||||
|
image.stride[AOM_PLANE_U] = uv_stride; |
||||
|
image.stride[AOM_PLANE_V] = uv_stride; |
||||
|
encoder->iterator = NULL; |
||||
|
return aom_codec_encode(&encoder->codec, &image, frame_index, 1, 0); |
||||
|
} |
||||
|
|
||||
|
// Drains delayed output at the end of the sequence even though lookahead is disabled.
|
||||
|
BENCHMARK_API int benchmark_flush(benchmark_encoder *encoder) { |
||||
|
encoder->iterator = NULL; |
||||
|
return aom_codec_encode(&encoder->codec, NULL, -1, 1, 0); |
||||
|
} |
||||
|
|
||||
|
// The returned packet storage belongs to libaom and is valid only until the next codec call.
|
||||
|
// The managed side copies it directly into the same kind of destination stream as its own encoder.
|
||||
|
BENCHMARK_API int benchmark_next_packet(benchmark_encoder *encoder, const void **data, size_t *length) { |
||||
|
const aom_codec_cx_pkt_t *packet; |
||||
|
while ((packet = aom_codec_get_cx_data(&encoder->codec, &encoder->iterator)) != NULL) { |
||||
|
if (packet->kind == AOM_CODEC_CX_FRAME_PKT) { |
||||
|
*data = packet->data.frame.buf; |
||||
|
*length = packet->data.frame.sz; |
||||
|
return 1; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
*data = NULL; |
||||
|
*length = 0; |
||||
|
return 0; |
||||
|
} |
||||
|
|
||||
|
// Releases exactly the native context allocated by benchmark_create.
|
||||
|
BENCHMARK_API int benchmark_destroy(benchmark_encoder *encoder) { |
||||
|
aom_codec_err_t status = aom_codec_destroy(&encoder->codec); |
||||
|
free(encoder); |
||||
|
return status; |
||||
|
} |
||||
|
|
||||
|
// Libaom owns this static UTF-8 error string; the caller must not free it.
|
||||
|
BENCHMARK_API const char *benchmark_error_string(int status) { |
||||
|
return aom_codec_err_to_string((aom_codec_err_t)status); |
||||
|
} |
||||
@ -0,0 +1,137 @@ |
|||||
|
# AV1 sequence encoder comparison |
||||
|
|
||||
|
`Av1SequenceEncoderBenchmarks` compares the managed AV1 sequence encoder with an optimized build of official libaom `main`. |
||||
|
The native adapter belongs only to this benchmark project. ImageSharp production code remains fully managed. |
||||
|
|
||||
|
## Measurement boundary |
||||
|
|
||||
|
Both methods start with the same three `Rgb24` photographic frames and finish with the complete raw AV1 OBU sequence in a reused `MemoryStream`. |
||||
|
The measured operation includes encoder/workspace construction, RGB-to-YUV conversion for every frame, encoding, output writes, flushing where required, and disposal. |
||||
|
Both use the existing ImageSharp SIMD color converter and write directly into the aligned source planes consumed by their encoder. |
||||
|
Libaom receives pinned plane pointers for each synchronous encode call; it does not read a preconverted input file. |
||||
|
|
||||
|
Image loading, resizing, half-pixel source translations, validation-file writes, decoding, and quality calculations are outside both measured methods. |
||||
|
MemoryStream capacity is retained between operations for both methods. These are RGB-to-OBU measurements, not public HEIF container-save measurements. |
||||
|
Do not compare them with `aomenc`'s internal encode-time report, which excludes the conversion boundary. |
||||
|
|
||||
|
The source is the existing `TestImages.Png.Bike` photograph. Frames one and two are independently translated by half a pixel and one pixel on both axes. |
||||
|
The parameter matrix contains 256x256 and 512x512 frames at ImageSharp efforts seven, eight, and nine. |
||||
|
Input is full-range BT.601, eight-bit 4:2:0; each sequence contains one key frame and two dependent frames. |
||||
|
Both codecs use one coding thread. Libaom uses good-quality mode, no lookahead, disabled automatic key frames, and `cpu-used=6`. |
||||
|
That speed is an independent reference setting, not a mapping from the ImageSharp effort scale. |
||||
|
|
||||
|
ImageSharp's base quantizer index is 120. Libaom's public quantizer 30 maps to that same index in the pinned reference. |
||||
|
The native minimum and maximum quantizers are both fixed to 30, in addition to the CQ setting, so frame-level CQ boosts cannot lower the base quantizer. |
||||
|
Other native coding tools remain enabled. Identical quantizers do not imply identical quality or bitrate; always retain size and decoded-quality results with timings. |
||||
|
|
||||
|
BenchmarkDotNet's allocation column measures managed allocations only. It does not measure libaom's native allocations, pooled native memory, or total peak memory. |
||||
|
Do not interpret its allocation ratio as a whole-encoder memory comparison. |
||||
|
|
||||
|
## Build the reference and benchmark adapter |
||||
|
|
||||
|
The verified source revision is `d565eec60f084421fa34fc0534b760c6452b6a6c`, libaom 3.15.0, exported at `D:\GitHub\ynse01\aom-d565eec6-source`. |
||||
|
The x64 Release build is `D:\GitHub\ynse01\aom-d565eec6-build-x64-release`. |
||||
|
It uses MSVC 19.51.36256.0 and NASM 3.02, runtime CPU detection, and SSE2, SSE4.1, AVX2, and AVX512 kernels. |
||||
|
The generic reference decoder build is suitable for conformance but must not be used as the timing reference. |
||||
|
|
||||
|
Run from the repository root in an x64 Visual Studio developer PowerShell with the existing CMake, Ninja, NASM, and Perl installations available: |
||||
|
|
||||
|
```powershell |
||||
|
$aomSourceDirectory = 'D:/GitHub/ynse01/aom-d565eec6-source' |
||||
|
$aomBuildDirectory = 'D:/GitHub/ynse01/aom-d565eec6-build-x64-release' |
||||
|
$adapterBuildDirectory = 'artifacts/av1-native-benchmark-release' |
||||
|
|
||||
|
cmake -S $aomSourceDirectory -B $aomBuildDirectory -G Ninja ` |
||||
|
-DCMAKE_BUILD_TYPE=Release -DAOM_TARGET_CPU=x86_64 -DENABLE_NASM=ON ` |
||||
|
-DENABLE_TESTS=OFF -DENABLE_EXAMPLES=ON -DENABLE_TOOLS=ON ` |
||||
|
-DCONFIG_WEBM_IO=OFF -DCONFIG_LIBYUV=OFF |
||||
|
if ($LASTEXITCODE -ne 0) { throw 'Reference configuration failed.' } |
||||
|
|
||||
|
cmake --build $aomBuildDirectory --target aomenc aomdec --parallel 4 |
||||
|
if ($LASTEXITCODE -ne 0) { throw 'Reference build failed.' } |
||||
|
|
||||
|
cmake -S tests/ImageSharp.Benchmarks/Codecs/Heif/Native -B $adapterBuildDirectory -G Ninja ` |
||||
|
-DCMAKE_BUILD_TYPE=Release "-DAOM_SOURCE_DIRECTORY=$aomSourceDirectory" "-DAOM_BUILD_DIRECTORY=$aomBuildDirectory" |
||||
|
if ($LASTEXITCODE -ne 0) { throw 'Adapter configuration failed.' } |
||||
|
|
||||
|
cmake --build $adapterBuildDirectory --parallel 1 |
||||
|
if ($LASTEXITCODE -ne 0) { throw 'Adapter build failed.' } |
||||
|
|
||||
|
dotnet build tests/ImageSharp.Benchmarks/ImageSharp.Benchmarks.csproj -c Release -f net11.0 ` |
||||
|
--no-restore --disable-build-servers -m:1 -p:UseSharedCompilation=false ` |
||||
|
-p:SIXLABORS_TESTING_PREVIEW=true -p:SIXLABORS_DISABLE_CONFIG_COPY=true |
||||
|
if ($LASTEXITCODE -ne 0) { throw 'Managed build failed.' } |
||||
|
|
||||
|
Copy-Item -LiteralPath "$adapterBuildDirectory/imagesharp_aom_benchmark.dll" ` |
||||
|
-Destination artifacts/bin/tests/ImageSharp.Benchmarks/Release/net11.0/imagesharp_aom_benchmark.dll |
||||
|
``` |
||||
|
|
||||
|
The adapter links the explicit reference build. It neither downloads tools nor adds a native production dependency. |
||||
|
Rebuild it whenever the source headers or reference library change; record the exact source revision and optimized build configuration with every report. |
||||
|
|
||||
|
## Validate, then measure |
||||
|
|
||||
|
Run one pair first, in the existing benchmark host with the in-process toolchain. Do not run builds or tests concurrently with timing. |
||||
|
|
||||
|
```powershell |
||||
|
$benchmarkAssembly = 'artifacts/bin/tests/ImageSharp.Benchmarks/Release/net11.0/ImageSharp.Benchmarks.dll' |
||||
|
$benchmarkFilter = '*Av1SequenceEncoderBenchmarks.*(Dimension: 256, Effort: 7)' |
||||
|
|
||||
|
dotnet $benchmarkAssembly --inProcess --job Dry --filter $benchmarkFilter ` |
||||
|
--stopOnFirstError --noOverwrite --artifacts artifacts/BenchmarkDotNet/av1-sequence-rgb-dry |
||||
|
``` |
||||
|
|
||||
|
The actual last measured payload for each method and the independently exported source planes are retained under |
||||
|
`tests/Images/ActualOutput/Heif/Av1/Av1SequenceEncoderBenchmarks`. |
||||
|
Both timed methods convert RGB on every invocation; the exported source file exists only for offline quality calculation. |
||||
|
Decode both payloads using the matching current-main `aomdec`, and require the expected complete three-frame YUV length: |
||||
|
|
||||
|
```powershell |
||||
|
$outputDirectory = 'tests/Images/ActualOutput/Heif/Av1/Av1SequenceEncoderBenchmarks' |
||||
|
foreach ($payloadName in 'bike-256-q120-effort7.obu', 'bike-256-q120-libaom-cpu6.obu') { |
||||
|
$payloadPath = Join-Path $outputDirectory $payloadName |
||||
|
& "$aomBuildDirectory/aomdec.exe" --codec=av1 --threads=1 --rawvideo -o "$payloadPath.yuv" $payloadPath |
||||
|
if ($LASTEXITCODE -ne 0) { throw "Reference decode failed: $payloadName" } |
||||
|
|
||||
|
if ((Get-Item -LiteralPath "$payloadPath.yuv").Length -ne (3 * 256 * 256 * 3 / 2)) { |
||||
|
throw "Incomplete decoded sequence: $payloadName" |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
dotnet $benchmarkAssembly --inProcess --job Short --filter $benchmarkFilter ` |
||||
|
--stopOnFirstError --noOverwrite --artifacts artifacts/BenchmarkDotNet/av1-sequence-rgb-short |
||||
|
``` |
||||
|
|
||||
|
Repeat output validation after the measured run. Compute per-plane PSNR against the exported source using |
||||
|
`10 * log10(255^2 * sampleCount / squaredErrorSum)`, accumulated over all three frames. |
||||
|
Aggregate YUV PSNR uses the total squared error and sample count, not the arithmetic mean of the three PSNR values. |
||||
|
Keep bitstream length, quality, absolute time, runtime, CPU information, reference settings, and any environment warnings together. |
||||
|
The short job is an initial diagnostic checkpoint, not a substitute for the complete size/quality/performance matrix or an equal-quality rate-distortion comparison. |
||||
|
|
||||
|
## Verified checkpoint: 2026-09-05 |
||||
|
|
||||
|
The final fixed-quantizer pair ran with BenchmarkDotNet 0.15.8, .NET 11.0.0-preview.7.26381.103, x64 RyuJIT, and AVX512 available. |
||||
|
The in-process Short job used three warmup iterations and three measured iterations. Each operation encoded all three 256x256 frames. |
||||
|
|
||||
|
| Encoder setting | Mean sequence time | OBU size | Y PSNR | U PSNR | V PSNR | Aggregate YUV PSNR | |
||||
|
| --- | ---: | ---: | ---: | ---: | ---: | ---: | |
||||
|
| ImageSharp effort 7, base index 120 | 2,053.31 ms | 11.54 KiB | 39.040 dB | 38.630 dB | 32.778 dB | 37.124 dB | |
||||
|
| Libaom cpu-used 6, fixed public quantizer 30 | 70.67 ms | 8.27 KiB | 39.275 dB | 39.637 dB | 37.351 dB | 38.942 dB | |
||||
|
|
||||
|
Both final measured payloads decode to exactly three complete native YUV frames using the optimized current-main reference decoder. |
||||
|
ImageSharp is about 29 times slower for this case while producing a larger, lower-PSNR sequence. The performance exit gate remains open. |
||||
|
The managed allocation counter reports 9.38 MiB per ImageSharp operation; its source still needs attribution, and it is not comparable with libaom's unmeasured native footprint. |
||||
|
|
||||
|
Evidence is retained in `artifacts/BenchmarkDotNet/av1-sequence-rgb-fixed-q-short-20260905/20260905-131149`. |
||||
|
The initial Dry run and earlier unrestricted-CQ Short run are preliminary checks, not the final comparison above. |
||||
|
BenchmarkDotNet could not change the power plan or query CPU model information in this environment; those warnings remain in the log. |
||||
|
This result is a local diagnostic, not a controlled-hardware or equal-quality performance claim. |
||||
|
|
||||
|
Final OBU SHA-256 values: |
||||
|
|
||||
|
- ImageSharp: `AB0327D33405FB7A6EC1895D465C9EE08E0B218DDA9D7E8EE7E0DCA6C5F8A73B` |
||||
|
- Libaom: `0C44A3475849762B5A09937397DEA4397264B83E79DAF930A06B70734977E4A9` |
||||
|
|
||||
|
The final Release benchmark build has zero errors and 39 existing benchmark warnings outside the new files. |
||||
|
Roslyn compiler and analyzer diagnostics contain no errors or warnings in the new benchmark files. |
||||
|
No production file changed in this benchmark checkpoint; the preceding 2,524-case encoder/entropy verification remains the latest production test run. |
||||
Loading…
Reference in new issue