mirror of https://github.com/SixLabors/ImageSharp
6 changed files with 618 additions and 0 deletions
@ -0,0 +1,201 @@ |
|||
// Copyright (c) Six Labors.
|
|||
// Licensed under the Six Labors Split License.
|
|||
|
|||
using System.Numerics; |
|||
using BenchmarkDotNet.Attributes; |
|||
using SixLabors.ImageSharp.Formats.Heif.Av1; |
|||
using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; |
|||
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; |
|||
using SixLabors.ImageSharp.Formats.Heif.Components; |
|||
using SixLabors.ImageSharp.Memory; |
|||
using SixLabors.ImageSharp.PixelFormats; |
|||
using SixLabors.ImageSharp.Processing; |
|||
using SixLabors.ImageSharp.Tests; |
|||
|
|||
namespace SixLabors.ImageSharp.Benchmarks.Codecs.Heif; |
|||
|
|||
/// <summary>
|
|||
/// Measures encoding a photographic sequence with fractional motion at the interpolation-search effort boundaries.
|
|||
/// </summary>
|
|||
[MemoryDiagnoser] |
|||
public class Av1SequenceEncoderBenchmarks |
|||
{ |
|||
/// <summary>
|
|||
/// The number of displayed pictures in each independently encoded sequence.
|
|||
/// </summary>
|
|||
private const int FrameCount = 3; |
|||
|
|||
/// <summary>
|
|||
/// The native AV1 quantizer index corresponding to libaom's public constant-quality level 30.
|
|||
/// </summary>
|
|||
private const int QIndex = 120; |
|||
|
|||
/// <summary>
|
|||
/// The fixed native speed baseline, independent of ImageSharp's effort scale.
|
|||
/// </summary>
|
|||
private const int NativeCpuUsed = 6; |
|||
|
|||
/// <summary>
|
|||
/// The native public quantizer corresponding to <see cref="QIndex"/>, also used for both rate-control bounds.
|
|||
/// </summary>
|
|||
private const int NativeQuality = 30; |
|||
|
|||
private Image<Rgb24> sequence; |
|||
private Configuration configuration; |
|||
private ObuColorConfig colorConfig; |
|||
private MemoryStream output; |
|||
private string outputDirectory; |
|||
|
|||
/// <summary>
|
|||
/// Gets or sets the square frame dimension.
|
|||
/// </summary>
|
|||
[Params(256, 512)] |
|||
public int Dimension { get; set; } |
|||
|
|||
/// <summary>
|
|||
/// Gets or sets the effort controlling fixed, common switchable, or independently switchable filters.
|
|||
/// </summary>
|
|||
[Params(7, 8, 9)] |
|||
public int Effort { get; set; } |
|||
|
|||
/// <summary>
|
|||
/// Prepares identical photographic RGB frames and a planar source file for checking reconstructed output quality.
|
|||
/// </summary>
|
|||
[GlobalSetup] |
|||
public void Setup() |
|||
{ |
|||
this.configuration = Configuration.Default.Clone(); |
|||
this.configuration.MaxDegreeOfParallelism = 1; |
|||
this.colorConfig = new ObuColorConfig |
|||
{ |
|||
BitDepth = Av1BitDepth.EightBit, |
|||
IsColorDescriptionPresent = true, |
|||
ColorPrimaries = ObuColorPrimaries.Bt601, |
|||
TransferCharacteristics = ObuTransferCharacteristics.Bt601, |
|||
MatrixCoefficients = ObuMatrixCoefficients.Bt601, |
|||
ColorRange = true, |
|||
SubSamplingX = true, |
|||
SubSamplingY = true, |
|||
ChromaSamplePosition = ObuChromoSamplePosition.Unknown |
|||
}; |
|||
|
|||
this.output = new MemoryStream(); |
|||
this.outputDirectory = TestEnvironment.CreateOutputDirectory("Heif", "Av1", nameof(Av1SequenceEncoderBenchmarks)); |
|||
string inputPath = Path.Combine(TestEnvironment.InputImagesDirectoryFullPath, TestImages.Png.Bike); |
|||
using Image<Rgb24> photograph = Image.Load<Rgb24>(inputPath); |
|||
|
|||
// Leave a source margin for the half-pixel translations. Resampling is setup work, not encoder time;
|
|||
// every invocation consumes the same three images rather than repeatedly translating a previous result.
|
|||
photograph.Mutate(context => context.Resize(new ResizeOptions |
|||
{ |
|||
Size = new Size(this.Dimension + FrameCount, this.Dimension + FrameCount), |
|||
Mode = ResizeMode.Crop |
|||
})); |
|||
|
|||
Rectangle sourceBounds = new(0, 0, photograph.Width, photograph.Height); |
|||
Size targetSize = new(this.Dimension, this.Dimension); |
|||
this.sequence = photograph.Clone(context => context.Crop(new Rectangle(Point.Empty, targetSize))); |
|||
for (int frameIndex = 1; frameIndex < FrameCount; frameIndex++) |
|||
{ |
|||
Matrix3x2 translation = Matrix3x2.CreateTranslation(-0.5F * frameIndex, -0.5F * frameIndex); |
|||
using Image<Rgb24> translated = photograph.Clone(context => |
|||
context.Transform(sourceBounds, translation, targetSize, KnownResamplers.Bicubic)); |
|||
|
|||
this.sequence.Frames.AddFrame(translated.Frames.RootFrame); |
|||
} |
|||
|
|||
// Export the production-converted source planes only for checking reconstructed output quality.
|
|||
// Neither timed encoder reads this file: both convert the original RGB frames during each operation.
|
|||
using Av1EncoderFrameBuffer<byte> planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0); |
|||
using FileStream raw = File.Create(Path.Combine(this.outputDirectory, $"bike-{this.Dimension}-3frames.source.yuv")); |
|||
foreach (ImageFrame<Rgb24> frame in this.sequence.Frames) |
|||
{ |
|||
Av1FrameEncoder.PrepareSource(this.configuration, frame, planar.Frame, this.colorConfig); |
|||
for (int planeIndex = 0; planeIndex < this.colorConfig.PlaneCount; planeIndex++) |
|||
{ |
|||
Buffer2DRegion<byte> plane = planar.Frame.View.GetPlane((Av1Plane)planeIndex); |
|||
for (int y = 0; y < plane.Height; y++) |
|||
{ |
|||
raw.Write(plane.DangerousGetRowSpan(y)); |
|||
} |
|||
} |
|||
} |
|||
} |
|||
|
|||
/// <summary>
|
|||
/// Encodes one key picture and two dependent pictures, returning the complete OBU payload length.
|
|||
/// </summary>
|
|||
/// <returns>The encoded sequence length.</returns>
|
|||
[Benchmark] |
|||
public long ImageSharp() |
|||
{ |
|||
this.output.SetLength(0); |
|||
using Av1FrameEncoder.SequenceEncoder encoder = Av1FrameEncoder.CreateColorSequenceEncoder( |
|||
this.configuration, this.Dimension, this.Dimension, this.colorConfig, QIndex, this.Effort); |
|||
|
|||
// One operation owns the real sequence lifetime: allocation, conversion, key/inter coding, and disposal.
|
|||
// The caller's destination is reused, excluding filesystem and MemoryStream growth from steady-state timing.
|
|||
encoder.EncodeKeyFrame(this.sequence.Frames.RootFrame, this.output); |
|||
for (int frameIndex = 1; frameIndex < FrameCount; frameIndex++) |
|||
{ |
|||
encoder.EncodeInterFrame(this.sequence.Frames[frameIndex], this.output); |
|||
} |
|||
|
|||
return this.output.Length; |
|||
} |
|||
|
|||
/// <summary>
|
|||
/// Encodes the same RGB sequence with current-main libaom, including conversion, allocation, output, and disposal.
|
|||
/// </summary>
|
|||
/// <returns>The encoded sequence length.</returns>
|
|||
[Benchmark(Baseline = true)] |
|||
public long Libaom() |
|||
{ |
|||
// cpu-used is a separate speed scale, not an ImageSharp effort mapping. Keep the reference at
|
|||
// good-quality speed six while comparing the three managed interpolation-search boundaries.
|
|||
this.output.SetLength(0); |
|||
using LibaomBenchmarkEncoder encoder = LibaomBenchmarkEncoder.Open(this.Dimension, this.Dimension, NativeQuality, NativeCpuUsed); |
|||
using Av1EncoderFrameBuffer<byte> planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0); |
|||
using Av1FrameEncoder.Av1EncoderConversionWorkspace conversion = new(this.configuration, this.Dimension, this.colorConfig, false, false); |
|||
Rectangle bounds = new(0, 0, this.Dimension, this.Dimension); |
|||
for (int frameIndex = 0; frameIndex < FrameCount; frameIndex++) |
|||
{ |
|||
// Conversion belongs inside both measured paths. Reuse the same row workspace and SIMD converter
|
|||
// as the managed sequence encoder, writing directly into the planes passed to native libaom.
|
|||
conversion.Convert<Rgb24, Av1EncoderFrame<byte>.PlanarView, byte, HeifByteSampleConverter>( |
|||
this.configuration, this.sequence.Frames[frameIndex], bounds, planar.Frame.View); |
|||
|
|||
encoder.Encode(planar.Frame, frameIndex, this.output); |
|||
} |
|||
|
|||
encoder.Finish(this.output); |
|||
return this.output.Length; |
|||
} |
|||
|
|||
/// <summary>
|
|||
/// Retains the measured managed encoder output and releases the input images and destination stream.
|
|||
/// </summary>
|
|||
[GlobalCleanup(Target = nameof(ImageSharp))] |
|||
public void CleanupImageSharp() => this.Cleanup($"bike-{this.Dimension}-q{QIndex}-effort{this.Effort}.obu"); |
|||
|
|||
/// <summary>
|
|||
/// Retains the measured reference encoder output and releases the input images and destination stream.
|
|||
/// </summary>
|
|||
[GlobalCleanup(Target = nameof(Libaom))] |
|||
public void CleanupLibaom() => this.Cleanup($"bike-{this.Dimension}-q{QIndex}-libaom-cpu{NativeCpuUsed}.obu"); |
|||
|
|||
/// <summary>
|
|||
/// Writes the measured payload without another encoding pass and releases the shared benchmark resources.
|
|||
/// </summary>
|
|||
/// <param name="outputName">The codec-specific payload file name.</param>
|
|||
private void Cleanup(string outputName) |
|||
{ |
|||
// Output validation and quality measurement use the actual measured payload, with no encode or decode
|
|||
// hidden inside the timed operation and no file-sized ToArray copy.
|
|||
using FileStream encoded = File.Create(Path.Combine(this.outputDirectory, outputName)); |
|||
this.output.Position = 0; |
|||
this.output.CopyTo(encoded); |
|||
this.output.Dispose(); |
|||
this.sequence.Dispose(); |
|||
} |
|||
} |
|||
@ -0,0 +1,135 @@ |
|||
// Copyright (c) Six Labors.
|
|||
// Licensed under the Six Labors Split License.
|
|||
|
|||
using System.Runtime.InteropServices; |
|||
using Microsoft.Win32.SafeHandles; |
|||
using SixLabors.ImageSharp.Formats.Heif.Av1; |
|||
using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; |
|||
using SixLabors.ImageSharp.Memory; |
|||
|
|||
namespace SixLabors.ImageSharp.Benchmarks.Codecs.Heif; |
|||
|
|||
/// <summary>
|
|||
/// Owns a benchmark-only libaom encoder through a C adapter compiled against the reference's actual headers.
|
|||
/// </summary>
|
|||
internal sealed unsafe partial class LibaomBenchmarkEncoder : SafeHandleZeroOrMinusOneIsInvalid |
|||
{ |
|||
/// <summary>
|
|||
/// The benchmark adapter's platform-independent library name.
|
|||
/// </summary>
|
|||
private const string LibraryName = "imagesharp_aom_benchmark"; |
|||
|
|||
/// <summary>
|
|||
/// Initializes an empty handle for the native create call's generated marshaller.
|
|||
/// </summary>
|
|||
public LibaomBenchmarkEncoder() |
|||
: base(ownsHandle: true) |
|||
{ |
|||
} |
|||
|
|||
/// <summary>
|
|||
/// Opens one native sequence with a single coding thread, no lookahead, and the requested quality and speed.
|
|||
/// </summary>
|
|||
public static LibaomBenchmarkEncoder Open(int width, int height, int quality, int speed) |
|||
{ |
|||
CheckStatus(Create((uint)width, (uint)height, (uint)quality, speed, out LibaomBenchmarkEncoder encoder)); |
|||
return encoder; |
|||
} |
|||
|
|||
/// <summary>
|
|||
/// Encodes the converted source planes and consumes every output packet before those bytes can be invalidated.
|
|||
/// </summary>
|
|||
public void Encode(Av1EncoderFrame<byte> frame, long frameIndex, Stream output) |
|||
{ |
|||
Buffer2DRegion<byte> y = frame.View.GetPlane(Av1Plane.Y); |
|||
Buffer2DRegion<byte> u = frame.View.GetPlane(Av1Plane.U); |
|||
Buffer2DRegion<byte> v = frame.View.GetPlane(Av1Plane.V); |
|||
fixed (byte* yPointer = y.DangerousGetRowSpan(0), uPointer = u.DangerousGetRowSpan(0), vPointer = v.DangerousGetRowSpan(0)) |
|||
{ |
|||
// The native call is synchronous. Its input descriptors borrow these pinned rows only until
|
|||
// EncodeFrame returns; the encoder owns any retained reference and lookahead storage itself.
|
|||
CheckStatus(EncodeFrame(this, yPointer, uPointer, vPointer, y.Stride, u.Stride, frameIndex)); |
|||
} |
|||
|
|||
this.WritePackets(output); |
|||
} |
|||
|
|||
/// <summary>
|
|||
/// Finishes the sequence and writes any remaining coded packets.
|
|||
/// </summary>
|
|||
public void Finish(Stream output) |
|||
{ |
|||
do |
|||
{ |
|||
CheckStatus(Flush(this)); |
|||
} |
|||
while (this.WritePackets(output)); |
|||
} |
|||
|
|||
/// <inheritdoc/>
|
|||
protected override bool ReleaseHandle() => Destroy(this.handle) == 0; |
|||
|
|||
/// <summary>
|
|||
/// Writes borrowed packet memory before the next call into the native codec invalidates it.
|
|||
/// </summary>
|
|||
private bool WritePackets(Stream output) |
|||
{ |
|||
bool wrotePacket = false; |
|||
while (NextPacket(this, out byte* data, out nuint length) != 0) |
|||
{ |
|||
// Packet lengths are bounded by the benchmark's image dimensions. Stream.Write consumes the
|
|||
// borrowed bytes synchronously, without retaining a native pointer or creating a managed array.
|
|||
output.Write(new ReadOnlySpan<byte>(data, (int)length)); |
|||
wrotePacket = true; |
|||
} |
|||
|
|||
return wrotePacket; |
|||
} |
|||
|
|||
/// <summary>
|
|||
/// Converts a native codec error into a managed benchmark failure instead of accepting invalid timing data.
|
|||
/// </summary>
|
|||
private static void CheckStatus(int status) |
|||
{ |
|||
if (status != 0) |
|||
{ |
|||
throw new InvalidOperationException(Marshal.PtrToStringUTF8(ErrorString(status))); |
|||
} |
|||
} |
|||
|
|||
/// <summary>
|
|||
/// Creates an owned opaque context; no libaom structure layout crosses the managed boundary.
|
|||
/// </summary>
|
|||
[LibraryImport(LibraryName, EntryPoint = "benchmark_create")] |
|||
private static partial int Create(uint width, uint height, uint quality, int speed, out LibaomBenchmarkEncoder encoder); |
|||
|
|||
/// <summary>
|
|||
/// Borrows three pinned planes for one synchronous native encode call.
|
|||
/// </summary>
|
|||
[LibraryImport(LibraryName, EntryPoint = "benchmark_encode")] |
|||
private static partial int EncodeFrame(LibaomBenchmarkEncoder encoder, byte* y, byte* u, byte* v, int yStride, int uvStride, long frameIndex); |
|||
|
|||
/// <summary>
|
|||
/// Signals the end of the native sequence.
|
|||
/// </summary>
|
|||
[LibraryImport(LibraryName, EntryPoint = "benchmark_flush")] |
|||
private static partial int Flush(LibaomBenchmarkEncoder encoder); |
|||
|
|||
/// <summary>
|
|||
/// Returns borrowed native packet storage and its pointer-sized length.
|
|||
/// </summary>
|
|||
[LibraryImport(LibraryName, EntryPoint = "benchmark_next_packet")] |
|||
private static partial int NextPacket(LibaomBenchmarkEncoder encoder, out byte* data, out nuint length); |
|||
|
|||
/// <summary>
|
|||
/// Destroys the context through the same native library that allocated it.
|
|||
/// </summary>
|
|||
[LibraryImport(LibraryName, EntryPoint = "benchmark_destroy")] |
|||
private static partial int Destroy(nint encoder); |
|||
|
|||
/// <summary>
|
|||
/// Returns a static UTF-8 error message owned by libaom.
|
|||
/// </summary>
|
|||
[LibraryImport(LibraryName, EntryPoint = "benchmark_error_string")] |
|||
private static partial nint ErrorString(int status); |
|||
} |
|||
@ -0,0 +1,12 @@ |
|||
cmake_minimum_required(VERSION 3.20) |
|||
project(imagesharp_aom_benchmark LANGUAGES C CXX) |
|||
|
|||
# Link the explicitly supplied current-main Release reference build. This adapter is benchmark-only; |
|||
# it is never referenced by ImageSharp or copied by the production project. |
|||
set(AOM_SOURCE_DIRECTORY "" CACHE PATH "Verified official libaom source directory") |
|||
set(AOM_BUILD_DIRECTORY "" CACHE PATH "Matching optimized Release libaom build directory") |
|||
add_library(imagesharp_aom_benchmark SHARED aom_benchmark.c) |
|||
target_include_directories(imagesharp_aom_benchmark PRIVATE "${AOM_SOURCE_DIRECTORY}") |
|||
find_library(AOM_LIBRARY NAMES aom PATHS "${AOM_BUILD_DIRECTORY}" NO_DEFAULT_PATH REQUIRED) |
|||
target_link_libraries(imagesharp_aom_benchmark PRIVATE "${AOM_LIBRARY}") |
|||
set_target_properties(imagesharp_aom_benchmark PROPERTIES C_STANDARD 11 LINKER_LANGUAGE CXX) |
|||
@ -0,0 +1,131 @@ |
|||
// Copyright (c) Six Labors.
|
|||
// Licensed under the Six Labors Split License.
|
|||
|
|||
#include <stdint.h> |
|||
#include <stdlib.h> |
|||
#include "aom/aom_encoder.h" |
|||
#include "aom/aomcx.h" |
|||
|
|||
#if defined(_WIN32) |
|||
#define BENCHMARK_API __declspec(dllexport) |
|||
#else |
|||
#define BENCHMARK_API __attribute__((visibility("default"))) |
|||
#endif |
|||
|
|||
// Keep libaom's version-dependent structures and variadic controls entirely on the C side.
|
|||
// The managed benchmark exchanges only opaque ownership, fixed-width integers, and borrowed buffers.
|
|||
typedef struct benchmark_encoder { |
|||
aom_codec_ctx_t codec; |
|||
aom_codec_iter_t iterator; |
|||
unsigned int width; |
|||
unsigned int height; |
|||
} benchmark_encoder; |
|||
|
|||
// Creates one single-threaded, unlagged, constant-quality sequence using the reference's normal tools.
|
|||
// A successful result belongs to the caller and must be released by benchmark_destroy.
|
|||
BENCHMARK_API int benchmark_create(unsigned int width, unsigned int height, |
|||
unsigned int quality, int speed, |
|||
benchmark_encoder **result) { |
|||
aom_codec_enc_cfg_t config; |
|||
aom_codec_iface_t *iface = aom_codec_av1_cx(); |
|||
aom_codec_err_t status = aom_codec_enc_config_default(iface, &config, AOM_USAGE_GOOD_QUALITY); |
|||
*result = NULL; |
|||
if (status != AOM_CODEC_OK) return status; |
|||
|
|||
benchmark_encoder *encoder = calloc(1, sizeof(*encoder)); |
|||
if (encoder == NULL) return AOM_CODEC_MEM_ERROR; |
|||
|
|||
config.g_w = width; |
|||
config.g_h = height; |
|||
config.g_threads = 1; |
|||
config.g_lag_in_frames = 0; |
|||
config.g_timebase.num = 1; |
|||
config.g_timebase.den = 30; |
|||
config.rc_end_usage = AOM_Q; |
|||
// Fix both bounds to the requested quantizer. CQ alone permits frame-quality boosts, whereas
|
|||
// the managed sequence uses this same base quantizer for every frame in the comparison.
|
|||
config.rc_min_quantizer = quality; |
|||
config.rc_max_quantizer = quality; |
|||
config.kf_mode = AOM_KF_DISABLED; |
|||
status = aom_codec_enc_init(&encoder->codec, iface, &config, 0); |
|||
if (status != AOM_CODEC_OK) { |
|||
free(encoder); |
|||
return status; |
|||
} |
|||
|
|||
// These calls remain type-checked against the current libaom headers, including each control's argument type.
|
|||
status = aom_codec_control(&encoder->codec, AOME_SET_CPUUSED, speed); |
|||
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AOME_SET_CQ_LEVEL, quality); |
|||
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_ROW_MT, 0u); |
|||
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_COLOR_PRIMARIES, AOM_CICP_CP_BT_601); |
|||
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_TRANSFER_CHARACTERISTICS, AOM_CICP_TC_BT_601); |
|||
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_MATRIX_COEFFICIENTS, AOM_CICP_MC_BT_601); |
|||
if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_COLOR_RANGE, AOM_CR_FULL_RANGE); |
|||
if (status != AOM_CODEC_OK) { |
|||
aom_codec_destroy(&encoder->codec); |
|||
free(encoder); |
|||
return status; |
|||
} |
|||
|
|||
encoder->width = width; |
|||
encoder->height = height; |
|||
*result = encoder; |
|||
return AOM_CODEC_OK; |
|||
} |
|||
|
|||
// Borrows the already converted planes for this synchronous encode call. The caller pins all three
|
|||
// pointers until it returns; libaom retains its own reference pictures, never these managed input pointers.
|
|||
BENCHMARK_API int benchmark_encode(benchmark_encoder *encoder, unsigned char *y, |
|||
unsigned char *u, unsigned char *v, |
|||
int y_stride, int uv_stride, int64_t frame_index) { |
|||
aom_image_t image; |
|||
if (aom_img_wrap(&image, AOM_IMG_FMT_I420, encoder->width, encoder->height, 1, y) == NULL) { |
|||
return AOM_CODEC_INVALID_PARAM; |
|||
} |
|||
|
|||
// ImageSharp's source planes include aligned borders. Describe those existing rows directly instead
|
|||
// of flattening them into another contiguous YUV allocation before calling the reference encoder.
|
|||
image.planes[AOM_PLANE_Y] = y; |
|||
image.planes[AOM_PLANE_U] = u; |
|||
image.planes[AOM_PLANE_V] = v; |
|||
image.stride[AOM_PLANE_Y] = y_stride; |
|||
image.stride[AOM_PLANE_U] = uv_stride; |
|||
image.stride[AOM_PLANE_V] = uv_stride; |
|||
encoder->iterator = NULL; |
|||
return aom_codec_encode(&encoder->codec, &image, frame_index, 1, 0); |
|||
} |
|||
|
|||
// Drains delayed output at the end of the sequence even though lookahead is disabled.
|
|||
BENCHMARK_API int benchmark_flush(benchmark_encoder *encoder) { |
|||
encoder->iterator = NULL; |
|||
return aom_codec_encode(&encoder->codec, NULL, -1, 1, 0); |
|||
} |
|||
|
|||
// The returned packet storage belongs to libaom and is valid only until the next codec call.
|
|||
// The managed side copies it directly into the same kind of destination stream as its own encoder.
|
|||
BENCHMARK_API int benchmark_next_packet(benchmark_encoder *encoder, const void **data, size_t *length) { |
|||
const aom_codec_cx_pkt_t *packet; |
|||
while ((packet = aom_codec_get_cx_data(&encoder->codec, &encoder->iterator)) != NULL) { |
|||
if (packet->kind == AOM_CODEC_CX_FRAME_PKT) { |
|||
*data = packet->data.frame.buf; |
|||
*length = packet->data.frame.sz; |
|||
return 1; |
|||
} |
|||
} |
|||
|
|||
*data = NULL; |
|||
*length = 0; |
|||
return 0; |
|||
} |
|||
|
|||
// Releases exactly the native context allocated by benchmark_create.
|
|||
BENCHMARK_API int benchmark_destroy(benchmark_encoder *encoder) { |
|||
aom_codec_err_t status = aom_codec_destroy(&encoder->codec); |
|||
free(encoder); |
|||
return status; |
|||
} |
|||
|
|||
// Libaom owns this static UTF-8 error string; the caller must not free it.
|
|||
BENCHMARK_API const char *benchmark_error_string(int status) { |
|||
return aom_codec_err_to_string((aom_codec_err_t)status); |
|||
} |
|||
@ -0,0 +1,137 @@ |
|||
# AV1 sequence encoder comparison |
|||
|
|||
`Av1SequenceEncoderBenchmarks` compares the managed AV1 sequence encoder with an optimized build of official libaom `main`. |
|||
The native adapter belongs only to this benchmark project. ImageSharp production code remains fully managed. |
|||
|
|||
## Measurement boundary |
|||
|
|||
Both methods start with the same three `Rgb24` photographic frames and finish with the complete raw AV1 OBU sequence in a reused `MemoryStream`. |
|||
The measured operation includes encoder/workspace construction, RGB-to-YUV conversion for every frame, encoding, output writes, flushing where required, and disposal. |
|||
Both use the existing ImageSharp SIMD color converter and write directly into the aligned source planes consumed by their encoder. |
|||
Libaom receives pinned plane pointers for each synchronous encode call; it does not read a preconverted input file. |
|||
|
|||
Image loading, resizing, half-pixel source translations, validation-file writes, decoding, and quality calculations are outside both measured methods. |
|||
MemoryStream capacity is retained between operations for both methods. These are RGB-to-OBU measurements, not public HEIF container-save measurements. |
|||
Do not compare them with `aomenc`'s internal encode-time report, which excludes the conversion boundary. |
|||
|
|||
The source is the existing `TestImages.Png.Bike` photograph. Frames one and two are independently translated by half a pixel and one pixel on both axes. |
|||
The parameter matrix contains 256x256 and 512x512 frames at ImageSharp efforts seven, eight, and nine. |
|||
Input is full-range BT.601, eight-bit 4:2:0; each sequence contains one key frame and two dependent frames. |
|||
Both codecs use one coding thread. Libaom uses good-quality mode, no lookahead, disabled automatic key frames, and `cpu-used=6`. |
|||
That speed is an independent reference setting, not a mapping from the ImageSharp effort scale. |
|||
|
|||
ImageSharp's base quantizer index is 120. Libaom's public quantizer 30 maps to that same index in the pinned reference. |
|||
The native minimum and maximum quantizers are both fixed to 30, in addition to the CQ setting, so frame-level CQ boosts cannot lower the base quantizer. |
|||
Other native coding tools remain enabled. Identical quantizers do not imply identical quality or bitrate; always retain size and decoded-quality results with timings. |
|||
|
|||
BenchmarkDotNet's allocation column measures managed allocations only. It does not measure libaom's native allocations, pooled native memory, or total peak memory. |
|||
Do not interpret its allocation ratio as a whole-encoder memory comparison. |
|||
|
|||
## Build the reference and benchmark adapter |
|||
|
|||
The verified source revision is `d565eec60f084421fa34fc0534b760c6452b6a6c`, libaom 3.15.0, exported at `D:\GitHub\ynse01\aom-d565eec6-source`. |
|||
The x64 Release build is `D:\GitHub\ynse01\aom-d565eec6-build-x64-release`. |
|||
It uses MSVC 19.51.36256.0 and NASM 3.02, runtime CPU detection, and SSE2, SSE4.1, AVX2, and AVX512 kernels. |
|||
The generic reference decoder build is suitable for conformance but must not be used as the timing reference. |
|||
|
|||
Run from the repository root in an x64 Visual Studio developer PowerShell with the existing CMake, Ninja, NASM, and Perl installations available: |
|||
|
|||
```powershell |
|||
$aomSourceDirectory = 'D:/GitHub/ynse01/aom-d565eec6-source' |
|||
$aomBuildDirectory = 'D:/GitHub/ynse01/aom-d565eec6-build-x64-release' |
|||
$adapterBuildDirectory = 'artifacts/av1-native-benchmark-release' |
|||
|
|||
cmake -S $aomSourceDirectory -B $aomBuildDirectory -G Ninja ` |
|||
-DCMAKE_BUILD_TYPE=Release -DAOM_TARGET_CPU=x86_64 -DENABLE_NASM=ON ` |
|||
-DENABLE_TESTS=OFF -DENABLE_EXAMPLES=ON -DENABLE_TOOLS=ON ` |
|||
-DCONFIG_WEBM_IO=OFF -DCONFIG_LIBYUV=OFF |
|||
if ($LASTEXITCODE -ne 0) { throw 'Reference configuration failed.' } |
|||
|
|||
cmake --build $aomBuildDirectory --target aomenc aomdec --parallel 4 |
|||
if ($LASTEXITCODE -ne 0) { throw 'Reference build failed.' } |
|||
|
|||
cmake -S tests/ImageSharp.Benchmarks/Codecs/Heif/Native -B $adapterBuildDirectory -G Ninja ` |
|||
-DCMAKE_BUILD_TYPE=Release "-DAOM_SOURCE_DIRECTORY=$aomSourceDirectory" "-DAOM_BUILD_DIRECTORY=$aomBuildDirectory" |
|||
if ($LASTEXITCODE -ne 0) { throw 'Adapter configuration failed.' } |
|||
|
|||
cmake --build $adapterBuildDirectory --parallel 1 |
|||
if ($LASTEXITCODE -ne 0) { throw 'Adapter build failed.' } |
|||
|
|||
dotnet build tests/ImageSharp.Benchmarks/ImageSharp.Benchmarks.csproj -c Release -f net11.0 ` |
|||
--no-restore --disable-build-servers -m:1 -p:UseSharedCompilation=false ` |
|||
-p:SIXLABORS_TESTING_PREVIEW=true -p:SIXLABORS_DISABLE_CONFIG_COPY=true |
|||
if ($LASTEXITCODE -ne 0) { throw 'Managed build failed.' } |
|||
|
|||
Copy-Item -LiteralPath "$adapterBuildDirectory/imagesharp_aom_benchmark.dll" ` |
|||
-Destination artifacts/bin/tests/ImageSharp.Benchmarks/Release/net11.0/imagesharp_aom_benchmark.dll |
|||
``` |
|||
|
|||
The adapter links the explicit reference build. It neither downloads tools nor adds a native production dependency. |
|||
Rebuild it whenever the source headers or reference library change; record the exact source revision and optimized build configuration with every report. |
|||
|
|||
## Validate, then measure |
|||
|
|||
Run one pair first, in the existing benchmark host with the in-process toolchain. Do not run builds or tests concurrently with timing. |
|||
|
|||
```powershell |
|||
$benchmarkAssembly = 'artifacts/bin/tests/ImageSharp.Benchmarks/Release/net11.0/ImageSharp.Benchmarks.dll' |
|||
$benchmarkFilter = '*Av1SequenceEncoderBenchmarks.*(Dimension: 256, Effort: 7)' |
|||
|
|||
dotnet $benchmarkAssembly --inProcess --job Dry --filter $benchmarkFilter ` |
|||
--stopOnFirstError --noOverwrite --artifacts artifacts/BenchmarkDotNet/av1-sequence-rgb-dry |
|||
``` |
|||
|
|||
The actual last measured payload for each method and the independently exported source planes are retained under |
|||
`tests/Images/ActualOutput/Heif/Av1/Av1SequenceEncoderBenchmarks`. |
|||
Both timed methods convert RGB on every invocation; the exported source file exists only for offline quality calculation. |
|||
Decode both payloads using the matching current-main `aomdec`, and require the expected complete three-frame YUV length: |
|||
|
|||
```powershell |
|||
$outputDirectory = 'tests/Images/ActualOutput/Heif/Av1/Av1SequenceEncoderBenchmarks' |
|||
foreach ($payloadName in 'bike-256-q120-effort7.obu', 'bike-256-q120-libaom-cpu6.obu') { |
|||
$payloadPath = Join-Path $outputDirectory $payloadName |
|||
& "$aomBuildDirectory/aomdec.exe" --codec=av1 --threads=1 --rawvideo -o "$payloadPath.yuv" $payloadPath |
|||
if ($LASTEXITCODE -ne 0) { throw "Reference decode failed: $payloadName" } |
|||
|
|||
if ((Get-Item -LiteralPath "$payloadPath.yuv").Length -ne (3 * 256 * 256 * 3 / 2)) { |
|||
throw "Incomplete decoded sequence: $payloadName" |
|||
} |
|||
} |
|||
|
|||
dotnet $benchmarkAssembly --inProcess --job Short --filter $benchmarkFilter ` |
|||
--stopOnFirstError --noOverwrite --artifacts artifacts/BenchmarkDotNet/av1-sequence-rgb-short |
|||
``` |
|||
|
|||
Repeat output validation after the measured run. Compute per-plane PSNR against the exported source using |
|||
`10 * log10(255^2 * sampleCount / squaredErrorSum)`, accumulated over all three frames. |
|||
Aggregate YUV PSNR uses the total squared error and sample count, not the arithmetic mean of the three PSNR values. |
|||
Keep bitstream length, quality, absolute time, runtime, CPU information, reference settings, and any environment warnings together. |
|||
The short job is an initial diagnostic checkpoint, not a substitute for the complete size/quality/performance matrix or an equal-quality rate-distortion comparison. |
|||
|
|||
## Verified checkpoint: 2026-09-05 |
|||
|
|||
The final fixed-quantizer pair ran with BenchmarkDotNet 0.15.8, .NET 11.0.0-preview.7.26381.103, x64 RyuJIT, and AVX512 available. |
|||
The in-process Short job used three warmup iterations and three measured iterations. Each operation encoded all three 256x256 frames. |
|||
|
|||
| Encoder setting | Mean sequence time | OBU size | Y PSNR | U PSNR | V PSNR | Aggregate YUV PSNR | |
|||
| --- | ---: | ---: | ---: | ---: | ---: | ---: | |
|||
| ImageSharp effort 7, base index 120 | 2,053.31 ms | 11.54 KiB | 39.040 dB | 38.630 dB | 32.778 dB | 37.124 dB | |
|||
| Libaom cpu-used 6, fixed public quantizer 30 | 70.67 ms | 8.27 KiB | 39.275 dB | 39.637 dB | 37.351 dB | 38.942 dB | |
|||
|
|||
Both final measured payloads decode to exactly three complete native YUV frames using the optimized current-main reference decoder. |
|||
ImageSharp is about 29 times slower for this case while producing a larger, lower-PSNR sequence. The performance exit gate remains open. |
|||
The managed allocation counter reports 9.38 MiB per ImageSharp operation; its source still needs attribution, and it is not comparable with libaom's unmeasured native footprint. |
|||
|
|||
Evidence is retained in `artifacts/BenchmarkDotNet/av1-sequence-rgb-fixed-q-short-20260905/20260905-131149`. |
|||
The initial Dry run and earlier unrestricted-CQ Short run are preliminary checks, not the final comparison above. |
|||
BenchmarkDotNet could not change the power plan or query CPU model information in this environment; those warnings remain in the log. |
|||
This result is a local diagnostic, not a controlled-hardware or equal-quality performance claim. |
|||
|
|||
Final OBU SHA-256 values: |
|||
|
|||
- ImageSharp: `AB0327D33405FB7A6EC1895D465C9EE08E0B218DDA9D7E8EE7E0DCA6C5F8A73B` |
|||
- Libaom: `0C44A3475849762B5A09937397DEA4397264B83E79DAF930A06B70734977E4A9` |
|||
|
|||
The final Release benchmark build has zero errors and 39 existing benchmark warnings outside the new files. |
|||
Roslyn compiler and analyzer diagnostics contain no errors or warnings in the new benchmark files. |
|||
No production file changed in this benchmark checkpoint; the preceding 2,524-case encoder/entropy verification remains the latest production test run. |
|||
Loading…
Reference in new issue