diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md index fa5224d13c..c465ca7276 100644 --- a/HEIF_IMPLEMENTATION_PLAN.md +++ b/HEIF_IMPLEMENTATION_PLAN.md @@ -44,6 +44,8 @@ Reconciled with the worktree on 2026-09-05. Current interpolation-search checkpoint, implemented on 2026-09-05 with focused verification in progress: +- [x] The reference benchmark now measures identical RGB-to-OBU boundaries, including pixel conversion for every frame on both sides. A benchmark-only C adapter links the optimized current-main libaom build in-process and borrows the existing converted plane storage; production remains fully managed. Setup and file I/O are excluded equally. Native quantizer bounds are fixed to the managed base index, with independent speed settings and explicit size/quality reporting. Reproduction, native build provenance, lifetime documentation, and exact output hashes are in `tests/ImageSharp.Benchmarks/Codecs/Heif/README.md`. +- [ ] The corrected benchmark exposes a substantial remaining performance and compression gap. For three photographic 256x256 frames, ImageSharp effort seven takes 2,053.31 ms and writes 11.54 KiB at 37.124 dB aggregate native YUV PSNR; current-main libaom cpu-used six takes 70.67 ms and writes 8.27 KiB at 38.942 dB. Both include RGB conversion and both measured outputs decode to all three complete frames. This is fixed-base-quantizer evidence, not equal-quality evidence. The managed path records 9.38 MiB of managed allocations per operation, which still requires attribution; native memory is not measured by that counter. The Short-run evidence is `artifacts/BenchmarkDotNet/av1-sequence-rgb-fixed-q-short-20260905/20260905-131149`. Power-plan and CPU-query warnings remain documented. The new benchmark files build in net11.0 Release and have no Roslyn compiler/analyzer diagnostics. Do not close the interpolation-performance gate or advance to additional reference tools until the gap is addressed. - [x] The requested in-progress tree was committed as `433afd1a9` before further encoder work. That commit is a checkpoint, not a claim of completed interpolation or codec delivery. - [x] The subsequent partition/interpolation/lossless correction was committed as `7cf7fc4` after the 225-case affected encoder run and exact current-main native comparison. - [x] Odd-sized color sequence verification exposed and corrected two further production defects. Empty inter luma transforms now retain the inferred DCT type before chroma inherits it; normalizing only during writing was too late. Region-major coefficient writing now rounds chroma end coordinates in 4x4 units, preserving the shared minimum chroma transform on sub-8x8 partitions instead of truncating it away. Both rules match current libaom's transform-type inference and `av1_write_intra_coeffs_mb` region bounds. The corrections add no allocation or sample copy. diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs b/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs new file mode 100644 index 0000000000..619e3b30cd --- /dev/null +++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs @@ -0,0 +1,201 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using System.Numerics; +using BenchmarkDotNet.Attributes; +using SixLabors.ImageSharp.Formats.Heif.Av1; +using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit; +using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; +using SixLabors.ImageSharp.Formats.Heif.Components; +using SixLabors.ImageSharp.Memory; +using SixLabors.ImageSharp.PixelFormats; +using SixLabors.ImageSharp.Processing; +using SixLabors.ImageSharp.Tests; + +namespace SixLabors.ImageSharp.Benchmarks.Codecs.Heif; + +/// +/// Measures encoding a photographic sequence with fractional motion at the interpolation-search effort boundaries. +/// +[MemoryDiagnoser] +public class Av1SequenceEncoderBenchmarks +{ + /// + /// The number of displayed pictures in each independently encoded sequence. + /// + private const int FrameCount = 3; + + /// + /// The native AV1 quantizer index corresponding to libaom's public constant-quality level 30. + /// + private const int QIndex = 120; + + /// + /// The fixed native speed baseline, independent of ImageSharp's effort scale. + /// + private const int NativeCpuUsed = 6; + + /// + /// The native public quantizer corresponding to , also used for both rate-control bounds. + /// + private const int NativeQuality = 30; + + private Image sequence; + private Configuration configuration; + private ObuColorConfig colorConfig; + private MemoryStream output; + private string outputDirectory; + + /// + /// Gets or sets the square frame dimension. + /// + [Params(256, 512)] + public int Dimension { get; set; } + + /// + /// Gets or sets the effort controlling fixed, common switchable, or independently switchable filters. + /// + [Params(7, 8, 9)] + public int Effort { get; set; } + + /// + /// Prepares identical photographic RGB frames and a planar source file for checking reconstructed output quality. + /// + [GlobalSetup] + public void Setup() + { + this.configuration = Configuration.Default.Clone(); + this.configuration.MaxDegreeOfParallelism = 1; + this.colorConfig = new ObuColorConfig + { + BitDepth = Av1BitDepth.EightBit, + IsColorDescriptionPresent = true, + ColorPrimaries = ObuColorPrimaries.Bt601, + TransferCharacteristics = ObuTransferCharacteristics.Bt601, + MatrixCoefficients = ObuMatrixCoefficients.Bt601, + ColorRange = true, + SubSamplingX = true, + SubSamplingY = true, + ChromaSamplePosition = ObuChromoSamplePosition.Unknown + }; + + this.output = new MemoryStream(); + this.outputDirectory = TestEnvironment.CreateOutputDirectory("Heif", "Av1", nameof(Av1SequenceEncoderBenchmarks)); + string inputPath = Path.Combine(TestEnvironment.InputImagesDirectoryFullPath, TestImages.Png.Bike); + using Image photograph = Image.Load(inputPath); + + // Leave a source margin for the half-pixel translations. Resampling is setup work, not encoder time; + // every invocation consumes the same three images rather than repeatedly translating a previous result. + photograph.Mutate(context => context.Resize(new ResizeOptions + { + Size = new Size(this.Dimension + FrameCount, this.Dimension + FrameCount), + Mode = ResizeMode.Crop + })); + + Rectangle sourceBounds = new(0, 0, photograph.Width, photograph.Height); + Size targetSize = new(this.Dimension, this.Dimension); + this.sequence = photograph.Clone(context => context.Crop(new Rectangle(Point.Empty, targetSize))); + for (int frameIndex = 1; frameIndex < FrameCount; frameIndex++) + { + Matrix3x2 translation = Matrix3x2.CreateTranslation(-0.5F * frameIndex, -0.5F * frameIndex); + using Image translated = photograph.Clone(context => + context.Transform(sourceBounds, translation, targetSize, KnownResamplers.Bicubic)); + + this.sequence.Frames.AddFrame(translated.Frames.RootFrame); + } + + // Export the production-converted source planes only for checking reconstructed output quality. + // Neither timed encoder reads this file: both convert the original RGB frames during each operation. + using Av1EncoderFrameBuffer planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0); + using FileStream raw = File.Create(Path.Combine(this.outputDirectory, $"bike-{this.Dimension}-3frames.source.yuv")); + foreach (ImageFrame frame in this.sequence.Frames) + { + Av1FrameEncoder.PrepareSource(this.configuration, frame, planar.Frame, this.colorConfig); + for (int planeIndex = 0; planeIndex < this.colorConfig.PlaneCount; planeIndex++) + { + Buffer2DRegion plane = planar.Frame.View.GetPlane((Av1Plane)planeIndex); + for (int y = 0; y < plane.Height; y++) + { + raw.Write(plane.DangerousGetRowSpan(y)); + } + } + } + } + + /// + /// Encodes one key picture and two dependent pictures, returning the complete OBU payload length. + /// + /// The encoded sequence length. + [Benchmark] + public long ImageSharp() + { + this.output.SetLength(0); + using Av1FrameEncoder.SequenceEncoder encoder = Av1FrameEncoder.CreateColorSequenceEncoder( + this.configuration, this.Dimension, this.Dimension, this.colorConfig, QIndex, this.Effort); + + // One operation owns the real sequence lifetime: allocation, conversion, key/inter coding, and disposal. + // The caller's destination is reused, excluding filesystem and MemoryStream growth from steady-state timing. + encoder.EncodeKeyFrame(this.sequence.Frames.RootFrame, this.output); + for (int frameIndex = 1; frameIndex < FrameCount; frameIndex++) + { + encoder.EncodeInterFrame(this.sequence.Frames[frameIndex], this.output); + } + + return this.output.Length; + } + + /// + /// Encodes the same RGB sequence with current-main libaom, including conversion, allocation, output, and disposal. + /// + /// The encoded sequence length. + [Benchmark(Baseline = true)] + public long Libaom() + { + // cpu-used is a separate speed scale, not an ImageSharp effort mapping. Keep the reference at + // good-quality speed six while comparing the three managed interpolation-search boundaries. + this.output.SetLength(0); + using LibaomBenchmarkEncoder encoder = LibaomBenchmarkEncoder.Open(this.Dimension, this.Dimension, NativeQuality, NativeCpuUsed); + using Av1EncoderFrameBuffer planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0); + using Av1FrameEncoder.Av1EncoderConversionWorkspace conversion = new(this.configuration, this.Dimension, this.colorConfig, false, false); + Rectangle bounds = new(0, 0, this.Dimension, this.Dimension); + for (int frameIndex = 0; frameIndex < FrameCount; frameIndex++) + { + // Conversion belongs inside both measured paths. Reuse the same row workspace and SIMD converter + // as the managed sequence encoder, writing directly into the planes passed to native libaom. + conversion.Convert.PlanarView, byte, HeifByteSampleConverter>( + this.configuration, this.sequence.Frames[frameIndex], bounds, planar.Frame.View); + + encoder.Encode(planar.Frame, frameIndex, this.output); + } + + encoder.Finish(this.output); + return this.output.Length; + } + + /// + /// Retains the measured managed encoder output and releases the input images and destination stream. + /// + [GlobalCleanup(Target = nameof(ImageSharp))] + public void CleanupImageSharp() => this.Cleanup($"bike-{this.Dimension}-q{QIndex}-effort{this.Effort}.obu"); + + /// + /// Retains the measured reference encoder output and releases the input images and destination stream. + /// + [GlobalCleanup(Target = nameof(Libaom))] + public void CleanupLibaom() => this.Cleanup($"bike-{this.Dimension}-q{QIndex}-libaom-cpu{NativeCpuUsed}.obu"); + + /// + /// Writes the measured payload without another encoding pass and releases the shared benchmark resources. + /// + /// The codec-specific payload file name. + private void Cleanup(string outputName) + { + // Output validation and quality measurement use the actual measured payload, with no encode or decode + // hidden inside the timed operation and no file-sized ToArray copy. + using FileStream encoded = File.Create(Path.Combine(this.outputDirectory, outputName)); + this.output.Position = 0; + this.output.CopyTo(encoded); + this.output.Dispose(); + this.sequence.Dispose(); + } +} diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/LibaomBenchmarkEncoder.cs b/tests/ImageSharp.Benchmarks/Codecs/Heif/LibaomBenchmarkEncoder.cs new file mode 100644 index 0000000000..a0007d0a24 --- /dev/null +++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/LibaomBenchmarkEncoder.cs @@ -0,0 +1,135 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +using System.Runtime.InteropServices; +using Microsoft.Win32.SafeHandles; +using SixLabors.ImageSharp.Formats.Heif.Av1; +using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline; +using SixLabors.ImageSharp.Memory; + +namespace SixLabors.ImageSharp.Benchmarks.Codecs.Heif; + +/// +/// Owns a benchmark-only libaom encoder through a C adapter compiled against the reference's actual headers. +/// +internal sealed unsafe partial class LibaomBenchmarkEncoder : SafeHandleZeroOrMinusOneIsInvalid +{ + /// + /// The benchmark adapter's platform-independent library name. + /// + private const string LibraryName = "imagesharp_aom_benchmark"; + + /// + /// Initializes an empty handle for the native create call's generated marshaller. + /// + public LibaomBenchmarkEncoder() + : base(ownsHandle: true) + { + } + + /// + /// Opens one native sequence with a single coding thread, no lookahead, and the requested quality and speed. + /// + public static LibaomBenchmarkEncoder Open(int width, int height, int quality, int speed) + { + CheckStatus(Create((uint)width, (uint)height, (uint)quality, speed, out LibaomBenchmarkEncoder encoder)); + return encoder; + } + + /// + /// Encodes the converted source planes and consumes every output packet before those bytes can be invalidated. + /// + public void Encode(Av1EncoderFrame frame, long frameIndex, Stream output) + { + Buffer2DRegion y = frame.View.GetPlane(Av1Plane.Y); + Buffer2DRegion u = frame.View.GetPlane(Av1Plane.U); + Buffer2DRegion v = frame.View.GetPlane(Av1Plane.V); + fixed (byte* yPointer = y.DangerousGetRowSpan(0), uPointer = u.DangerousGetRowSpan(0), vPointer = v.DangerousGetRowSpan(0)) + { + // The native call is synchronous. Its input descriptors borrow these pinned rows only until + // EncodeFrame returns; the encoder owns any retained reference and lookahead storage itself. + CheckStatus(EncodeFrame(this, yPointer, uPointer, vPointer, y.Stride, u.Stride, frameIndex)); + } + + this.WritePackets(output); + } + + /// + /// Finishes the sequence and writes any remaining coded packets. + /// + public void Finish(Stream output) + { + do + { + CheckStatus(Flush(this)); + } + while (this.WritePackets(output)); + } + + /// + protected override bool ReleaseHandle() => Destroy(this.handle) == 0; + + /// + /// Writes borrowed packet memory before the next call into the native codec invalidates it. + /// + private bool WritePackets(Stream output) + { + bool wrotePacket = false; + while (NextPacket(this, out byte* data, out nuint length) != 0) + { + // Packet lengths are bounded by the benchmark's image dimensions. Stream.Write consumes the + // borrowed bytes synchronously, without retaining a native pointer or creating a managed array. + output.Write(new ReadOnlySpan(data, (int)length)); + wrotePacket = true; + } + + return wrotePacket; + } + + /// + /// Converts a native codec error into a managed benchmark failure instead of accepting invalid timing data. + /// + private static void CheckStatus(int status) + { + if (status != 0) + { + throw new InvalidOperationException(Marshal.PtrToStringUTF8(ErrorString(status))); + } + } + + /// + /// Creates an owned opaque context; no libaom structure layout crosses the managed boundary. + /// + [LibraryImport(LibraryName, EntryPoint = "benchmark_create")] + private static partial int Create(uint width, uint height, uint quality, int speed, out LibaomBenchmarkEncoder encoder); + + /// + /// Borrows three pinned planes for one synchronous native encode call. + /// + [LibraryImport(LibraryName, EntryPoint = "benchmark_encode")] + private static partial int EncodeFrame(LibaomBenchmarkEncoder encoder, byte* y, byte* u, byte* v, int yStride, int uvStride, long frameIndex); + + /// + /// Signals the end of the native sequence. + /// + [LibraryImport(LibraryName, EntryPoint = "benchmark_flush")] + private static partial int Flush(LibaomBenchmarkEncoder encoder); + + /// + /// Returns borrowed native packet storage and its pointer-sized length. + /// + [LibraryImport(LibraryName, EntryPoint = "benchmark_next_packet")] + private static partial int NextPacket(LibaomBenchmarkEncoder encoder, out byte* data, out nuint length); + + /// + /// Destroys the context through the same native library that allocated it. + /// + [LibraryImport(LibraryName, EntryPoint = "benchmark_destroy")] + private static partial int Destroy(nint encoder); + + /// + /// Returns a static UTF-8 error message owned by libaom. + /// + [LibraryImport(LibraryName, EntryPoint = "benchmark_error_string")] + private static partial nint ErrorString(int status); +} diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/CMakeLists.txt b/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/CMakeLists.txt new file mode 100644 index 0000000000..f19a785265 --- /dev/null +++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/CMakeLists.txt @@ -0,0 +1,12 @@ +cmake_minimum_required(VERSION 3.20) +project(imagesharp_aom_benchmark LANGUAGES C CXX) + +# Link the explicitly supplied current-main Release reference build. This adapter is benchmark-only; +# it is never referenced by ImageSharp or copied by the production project. +set(AOM_SOURCE_DIRECTORY "" CACHE PATH "Verified official libaom source directory") +set(AOM_BUILD_DIRECTORY "" CACHE PATH "Matching optimized Release libaom build directory") +add_library(imagesharp_aom_benchmark SHARED aom_benchmark.c) +target_include_directories(imagesharp_aom_benchmark PRIVATE "${AOM_SOURCE_DIRECTORY}") +find_library(AOM_LIBRARY NAMES aom PATHS "${AOM_BUILD_DIRECTORY}" NO_DEFAULT_PATH REQUIRED) +target_link_libraries(imagesharp_aom_benchmark PRIVATE "${AOM_LIBRARY}") +set_target_properties(imagesharp_aom_benchmark PROPERTIES C_STANDARD 11 LINKER_LANGUAGE CXX) diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/aom_benchmark.c b/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/aom_benchmark.c new file mode 100644 index 0000000000..ba5f00e154 --- /dev/null +++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/aom_benchmark.c @@ -0,0 +1,131 @@ +// Copyright (c) Six Labors. +// Licensed under the Six Labors Split License. + +#include +#include +#include "aom/aom_encoder.h" +#include "aom/aomcx.h" + +#if defined(_WIN32) +#define BENCHMARK_API __declspec(dllexport) +#else +#define BENCHMARK_API __attribute__((visibility("default"))) +#endif + +// Keep libaom's version-dependent structures and variadic controls entirely on the C side. +// The managed benchmark exchanges only opaque ownership, fixed-width integers, and borrowed buffers. +typedef struct benchmark_encoder { + aom_codec_ctx_t codec; + aom_codec_iter_t iterator; + unsigned int width; + unsigned int height; +} benchmark_encoder; + +// Creates one single-threaded, unlagged, constant-quality sequence using the reference's normal tools. +// A successful result belongs to the caller and must be released by benchmark_destroy. +BENCHMARK_API int benchmark_create(unsigned int width, unsigned int height, + unsigned int quality, int speed, + benchmark_encoder **result) { + aom_codec_enc_cfg_t config; + aom_codec_iface_t *iface = aom_codec_av1_cx(); + aom_codec_err_t status = aom_codec_enc_config_default(iface, &config, AOM_USAGE_GOOD_QUALITY); + *result = NULL; + if (status != AOM_CODEC_OK) return status; + + benchmark_encoder *encoder = calloc(1, sizeof(*encoder)); + if (encoder == NULL) return AOM_CODEC_MEM_ERROR; + + config.g_w = width; + config.g_h = height; + config.g_threads = 1; + config.g_lag_in_frames = 0; + config.g_timebase.num = 1; + config.g_timebase.den = 30; + config.rc_end_usage = AOM_Q; + // Fix both bounds to the requested quantizer. CQ alone permits frame-quality boosts, whereas + // the managed sequence uses this same base quantizer for every frame in the comparison. + config.rc_min_quantizer = quality; + config.rc_max_quantizer = quality; + config.kf_mode = AOM_KF_DISABLED; + status = aom_codec_enc_init(&encoder->codec, iface, &config, 0); + if (status != AOM_CODEC_OK) { + free(encoder); + return status; + } + + // These calls remain type-checked against the current libaom headers, including each control's argument type. + status = aom_codec_control(&encoder->codec, AOME_SET_CPUUSED, speed); + if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AOME_SET_CQ_LEVEL, quality); + if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_ROW_MT, 0u); + if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_COLOR_PRIMARIES, AOM_CICP_CP_BT_601); + if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_TRANSFER_CHARACTERISTICS, AOM_CICP_TC_BT_601); + if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_MATRIX_COEFFICIENTS, AOM_CICP_MC_BT_601); + if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_COLOR_RANGE, AOM_CR_FULL_RANGE); + if (status != AOM_CODEC_OK) { + aom_codec_destroy(&encoder->codec); + free(encoder); + return status; + } + + encoder->width = width; + encoder->height = height; + *result = encoder; + return AOM_CODEC_OK; +} + +// Borrows the already converted planes for this synchronous encode call. The caller pins all three +// pointers until it returns; libaom retains its own reference pictures, never these managed input pointers. +BENCHMARK_API int benchmark_encode(benchmark_encoder *encoder, unsigned char *y, + unsigned char *u, unsigned char *v, + int y_stride, int uv_stride, int64_t frame_index) { + aom_image_t image; + if (aom_img_wrap(&image, AOM_IMG_FMT_I420, encoder->width, encoder->height, 1, y) == NULL) { + return AOM_CODEC_INVALID_PARAM; + } + + // ImageSharp's source planes include aligned borders. Describe those existing rows directly instead + // of flattening them into another contiguous YUV allocation before calling the reference encoder. + image.planes[AOM_PLANE_Y] = y; + image.planes[AOM_PLANE_U] = u; + image.planes[AOM_PLANE_V] = v; + image.stride[AOM_PLANE_Y] = y_stride; + image.stride[AOM_PLANE_U] = uv_stride; + image.stride[AOM_PLANE_V] = uv_stride; + encoder->iterator = NULL; + return aom_codec_encode(&encoder->codec, &image, frame_index, 1, 0); +} + +// Drains delayed output at the end of the sequence even though lookahead is disabled. +BENCHMARK_API int benchmark_flush(benchmark_encoder *encoder) { + encoder->iterator = NULL; + return aom_codec_encode(&encoder->codec, NULL, -1, 1, 0); +} + +// The returned packet storage belongs to libaom and is valid only until the next codec call. +// The managed side copies it directly into the same kind of destination stream as its own encoder. +BENCHMARK_API int benchmark_next_packet(benchmark_encoder *encoder, const void **data, size_t *length) { + const aom_codec_cx_pkt_t *packet; + while ((packet = aom_codec_get_cx_data(&encoder->codec, &encoder->iterator)) != NULL) { + if (packet->kind == AOM_CODEC_CX_FRAME_PKT) { + *data = packet->data.frame.buf; + *length = packet->data.frame.sz; + return 1; + } + } + + *data = NULL; + *length = 0; + return 0; +} + +// Releases exactly the native context allocated by benchmark_create. +BENCHMARK_API int benchmark_destroy(benchmark_encoder *encoder) { + aom_codec_err_t status = aom_codec_destroy(&encoder->codec); + free(encoder); + return status; +} + +// Libaom owns this static UTF-8 error string; the caller must not free it. +BENCHMARK_API const char *benchmark_error_string(int status) { + return aom_codec_err_to_string((aom_codec_err_t)status); +} diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/README.md b/tests/ImageSharp.Benchmarks/Codecs/Heif/README.md new file mode 100644 index 0000000000..5bc9e1d613 --- /dev/null +++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/README.md @@ -0,0 +1,137 @@ +# AV1 sequence encoder comparison + +`Av1SequenceEncoderBenchmarks` compares the managed AV1 sequence encoder with an optimized build of official libaom `main`. +The native adapter belongs only to this benchmark project. ImageSharp production code remains fully managed. + +## Measurement boundary + +Both methods start with the same three `Rgb24` photographic frames and finish with the complete raw AV1 OBU sequence in a reused `MemoryStream`. +The measured operation includes encoder/workspace construction, RGB-to-YUV conversion for every frame, encoding, output writes, flushing where required, and disposal. +Both use the existing ImageSharp SIMD color converter and write directly into the aligned source planes consumed by their encoder. +Libaom receives pinned plane pointers for each synchronous encode call; it does not read a preconverted input file. + +Image loading, resizing, half-pixel source translations, validation-file writes, decoding, and quality calculations are outside both measured methods. +MemoryStream capacity is retained between operations for both methods. These are RGB-to-OBU measurements, not public HEIF container-save measurements. +Do not compare them with `aomenc`'s internal encode-time report, which excludes the conversion boundary. + +The source is the existing `TestImages.Png.Bike` photograph. Frames one and two are independently translated by half a pixel and one pixel on both axes. +The parameter matrix contains 256x256 and 512x512 frames at ImageSharp efforts seven, eight, and nine. +Input is full-range BT.601, eight-bit 4:2:0; each sequence contains one key frame and two dependent frames. +Both codecs use one coding thread. Libaom uses good-quality mode, no lookahead, disabled automatic key frames, and `cpu-used=6`. +That speed is an independent reference setting, not a mapping from the ImageSharp effort scale. + +ImageSharp's base quantizer index is 120. Libaom's public quantizer 30 maps to that same index in the pinned reference. +The native minimum and maximum quantizers are both fixed to 30, in addition to the CQ setting, so frame-level CQ boosts cannot lower the base quantizer. +Other native coding tools remain enabled. Identical quantizers do not imply identical quality or bitrate; always retain size and decoded-quality results with timings. + +BenchmarkDotNet's allocation column measures managed allocations only. It does not measure libaom's native allocations, pooled native memory, or total peak memory. +Do not interpret its allocation ratio as a whole-encoder memory comparison. + +## Build the reference and benchmark adapter + +The verified source revision is `d565eec60f084421fa34fc0534b760c6452b6a6c`, libaom 3.15.0, exported at `D:\GitHub\ynse01\aom-d565eec6-source`. +The x64 Release build is `D:\GitHub\ynse01\aom-d565eec6-build-x64-release`. +It uses MSVC 19.51.36256.0 and NASM 3.02, runtime CPU detection, and SSE2, SSE4.1, AVX2, and AVX512 kernels. +The generic reference decoder build is suitable for conformance but must not be used as the timing reference. + +Run from the repository root in an x64 Visual Studio developer PowerShell with the existing CMake, Ninja, NASM, and Perl installations available: + +```powershell +$aomSourceDirectory = 'D:/GitHub/ynse01/aom-d565eec6-source' +$aomBuildDirectory = 'D:/GitHub/ynse01/aom-d565eec6-build-x64-release' +$adapterBuildDirectory = 'artifacts/av1-native-benchmark-release' + +cmake -S $aomSourceDirectory -B $aomBuildDirectory -G Ninja ` + -DCMAKE_BUILD_TYPE=Release -DAOM_TARGET_CPU=x86_64 -DENABLE_NASM=ON ` + -DENABLE_TESTS=OFF -DENABLE_EXAMPLES=ON -DENABLE_TOOLS=ON ` + -DCONFIG_WEBM_IO=OFF -DCONFIG_LIBYUV=OFF +if ($LASTEXITCODE -ne 0) { throw 'Reference configuration failed.' } + +cmake --build $aomBuildDirectory --target aomenc aomdec --parallel 4 +if ($LASTEXITCODE -ne 0) { throw 'Reference build failed.' } + +cmake -S tests/ImageSharp.Benchmarks/Codecs/Heif/Native -B $adapterBuildDirectory -G Ninja ` + -DCMAKE_BUILD_TYPE=Release "-DAOM_SOURCE_DIRECTORY=$aomSourceDirectory" "-DAOM_BUILD_DIRECTORY=$aomBuildDirectory" +if ($LASTEXITCODE -ne 0) { throw 'Adapter configuration failed.' } + +cmake --build $adapterBuildDirectory --parallel 1 +if ($LASTEXITCODE -ne 0) { throw 'Adapter build failed.' } + +dotnet build tests/ImageSharp.Benchmarks/ImageSharp.Benchmarks.csproj -c Release -f net11.0 ` + --no-restore --disable-build-servers -m:1 -p:UseSharedCompilation=false ` + -p:SIXLABORS_TESTING_PREVIEW=true -p:SIXLABORS_DISABLE_CONFIG_COPY=true +if ($LASTEXITCODE -ne 0) { throw 'Managed build failed.' } + +Copy-Item -LiteralPath "$adapterBuildDirectory/imagesharp_aom_benchmark.dll" ` + -Destination artifacts/bin/tests/ImageSharp.Benchmarks/Release/net11.0/imagesharp_aom_benchmark.dll +``` + +The adapter links the explicit reference build. It neither downloads tools nor adds a native production dependency. +Rebuild it whenever the source headers or reference library change; record the exact source revision and optimized build configuration with every report. + +## Validate, then measure + +Run one pair first, in the existing benchmark host with the in-process toolchain. Do not run builds or tests concurrently with timing. + +```powershell +$benchmarkAssembly = 'artifacts/bin/tests/ImageSharp.Benchmarks/Release/net11.0/ImageSharp.Benchmarks.dll' +$benchmarkFilter = '*Av1SequenceEncoderBenchmarks.*(Dimension: 256, Effort: 7)' + +dotnet $benchmarkAssembly --inProcess --job Dry --filter $benchmarkFilter ` + --stopOnFirstError --noOverwrite --artifacts artifacts/BenchmarkDotNet/av1-sequence-rgb-dry +``` + +The actual last measured payload for each method and the independently exported source planes are retained under +`tests/Images/ActualOutput/Heif/Av1/Av1SequenceEncoderBenchmarks`. +Both timed methods convert RGB on every invocation; the exported source file exists only for offline quality calculation. +Decode both payloads using the matching current-main `aomdec`, and require the expected complete three-frame YUV length: + +```powershell +$outputDirectory = 'tests/Images/ActualOutput/Heif/Av1/Av1SequenceEncoderBenchmarks' +foreach ($payloadName in 'bike-256-q120-effort7.obu', 'bike-256-q120-libaom-cpu6.obu') { + $payloadPath = Join-Path $outputDirectory $payloadName + & "$aomBuildDirectory/aomdec.exe" --codec=av1 --threads=1 --rawvideo -o "$payloadPath.yuv" $payloadPath + if ($LASTEXITCODE -ne 0) { throw "Reference decode failed: $payloadName" } + + if ((Get-Item -LiteralPath "$payloadPath.yuv").Length -ne (3 * 256 * 256 * 3 / 2)) { + throw "Incomplete decoded sequence: $payloadName" + } +} + +dotnet $benchmarkAssembly --inProcess --job Short --filter $benchmarkFilter ` + --stopOnFirstError --noOverwrite --artifacts artifacts/BenchmarkDotNet/av1-sequence-rgb-short +``` + +Repeat output validation after the measured run. Compute per-plane PSNR against the exported source using +`10 * log10(255^2 * sampleCount / squaredErrorSum)`, accumulated over all three frames. +Aggregate YUV PSNR uses the total squared error and sample count, not the arithmetic mean of the three PSNR values. +Keep bitstream length, quality, absolute time, runtime, CPU information, reference settings, and any environment warnings together. +The short job is an initial diagnostic checkpoint, not a substitute for the complete size/quality/performance matrix or an equal-quality rate-distortion comparison. + +## Verified checkpoint: 2026-09-05 + +The final fixed-quantizer pair ran with BenchmarkDotNet 0.15.8, .NET 11.0.0-preview.7.26381.103, x64 RyuJIT, and AVX512 available. +The in-process Short job used three warmup iterations and three measured iterations. Each operation encoded all three 256x256 frames. + +| Encoder setting | Mean sequence time | OBU size | Y PSNR | U PSNR | V PSNR | Aggregate YUV PSNR | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| ImageSharp effort 7, base index 120 | 2,053.31 ms | 11.54 KiB | 39.040 dB | 38.630 dB | 32.778 dB | 37.124 dB | +| Libaom cpu-used 6, fixed public quantizer 30 | 70.67 ms | 8.27 KiB | 39.275 dB | 39.637 dB | 37.351 dB | 38.942 dB | + +Both final measured payloads decode to exactly three complete native YUV frames using the optimized current-main reference decoder. +ImageSharp is about 29 times slower for this case while producing a larger, lower-PSNR sequence. The performance exit gate remains open. +The managed allocation counter reports 9.38 MiB per ImageSharp operation; its source still needs attribution, and it is not comparable with libaom's unmeasured native footprint. + +Evidence is retained in `artifacts/BenchmarkDotNet/av1-sequence-rgb-fixed-q-short-20260905/20260905-131149`. +The initial Dry run and earlier unrestricted-CQ Short run are preliminary checks, not the final comparison above. +BenchmarkDotNet could not change the power plan or query CPU model information in this environment; those warnings remain in the log. +This result is a local diagnostic, not a controlled-hardware or equal-quality performance claim. + +Final OBU SHA-256 values: + +- ImageSharp: `AB0327D33405FB7A6EC1895D465C9EE08E0B218DDA9D7E8EE7E0DCA6C5F8A73B` +- Libaom: `0C44A3475849762B5A09937397DEA4397264B83E79DAF930A06B70734977E4A9` + +The final Release benchmark build has zero errors and 39 existing benchmark warnings outside the new files. +Roslyn compiler and analyzer diagnostics contain no errors or warnings in the new benchmark files. +No production file changed in this benchmark checkpoint; the preceding 2,524-case encoder/entropy verification remains the latest production test run.