diff --git a/HEIF_IMPLEMENTATION_PLAN.md b/HEIF_IMPLEMENTATION_PLAN.md
index fa5224d13c..c465ca7276 100644
--- a/HEIF_IMPLEMENTATION_PLAN.md
+++ b/HEIF_IMPLEMENTATION_PLAN.md
@@ -44,6 +44,8 @@ Reconciled with the worktree on 2026-09-05.
Current interpolation-search checkpoint, implemented on 2026-09-05 with focused verification in progress:
+- [x] The reference benchmark now measures identical RGB-to-OBU boundaries, including pixel conversion for every frame on both sides. A benchmark-only C adapter links the optimized current-main libaom build in-process and borrows the existing converted plane storage; production remains fully managed. Setup and file I/O are excluded equally. Native quantizer bounds are fixed to the managed base index, with independent speed settings and explicit size/quality reporting. Reproduction, native build provenance, lifetime documentation, and exact output hashes are in `tests/ImageSharp.Benchmarks/Codecs/Heif/README.md`.
+- [ ] The corrected benchmark exposes a substantial remaining performance and compression gap. For three photographic 256x256 frames, ImageSharp effort seven takes 2,053.31 ms and writes 11.54 KiB at 37.124 dB aggregate native YUV PSNR; current-main libaom cpu-used six takes 70.67 ms and writes 8.27 KiB at 38.942 dB. Both include RGB conversion and both measured outputs decode to all three complete frames. This is fixed-base-quantizer evidence, not equal-quality evidence. The managed path records 9.38 MiB of managed allocations per operation, which still requires attribution; native memory is not measured by that counter. The Short-run evidence is `artifacts/BenchmarkDotNet/av1-sequence-rgb-fixed-q-short-20260905/20260905-131149`. Power-plan and CPU-query warnings remain documented. The new benchmark files build in net11.0 Release and have no Roslyn compiler/analyzer diagnostics. Do not close the interpolation-performance gate or advance to additional reference tools until the gap is addressed.
- [x] The requested in-progress tree was committed as `433afd1a9` before further encoder work. That commit is a checkpoint, not a claim of completed interpolation or codec delivery.
- [x] The subsequent partition/interpolation/lossless correction was committed as `7cf7fc4` after the 225-case affected encoder run and exact current-main native comparison.
- [x] Odd-sized color sequence verification exposed and corrected two further production defects. Empty inter luma transforms now retain the inferred DCT type before chroma inherits it; normalizing only during writing was too late. Region-major coefficient writing now rounds chroma end coordinates in 4x4 units, preserving the shared minimum chroma transform on sub-8x8 partitions instead of truncating it away. Both rules match current libaom's transform-type inference and `av1_write_intra_coeffs_mb` region bounds. The corrections add no allocation or sample copy.
diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs b/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs
new file mode 100644
index 0000000000..619e3b30cd
--- /dev/null
+++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/Av1SequenceEncoderBenchmarks.cs
@@ -0,0 +1,201 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Numerics;
+using BenchmarkDotNet.Attributes;
+using SixLabors.ImageSharp.Formats.Heif.Av1;
+using SixLabors.ImageSharp.Formats.Heif.Av1.OpenBitstreamUnit;
+using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
+using SixLabors.ImageSharp.Formats.Heif.Components;
+using SixLabors.ImageSharp.Memory;
+using SixLabors.ImageSharp.PixelFormats;
+using SixLabors.ImageSharp.Processing;
+using SixLabors.ImageSharp.Tests;
+
+namespace SixLabors.ImageSharp.Benchmarks.Codecs.Heif;
+
+///
+/// Measures encoding a photographic sequence with fractional motion at the interpolation-search effort boundaries.
+///
+[MemoryDiagnoser]
+public class Av1SequenceEncoderBenchmarks
+{
+ ///
+ /// The number of displayed pictures in each independently encoded sequence.
+ ///
+ private const int FrameCount = 3;
+
+ ///
+ /// The native AV1 quantizer index corresponding to libaom's public constant-quality level 30.
+ ///
+ private const int QIndex = 120;
+
+ ///
+ /// The fixed native speed baseline, independent of ImageSharp's effort scale.
+ ///
+ private const int NativeCpuUsed = 6;
+
+ ///
+ /// The native public quantizer corresponding to , also used for both rate-control bounds.
+ ///
+ private const int NativeQuality = 30;
+
+ private Image sequence;
+ private Configuration configuration;
+ private ObuColorConfig colorConfig;
+ private MemoryStream output;
+ private string outputDirectory;
+
+ ///
+ /// Gets or sets the square frame dimension.
+ ///
+ [Params(256, 512)]
+ public int Dimension { get; set; }
+
+ ///
+ /// Gets or sets the effort controlling fixed, common switchable, or independently switchable filters.
+ ///
+ [Params(7, 8, 9)]
+ public int Effort { get; set; }
+
+ ///
+ /// Prepares identical photographic RGB frames and a planar source file for checking reconstructed output quality.
+ ///
+ [GlobalSetup]
+ public void Setup()
+ {
+ this.configuration = Configuration.Default.Clone();
+ this.configuration.MaxDegreeOfParallelism = 1;
+ this.colorConfig = new ObuColorConfig
+ {
+ BitDepth = Av1BitDepth.EightBit,
+ IsColorDescriptionPresent = true,
+ ColorPrimaries = ObuColorPrimaries.Bt601,
+ TransferCharacteristics = ObuTransferCharacteristics.Bt601,
+ MatrixCoefficients = ObuMatrixCoefficients.Bt601,
+ ColorRange = true,
+ SubSamplingX = true,
+ SubSamplingY = true,
+ ChromaSamplePosition = ObuChromoSamplePosition.Unknown
+ };
+
+ this.output = new MemoryStream();
+ this.outputDirectory = TestEnvironment.CreateOutputDirectory("Heif", "Av1", nameof(Av1SequenceEncoderBenchmarks));
+ string inputPath = Path.Combine(TestEnvironment.InputImagesDirectoryFullPath, TestImages.Png.Bike);
+ using Image photograph = Image.Load(inputPath);
+
+ // Leave a source margin for the half-pixel translations. Resampling is setup work, not encoder time;
+ // every invocation consumes the same three images rather than repeatedly translating a previous result.
+ photograph.Mutate(context => context.Resize(new ResizeOptions
+ {
+ Size = new Size(this.Dimension + FrameCount, this.Dimension + FrameCount),
+ Mode = ResizeMode.Crop
+ }));
+
+ Rectangle sourceBounds = new(0, 0, photograph.Width, photograph.Height);
+ Size targetSize = new(this.Dimension, this.Dimension);
+ this.sequence = photograph.Clone(context => context.Crop(new Rectangle(Point.Empty, targetSize)));
+ for (int frameIndex = 1; frameIndex < FrameCount; frameIndex++)
+ {
+ Matrix3x2 translation = Matrix3x2.CreateTranslation(-0.5F * frameIndex, -0.5F * frameIndex);
+ using Image translated = photograph.Clone(context =>
+ context.Transform(sourceBounds, translation, targetSize, KnownResamplers.Bicubic));
+
+ this.sequence.Frames.AddFrame(translated.Frames.RootFrame);
+ }
+
+ // Export the production-converted source planes only for checking reconstructed output quality.
+ // Neither timed encoder reads this file: both convert the original RGB frames during each operation.
+ using Av1EncoderFrameBuffer planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0);
+ using FileStream raw = File.Create(Path.Combine(this.outputDirectory, $"bike-{this.Dimension}-3frames.source.yuv"));
+ foreach (ImageFrame frame in this.sequence.Frames)
+ {
+ Av1FrameEncoder.PrepareSource(this.configuration, frame, planar.Frame, this.colorConfig);
+ for (int planeIndex = 0; planeIndex < this.colorConfig.PlaneCount; planeIndex++)
+ {
+ Buffer2DRegion plane = planar.Frame.View.GetPlane((Av1Plane)planeIndex);
+ for (int y = 0; y < plane.Height; y++)
+ {
+ raw.Write(plane.DangerousGetRowSpan(y));
+ }
+ }
+ }
+ }
+
+ ///
+ /// Encodes one key picture and two dependent pictures, returning the complete OBU payload length.
+ ///
+ /// The encoded sequence length.
+ [Benchmark]
+ public long ImageSharp()
+ {
+ this.output.SetLength(0);
+ using Av1FrameEncoder.SequenceEncoder encoder = Av1FrameEncoder.CreateColorSequenceEncoder(
+ this.configuration, this.Dimension, this.Dimension, this.colorConfig, QIndex, this.Effort);
+
+ // One operation owns the real sequence lifetime: allocation, conversion, key/inter coding, and disposal.
+ // The caller's destination is reused, excluding filesystem and MemoryStream growth from steady-state timing.
+ encoder.EncodeKeyFrame(this.sequence.Frames.RootFrame, this.output);
+ for (int frameIndex = 1; frameIndex < FrameCount; frameIndex++)
+ {
+ encoder.EncodeInterFrame(this.sequence.Frames[frameIndex], this.output);
+ }
+
+ return this.output.Length;
+ }
+
+ ///
+ /// Encodes the same RGB sequence with current-main libaom, including conversion, allocation, output, and disposal.
+ ///
+ /// The encoded sequence length.
+ [Benchmark(Baseline = true)]
+ public long Libaom()
+ {
+ // cpu-used is a separate speed scale, not an ImageSharp effort mapping. Keep the reference at
+ // good-quality speed six while comparing the three managed interpolation-search boundaries.
+ this.output.SetLength(0);
+ using LibaomBenchmarkEncoder encoder = LibaomBenchmarkEncoder.Open(this.Dimension, this.Dimension, NativeQuality, NativeCpuUsed);
+ using Av1EncoderFrameBuffer planar = new(this.configuration, this.Dimension, this.Dimension, 8, Av1ColorFormat.Yuv420, 0, 0);
+ using Av1FrameEncoder.Av1EncoderConversionWorkspace conversion = new(this.configuration, this.Dimension, this.colorConfig, false, false);
+ Rectangle bounds = new(0, 0, this.Dimension, this.Dimension);
+ for (int frameIndex = 0; frameIndex < FrameCount; frameIndex++)
+ {
+ // Conversion belongs inside both measured paths. Reuse the same row workspace and SIMD converter
+ // as the managed sequence encoder, writing directly into the planes passed to native libaom.
+ conversion.Convert.PlanarView, byte, HeifByteSampleConverter>(
+ this.configuration, this.sequence.Frames[frameIndex], bounds, planar.Frame.View);
+
+ encoder.Encode(planar.Frame, frameIndex, this.output);
+ }
+
+ encoder.Finish(this.output);
+ return this.output.Length;
+ }
+
+ ///
+ /// Retains the measured managed encoder output and releases the input images and destination stream.
+ ///
+ [GlobalCleanup(Target = nameof(ImageSharp))]
+ public void CleanupImageSharp() => this.Cleanup($"bike-{this.Dimension}-q{QIndex}-effort{this.Effort}.obu");
+
+ ///
+ /// Retains the measured reference encoder output and releases the input images and destination stream.
+ ///
+ [GlobalCleanup(Target = nameof(Libaom))]
+ public void CleanupLibaom() => this.Cleanup($"bike-{this.Dimension}-q{QIndex}-libaom-cpu{NativeCpuUsed}.obu");
+
+ ///
+ /// Writes the measured payload without another encoding pass and releases the shared benchmark resources.
+ ///
+ /// The codec-specific payload file name.
+ private void Cleanup(string outputName)
+ {
+ // Output validation and quality measurement use the actual measured payload, with no encode or decode
+ // hidden inside the timed operation and no file-sized ToArray copy.
+ using FileStream encoded = File.Create(Path.Combine(this.outputDirectory, outputName));
+ this.output.Position = 0;
+ this.output.CopyTo(encoded);
+ this.output.Dispose();
+ this.sequence.Dispose();
+ }
+}
diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/LibaomBenchmarkEncoder.cs b/tests/ImageSharp.Benchmarks/Codecs/Heif/LibaomBenchmarkEncoder.cs
new file mode 100644
index 0000000000..a0007d0a24
--- /dev/null
+++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/LibaomBenchmarkEncoder.cs
@@ -0,0 +1,135 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+using System.Runtime.InteropServices;
+using Microsoft.Win32.SafeHandles;
+using SixLabors.ImageSharp.Formats.Heif.Av1;
+using SixLabors.ImageSharp.Formats.Heif.Av1.Pipeline;
+using SixLabors.ImageSharp.Memory;
+
+namespace SixLabors.ImageSharp.Benchmarks.Codecs.Heif;
+
+///
+/// Owns a benchmark-only libaom encoder through a C adapter compiled against the reference's actual headers.
+///
+internal sealed unsafe partial class LibaomBenchmarkEncoder : SafeHandleZeroOrMinusOneIsInvalid
+{
+ ///
+ /// The benchmark adapter's platform-independent library name.
+ ///
+ private const string LibraryName = "imagesharp_aom_benchmark";
+
+ ///
+ /// Initializes an empty handle for the native create call's generated marshaller.
+ ///
+ public LibaomBenchmarkEncoder()
+ : base(ownsHandle: true)
+ {
+ }
+
+ ///
+ /// Opens one native sequence with a single coding thread, no lookahead, and the requested quality and speed.
+ ///
+ public static LibaomBenchmarkEncoder Open(int width, int height, int quality, int speed)
+ {
+ CheckStatus(Create((uint)width, (uint)height, (uint)quality, speed, out LibaomBenchmarkEncoder encoder));
+ return encoder;
+ }
+
+ ///
+ /// Encodes the converted source planes and consumes every output packet before those bytes can be invalidated.
+ ///
+ public void Encode(Av1EncoderFrame frame, long frameIndex, Stream output)
+ {
+ Buffer2DRegion y = frame.View.GetPlane(Av1Plane.Y);
+ Buffer2DRegion u = frame.View.GetPlane(Av1Plane.U);
+ Buffer2DRegion v = frame.View.GetPlane(Av1Plane.V);
+ fixed (byte* yPointer = y.DangerousGetRowSpan(0), uPointer = u.DangerousGetRowSpan(0), vPointer = v.DangerousGetRowSpan(0))
+ {
+ // The native call is synchronous. Its input descriptors borrow these pinned rows only until
+ // EncodeFrame returns; the encoder owns any retained reference and lookahead storage itself.
+ CheckStatus(EncodeFrame(this, yPointer, uPointer, vPointer, y.Stride, u.Stride, frameIndex));
+ }
+
+ this.WritePackets(output);
+ }
+
+ ///
+ /// Finishes the sequence and writes any remaining coded packets.
+ ///
+ public void Finish(Stream output)
+ {
+ do
+ {
+ CheckStatus(Flush(this));
+ }
+ while (this.WritePackets(output));
+ }
+
+ ///
+ protected override bool ReleaseHandle() => Destroy(this.handle) == 0;
+
+ ///
+ /// Writes borrowed packet memory before the next call into the native codec invalidates it.
+ ///
+ private bool WritePackets(Stream output)
+ {
+ bool wrotePacket = false;
+ while (NextPacket(this, out byte* data, out nuint length) != 0)
+ {
+ // Packet lengths are bounded by the benchmark's image dimensions. Stream.Write consumes the
+ // borrowed bytes synchronously, without retaining a native pointer or creating a managed array.
+ output.Write(new ReadOnlySpan(data, (int)length));
+ wrotePacket = true;
+ }
+
+ return wrotePacket;
+ }
+
+ ///
+ /// Converts a native codec error into a managed benchmark failure instead of accepting invalid timing data.
+ ///
+ private static void CheckStatus(int status)
+ {
+ if (status != 0)
+ {
+ throw new InvalidOperationException(Marshal.PtrToStringUTF8(ErrorString(status)));
+ }
+ }
+
+ ///
+ /// Creates an owned opaque context; no libaom structure layout crosses the managed boundary.
+ ///
+ [LibraryImport(LibraryName, EntryPoint = "benchmark_create")]
+ private static partial int Create(uint width, uint height, uint quality, int speed, out LibaomBenchmarkEncoder encoder);
+
+ ///
+ /// Borrows three pinned planes for one synchronous native encode call.
+ ///
+ [LibraryImport(LibraryName, EntryPoint = "benchmark_encode")]
+ private static partial int EncodeFrame(LibaomBenchmarkEncoder encoder, byte* y, byte* u, byte* v, int yStride, int uvStride, long frameIndex);
+
+ ///
+ /// Signals the end of the native sequence.
+ ///
+ [LibraryImport(LibraryName, EntryPoint = "benchmark_flush")]
+ private static partial int Flush(LibaomBenchmarkEncoder encoder);
+
+ ///
+ /// Returns borrowed native packet storage and its pointer-sized length.
+ ///
+ [LibraryImport(LibraryName, EntryPoint = "benchmark_next_packet")]
+ private static partial int NextPacket(LibaomBenchmarkEncoder encoder, out byte* data, out nuint length);
+
+ ///
+ /// Destroys the context through the same native library that allocated it.
+ ///
+ [LibraryImport(LibraryName, EntryPoint = "benchmark_destroy")]
+ private static partial int Destroy(nint encoder);
+
+ ///
+ /// Returns a static UTF-8 error message owned by libaom.
+ ///
+ [LibraryImport(LibraryName, EntryPoint = "benchmark_error_string")]
+ private static partial nint ErrorString(int status);
+}
diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/CMakeLists.txt b/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/CMakeLists.txt
new file mode 100644
index 0000000000..f19a785265
--- /dev/null
+++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/CMakeLists.txt
@@ -0,0 +1,12 @@
+cmake_minimum_required(VERSION 3.20)
+project(imagesharp_aom_benchmark LANGUAGES C CXX)
+
+# Link the explicitly supplied current-main Release reference build. This adapter is benchmark-only;
+# it is never referenced by ImageSharp or copied by the production project.
+set(AOM_SOURCE_DIRECTORY "" CACHE PATH "Verified official libaom source directory")
+set(AOM_BUILD_DIRECTORY "" CACHE PATH "Matching optimized Release libaom build directory")
+add_library(imagesharp_aom_benchmark SHARED aom_benchmark.c)
+target_include_directories(imagesharp_aom_benchmark PRIVATE "${AOM_SOURCE_DIRECTORY}")
+find_library(AOM_LIBRARY NAMES aom PATHS "${AOM_BUILD_DIRECTORY}" NO_DEFAULT_PATH REQUIRED)
+target_link_libraries(imagesharp_aom_benchmark PRIVATE "${AOM_LIBRARY}")
+set_target_properties(imagesharp_aom_benchmark PROPERTIES C_STANDARD 11 LINKER_LANGUAGE CXX)
diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/aom_benchmark.c b/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/aom_benchmark.c
new file mode 100644
index 0000000000..ba5f00e154
--- /dev/null
+++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/Native/aom_benchmark.c
@@ -0,0 +1,131 @@
+// Copyright (c) Six Labors.
+// Licensed under the Six Labors Split License.
+
+#include
+#include
+#include "aom/aom_encoder.h"
+#include "aom/aomcx.h"
+
+#if defined(_WIN32)
+#define BENCHMARK_API __declspec(dllexport)
+#else
+#define BENCHMARK_API __attribute__((visibility("default")))
+#endif
+
+// Keep libaom's version-dependent structures and variadic controls entirely on the C side.
+// The managed benchmark exchanges only opaque ownership, fixed-width integers, and borrowed buffers.
+typedef struct benchmark_encoder {
+ aom_codec_ctx_t codec;
+ aom_codec_iter_t iterator;
+ unsigned int width;
+ unsigned int height;
+} benchmark_encoder;
+
+// Creates one single-threaded, unlagged, constant-quality sequence using the reference's normal tools.
+// A successful result belongs to the caller and must be released by benchmark_destroy.
+BENCHMARK_API int benchmark_create(unsigned int width, unsigned int height,
+ unsigned int quality, int speed,
+ benchmark_encoder **result) {
+ aom_codec_enc_cfg_t config;
+ aom_codec_iface_t *iface = aom_codec_av1_cx();
+ aom_codec_err_t status = aom_codec_enc_config_default(iface, &config, AOM_USAGE_GOOD_QUALITY);
+ *result = NULL;
+ if (status != AOM_CODEC_OK) return status;
+
+ benchmark_encoder *encoder = calloc(1, sizeof(*encoder));
+ if (encoder == NULL) return AOM_CODEC_MEM_ERROR;
+
+ config.g_w = width;
+ config.g_h = height;
+ config.g_threads = 1;
+ config.g_lag_in_frames = 0;
+ config.g_timebase.num = 1;
+ config.g_timebase.den = 30;
+ config.rc_end_usage = AOM_Q;
+ // Fix both bounds to the requested quantizer. CQ alone permits frame-quality boosts, whereas
+ // the managed sequence uses this same base quantizer for every frame in the comparison.
+ config.rc_min_quantizer = quality;
+ config.rc_max_quantizer = quality;
+ config.kf_mode = AOM_KF_DISABLED;
+ status = aom_codec_enc_init(&encoder->codec, iface, &config, 0);
+ if (status != AOM_CODEC_OK) {
+ free(encoder);
+ return status;
+ }
+
+ // These calls remain type-checked against the current libaom headers, including each control's argument type.
+ status = aom_codec_control(&encoder->codec, AOME_SET_CPUUSED, speed);
+ if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AOME_SET_CQ_LEVEL, quality);
+ if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_ROW_MT, 0u);
+ if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_COLOR_PRIMARIES, AOM_CICP_CP_BT_601);
+ if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_TRANSFER_CHARACTERISTICS, AOM_CICP_TC_BT_601);
+ if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_MATRIX_COEFFICIENTS, AOM_CICP_MC_BT_601);
+ if (status == AOM_CODEC_OK) status = aom_codec_control(&encoder->codec, AV1E_SET_COLOR_RANGE, AOM_CR_FULL_RANGE);
+ if (status != AOM_CODEC_OK) {
+ aom_codec_destroy(&encoder->codec);
+ free(encoder);
+ return status;
+ }
+
+ encoder->width = width;
+ encoder->height = height;
+ *result = encoder;
+ return AOM_CODEC_OK;
+}
+
+// Borrows the already converted planes for this synchronous encode call. The caller pins all three
+// pointers until it returns; libaom retains its own reference pictures, never these managed input pointers.
+BENCHMARK_API int benchmark_encode(benchmark_encoder *encoder, unsigned char *y,
+ unsigned char *u, unsigned char *v,
+ int y_stride, int uv_stride, int64_t frame_index) {
+ aom_image_t image;
+ if (aom_img_wrap(&image, AOM_IMG_FMT_I420, encoder->width, encoder->height, 1, y) == NULL) {
+ return AOM_CODEC_INVALID_PARAM;
+ }
+
+ // ImageSharp's source planes include aligned borders. Describe those existing rows directly instead
+ // of flattening them into another contiguous YUV allocation before calling the reference encoder.
+ image.planes[AOM_PLANE_Y] = y;
+ image.planes[AOM_PLANE_U] = u;
+ image.planes[AOM_PLANE_V] = v;
+ image.stride[AOM_PLANE_Y] = y_stride;
+ image.stride[AOM_PLANE_U] = uv_stride;
+ image.stride[AOM_PLANE_V] = uv_stride;
+ encoder->iterator = NULL;
+ return aom_codec_encode(&encoder->codec, &image, frame_index, 1, 0);
+}
+
+// Drains delayed output at the end of the sequence even though lookahead is disabled.
+BENCHMARK_API int benchmark_flush(benchmark_encoder *encoder) {
+ encoder->iterator = NULL;
+ return aom_codec_encode(&encoder->codec, NULL, -1, 1, 0);
+}
+
+// The returned packet storage belongs to libaom and is valid only until the next codec call.
+// The managed side copies it directly into the same kind of destination stream as its own encoder.
+BENCHMARK_API int benchmark_next_packet(benchmark_encoder *encoder, const void **data, size_t *length) {
+ const aom_codec_cx_pkt_t *packet;
+ while ((packet = aom_codec_get_cx_data(&encoder->codec, &encoder->iterator)) != NULL) {
+ if (packet->kind == AOM_CODEC_CX_FRAME_PKT) {
+ *data = packet->data.frame.buf;
+ *length = packet->data.frame.sz;
+ return 1;
+ }
+ }
+
+ *data = NULL;
+ *length = 0;
+ return 0;
+}
+
+// Releases exactly the native context allocated by benchmark_create.
+BENCHMARK_API int benchmark_destroy(benchmark_encoder *encoder) {
+ aom_codec_err_t status = aom_codec_destroy(&encoder->codec);
+ free(encoder);
+ return status;
+}
+
+// Libaom owns this static UTF-8 error string; the caller must not free it.
+BENCHMARK_API const char *benchmark_error_string(int status) {
+ return aom_codec_err_to_string((aom_codec_err_t)status);
+}
diff --git a/tests/ImageSharp.Benchmarks/Codecs/Heif/README.md b/tests/ImageSharp.Benchmarks/Codecs/Heif/README.md
new file mode 100644
index 0000000000..5bc9e1d613
--- /dev/null
+++ b/tests/ImageSharp.Benchmarks/Codecs/Heif/README.md
@@ -0,0 +1,137 @@
+# AV1 sequence encoder comparison
+
+`Av1SequenceEncoderBenchmarks` compares the managed AV1 sequence encoder with an optimized build of official libaom `main`.
+The native adapter belongs only to this benchmark project. ImageSharp production code remains fully managed.
+
+## Measurement boundary
+
+Both methods start with the same three `Rgb24` photographic frames and finish with the complete raw AV1 OBU sequence in a reused `MemoryStream`.
+The measured operation includes encoder/workspace construction, RGB-to-YUV conversion for every frame, encoding, output writes, flushing where required, and disposal.
+Both use the existing ImageSharp SIMD color converter and write directly into the aligned source planes consumed by their encoder.
+Libaom receives pinned plane pointers for each synchronous encode call; it does not read a preconverted input file.
+
+Image loading, resizing, half-pixel source translations, validation-file writes, decoding, and quality calculations are outside both measured methods.
+MemoryStream capacity is retained between operations for both methods. These are RGB-to-OBU measurements, not public HEIF container-save measurements.
+Do not compare them with `aomenc`'s internal encode-time report, which excludes the conversion boundary.
+
+The source is the existing `TestImages.Png.Bike` photograph. Frames one and two are independently translated by half a pixel and one pixel on both axes.
+The parameter matrix contains 256x256 and 512x512 frames at ImageSharp efforts seven, eight, and nine.
+Input is full-range BT.601, eight-bit 4:2:0; each sequence contains one key frame and two dependent frames.
+Both codecs use one coding thread. Libaom uses good-quality mode, no lookahead, disabled automatic key frames, and `cpu-used=6`.
+That speed is an independent reference setting, not a mapping from the ImageSharp effort scale.
+
+ImageSharp's base quantizer index is 120. Libaom's public quantizer 30 maps to that same index in the pinned reference.
+The native minimum and maximum quantizers are both fixed to 30, in addition to the CQ setting, so frame-level CQ boosts cannot lower the base quantizer.
+Other native coding tools remain enabled. Identical quantizers do not imply identical quality or bitrate; always retain size and decoded-quality results with timings.
+
+BenchmarkDotNet's allocation column measures managed allocations only. It does not measure libaom's native allocations, pooled native memory, or total peak memory.
+Do not interpret its allocation ratio as a whole-encoder memory comparison.
+
+## Build the reference and benchmark adapter
+
+The verified source revision is `d565eec60f084421fa34fc0534b760c6452b6a6c`, libaom 3.15.0, exported at `D:\GitHub\ynse01\aom-d565eec6-source`.
+The x64 Release build is `D:\GitHub\ynse01\aom-d565eec6-build-x64-release`.
+It uses MSVC 19.51.36256.0 and NASM 3.02, runtime CPU detection, and SSE2, SSE4.1, AVX2, and AVX512 kernels.
+The generic reference decoder build is suitable for conformance but must not be used as the timing reference.
+
+Run from the repository root in an x64 Visual Studio developer PowerShell with the existing CMake, Ninja, NASM, and Perl installations available:
+
+```powershell
+$aomSourceDirectory = 'D:/GitHub/ynse01/aom-d565eec6-source'
+$aomBuildDirectory = 'D:/GitHub/ynse01/aom-d565eec6-build-x64-release'
+$adapterBuildDirectory = 'artifacts/av1-native-benchmark-release'
+
+cmake -S $aomSourceDirectory -B $aomBuildDirectory -G Ninja `
+ -DCMAKE_BUILD_TYPE=Release -DAOM_TARGET_CPU=x86_64 -DENABLE_NASM=ON `
+ -DENABLE_TESTS=OFF -DENABLE_EXAMPLES=ON -DENABLE_TOOLS=ON `
+ -DCONFIG_WEBM_IO=OFF -DCONFIG_LIBYUV=OFF
+if ($LASTEXITCODE -ne 0) { throw 'Reference configuration failed.' }
+
+cmake --build $aomBuildDirectory --target aomenc aomdec --parallel 4
+if ($LASTEXITCODE -ne 0) { throw 'Reference build failed.' }
+
+cmake -S tests/ImageSharp.Benchmarks/Codecs/Heif/Native -B $adapterBuildDirectory -G Ninja `
+ -DCMAKE_BUILD_TYPE=Release "-DAOM_SOURCE_DIRECTORY=$aomSourceDirectory" "-DAOM_BUILD_DIRECTORY=$aomBuildDirectory"
+if ($LASTEXITCODE -ne 0) { throw 'Adapter configuration failed.' }
+
+cmake --build $adapterBuildDirectory --parallel 1
+if ($LASTEXITCODE -ne 0) { throw 'Adapter build failed.' }
+
+dotnet build tests/ImageSharp.Benchmarks/ImageSharp.Benchmarks.csproj -c Release -f net11.0 `
+ --no-restore --disable-build-servers -m:1 -p:UseSharedCompilation=false `
+ -p:SIXLABORS_TESTING_PREVIEW=true -p:SIXLABORS_DISABLE_CONFIG_COPY=true
+if ($LASTEXITCODE -ne 0) { throw 'Managed build failed.' }
+
+Copy-Item -LiteralPath "$adapterBuildDirectory/imagesharp_aom_benchmark.dll" `
+ -Destination artifacts/bin/tests/ImageSharp.Benchmarks/Release/net11.0/imagesharp_aom_benchmark.dll
+```
+
+The adapter links the explicit reference build. It neither downloads tools nor adds a native production dependency.
+Rebuild it whenever the source headers or reference library change; record the exact source revision and optimized build configuration with every report.
+
+## Validate, then measure
+
+Run one pair first, in the existing benchmark host with the in-process toolchain. Do not run builds or tests concurrently with timing.
+
+```powershell
+$benchmarkAssembly = 'artifacts/bin/tests/ImageSharp.Benchmarks/Release/net11.0/ImageSharp.Benchmarks.dll'
+$benchmarkFilter = '*Av1SequenceEncoderBenchmarks.*(Dimension: 256, Effort: 7)'
+
+dotnet $benchmarkAssembly --inProcess --job Dry --filter $benchmarkFilter `
+ --stopOnFirstError --noOverwrite --artifacts artifacts/BenchmarkDotNet/av1-sequence-rgb-dry
+```
+
+The actual last measured payload for each method and the independently exported source planes are retained under
+`tests/Images/ActualOutput/Heif/Av1/Av1SequenceEncoderBenchmarks`.
+Both timed methods convert RGB on every invocation; the exported source file exists only for offline quality calculation.
+Decode both payloads using the matching current-main `aomdec`, and require the expected complete three-frame YUV length:
+
+```powershell
+$outputDirectory = 'tests/Images/ActualOutput/Heif/Av1/Av1SequenceEncoderBenchmarks'
+foreach ($payloadName in 'bike-256-q120-effort7.obu', 'bike-256-q120-libaom-cpu6.obu') {
+ $payloadPath = Join-Path $outputDirectory $payloadName
+ & "$aomBuildDirectory/aomdec.exe" --codec=av1 --threads=1 --rawvideo -o "$payloadPath.yuv" $payloadPath
+ if ($LASTEXITCODE -ne 0) { throw "Reference decode failed: $payloadName" }
+
+ if ((Get-Item -LiteralPath "$payloadPath.yuv").Length -ne (3 * 256 * 256 * 3 / 2)) {
+ throw "Incomplete decoded sequence: $payloadName"
+ }
+}
+
+dotnet $benchmarkAssembly --inProcess --job Short --filter $benchmarkFilter `
+ --stopOnFirstError --noOverwrite --artifacts artifacts/BenchmarkDotNet/av1-sequence-rgb-short
+```
+
+Repeat output validation after the measured run. Compute per-plane PSNR against the exported source using
+`10 * log10(255^2 * sampleCount / squaredErrorSum)`, accumulated over all three frames.
+Aggregate YUV PSNR uses the total squared error and sample count, not the arithmetic mean of the three PSNR values.
+Keep bitstream length, quality, absolute time, runtime, CPU information, reference settings, and any environment warnings together.
+The short job is an initial diagnostic checkpoint, not a substitute for the complete size/quality/performance matrix or an equal-quality rate-distortion comparison.
+
+## Verified checkpoint: 2026-09-05
+
+The final fixed-quantizer pair ran with BenchmarkDotNet 0.15.8, .NET 11.0.0-preview.7.26381.103, x64 RyuJIT, and AVX512 available.
+The in-process Short job used three warmup iterations and three measured iterations. Each operation encoded all three 256x256 frames.
+
+| Encoder setting | Mean sequence time | OBU size | Y PSNR | U PSNR | V PSNR | Aggregate YUV PSNR |
+| --- | ---: | ---: | ---: | ---: | ---: | ---: |
+| ImageSharp effort 7, base index 120 | 2,053.31 ms | 11.54 KiB | 39.040 dB | 38.630 dB | 32.778 dB | 37.124 dB |
+| Libaom cpu-used 6, fixed public quantizer 30 | 70.67 ms | 8.27 KiB | 39.275 dB | 39.637 dB | 37.351 dB | 38.942 dB |
+
+Both final measured payloads decode to exactly three complete native YUV frames using the optimized current-main reference decoder.
+ImageSharp is about 29 times slower for this case while producing a larger, lower-PSNR sequence. The performance exit gate remains open.
+The managed allocation counter reports 9.38 MiB per ImageSharp operation; its source still needs attribution, and it is not comparable with libaom's unmeasured native footprint.
+
+Evidence is retained in `artifacts/BenchmarkDotNet/av1-sequence-rgb-fixed-q-short-20260905/20260905-131149`.
+The initial Dry run and earlier unrestricted-CQ Short run are preliminary checks, not the final comparison above.
+BenchmarkDotNet could not change the power plan or query CPU model information in this environment; those warnings remain in the log.
+This result is a local diagnostic, not a controlled-hardware or equal-quality performance claim.
+
+Final OBU SHA-256 values:
+
+- ImageSharp: `AB0327D33405FB7A6EC1895D465C9EE08E0B218DDA9D7E8EE7E0DCA6C5F8A73B`
+- Libaom: `0C44A3475849762B5A09937397DEA4397264B83E79DAF930A06B70734977E4A9`
+
+The final Release benchmark build has zero errors and 39 existing benchmark warnings outside the new files.
+Roslyn compiler and analyzer diagnostics contain no errors or warnings in the new benchmark files.
+No production file changed in this benchmark checkpoint; the preceding 2,524-case encoder/entropy verification remains the latest production test run.