Commit ef1a201f for libheif
commit ef1a201f78982eb3303334292dd53d9ced1fb178
Author: Dirk Farin <dirk.farin@gmail.com>
Date: Sun Oct 4 00:39:17 2026 +0200
Factor the 4:2:0 bilinear chroma upsampling into a per-plane helper
The Cb and Cr planes were processed by two copies of every border loop and of
the inner loop, with the index arithmetic repeated in each of the ~150 character
lines. This is how the border indexing bug fixed two commits ago could hide.
Move the upsampling of one plane into upsample_chroma_plane_bilinear() and
express the four borders through a row and a column lambda, so that each index
expression and the 3/4:1/4 interpolation appear only once. The arithmetic and
the output are unchanged.
diff --git a/libheif/color-conversion/chroma_sampling.cc b/libheif/color-conversion/chroma_sampling.cc
index a38139e2..a6b3620f 100644
--- a/libheif/color-conversion/chroma_sampling.cc
+++ b/libheif/color-conversion/chroma_sampling.cc
@@ -478,6 +478,98 @@ Op_YCbCr420_bilinear_to_YCbCr444<Pixel>::state_after_conversion(const ColorState
}
+
+/*
+ * Upsample one 4:2:0 chroma plane to the full image resolution.
+ * Strides are given in units of Pixel.
+ *
+ * We assume that chroma pixels are located in the center of 2x2 luma pixels.
+ * The image border 'b' is handled separately.
+ * The right and bottom border are not processed when the size is odd.
+ * Then, each 2x2 square between 4 chroma samples is computed in one iteration.
+ *
+ * Upsampling weights are 3/4, 1/4. For example:
+ * A = 3/4*3/4 * C1 + 3/4*1/4 * C2 + 1/4*3/4 * C3 + 1/4*1/4 * C4
+ *
+ * +---+---+---+---+
+ * | b | b | b | b |
+ * +---C1--+---C2--+
+ * | b | A | | b |
+ * +---+---+---+---+
+ * | b | | | b |
+ * +---C3--+---C4--+
+ * | b | b | b | b |
+ * +---+---+---+---+
+ */
+template<class Pixel>
+static void upsample_chroma_plane_bilinear(const Pixel* in, size_t in_stride,
+ Pixel* out, size_t out_stride,
+ uint32_t width, uint32_t height)
+{
+ // interpolate between two chroma samples with weights 3/4 and 1/4
+ auto mix_3_1 = [](Pixel a, Pixel b) { return (Pixel) ((3 * a + 1 * b + 2) / 4); };
+
+ // --- fill borders
+
+ // Fill one image row from one chroma row, including both corner pixels.
+ auto upsample_row = [&](const Pixel* in_row, Pixel* out_row) {
+ out_row[0] = in_row[0];
+
+ for (uint32_t cx = 0; cx < (width - 1) / 2; cx++) {
+ Pixel a = in_row[cx];
+ Pixel b = in_row[cx + 1];
+ out_row[2 * cx + 1] = mix_3_1(a, b);
+ out_row[2 * cx + 2] = mix_3_1(b, a);
+ }
+
+ if (width % 2 == 0) {
+ out_row[width - 1] = in_row[width / 2 - 1];
+ }
+ };
+
+ // Fill one image column from one chroma column. The corner pixels are already set by upsample_row().
+ auto upsample_column = [&](const Pixel* in_column, Pixel* out_column) {
+ for (uint32_t cy = 0; cy < (height - 1) / 2; cy++) {
+ Pixel a = in_column[cy * in_stride];
+ Pixel b = in_column[(cy + 1) * in_stride];
+ out_column[(2 * cy + 1) * out_stride] = mix_3_1(a, b);
+ out_column[(2 * cy + 2) * out_stride] = mix_3_1(b, a);
+ }
+ };
+
+ upsample_row(in, out); // top border
+ upsample_column(in, out); // left border
+
+ if (width % 2 == 0) {
+ upsample_column(in + width / 2 - 1, out + width - 1); // right border
+ }
+
+ if (height % 2 == 0) {
+ upsample_row(in + (height / 2 - 1) * in_stride, out + (height - 1) * out_stride); // bottom border
+ }
+
+
+ // --- bilinear filtering of inner part
+
+ for (uint32_t y = 1; y < height - 1; y += 2) {
+ for (uint32_t x = 1; x < width - 1; x += 2) {
+ uint32_t cx = x / 2;
+ uint32_t cy = y / 2;
+
+ Pixel c00 = in[cy * in_stride + cx];
+ Pixel c01 = in[cy * in_stride + cx + 1];
+ Pixel c10 = in[(cy + 1) * in_stride + cx];
+ Pixel c11 = in[(cy + 1) * in_stride + cx + 1];
+
+ out[(y + 0) * out_stride + x + 0] = (Pixel) ((c00 * 3 * 3 + c01 * 1 * 3 + c10 * 3 * 1 + c11 * 1 * 1 + 8) / 16);
+ out[(y + 0) * out_stride + x + 1] = (Pixel) ((c00 * 1 * 3 + c01 * 3 * 3 + c10 * 1 * 1 + c11 * 3 * 1 + 8) / 16);
+ out[(y + 1) * out_stride + x + 0] = (Pixel) ((c00 * 3 * 1 + c01 * 1 * 1 + c10 * 3 * 3 + c11 * 1 * 3 + 8) / 16);
+ out[(y + 1) * out_stride + x + 1] = (Pixel) ((c00 * 1 * 1 + c01 * 3 * 1 + c10 * 1 * 3 + c11 * 3 * 3 + 8) / 16);
+ }
+ }
+}
+
+
template<class Pixel>
Result<std::shared_ptr<HeifPixelImage>>
Op_YCbCr420_bilinear_to_YCbCr444<Pixel>::convert_colorspace(const std::shared_ptr<const HeifPixelImage>& input,
@@ -564,119 +656,12 @@ Op_YCbCr420_bilinear_to_YCbCr444<Pixel>::convert_colorspace(const std::shared_pt
out_cb_stride /= sizeof(Pixel);
out_cr_stride /= sizeof(Pixel);
- /*
- * We assume that chroma pixels are located in the center of 2x2 luma pixels.
- * The image border 'b' is handled separately.
- * The right and bottom border are not processed when the size is odd.
- * Then, each 2x2 square between 4 chroma samples is computed in one iteration.
- *
- * Upsampling weights are 3/4, 1/4. For example:
- * A = 3/4*3/4 * C1 + 3/4*1/4 * C2 + 1/4*3/4 * C3 + 1/4*1/4 * C4
- *
- * +---+---+---+---+
- * | b | b | b | b |
- * +---C1--+---C2--+
- * | b | A | | b |
- * +---+---+---+---+
- * | b | | | b |
- * +---C3--+---C4--+
- * | b | b | b | b |
- * +---+---+---+---+
- */
-
- // --- fill borders
-
- // top left corner
- out_cb[0] = in_cb[0];
- out_cr[0] = in_cr[0];
-
- // top border
- for (uint32_t cx = 0; cx < (width - 1) / 2; cx++) {
- out_cb[0 * out_cb_stride + 2 * cx + 1] = (Pixel) ((3 * in_cb[cx] + 1 * in_cb[cx + 1] + 2) / 4);
- out_cb[0 * out_cb_stride + 2 * cx + 2] = (Pixel) ((1 * in_cb[cx] + 3 * in_cb[cx + 1] + 2) / 4);
- out_cr[0 * out_cr_stride + 2 * cx + 1] = (Pixel) ((3 * in_cr[cx] + 1 * in_cr[cx + 1] + 2) / 4);
- out_cr[0 * out_cr_stride + 2 * cx + 2] = (Pixel) ((1 * in_cr[cx] + 3 * in_cr[cx + 1] + 2) / 4);
- }
-
- // top right corner
- if (width % 2 == 0) {
- out_cb[width - 1] = in_cb[width / 2 - 1];
- out_cr[width - 1] = in_cr[width / 2 - 1];
- }
-
- // left border
- for (uint32_t cy = 0; cy < (height - 1) / 2; cy++) {
- out_cb[(2 * cy + 1) * out_cb_stride + 0] = (Pixel) ((3 * in_cb[cy * in_cb_stride] + 1 * in_cb[(cy + 1) * in_cb_stride] + 2) / 4);
- out_cb[(2 * cy + 2) * out_cb_stride + 0] = (Pixel) ((1 * in_cb[cy * in_cb_stride] + 3 * in_cb[(cy + 1) * in_cb_stride] + 2) / 4);
- out_cr[(2 * cy + 1) * out_cr_stride + 0] = (Pixel) ((3 * in_cr[cy * in_cr_stride] + 1 * in_cr[(cy + 1) * in_cr_stride] + 2) / 4);
- out_cr[(2 * cy + 2) * out_cr_stride + 0] = (Pixel) ((1 * in_cr[cy * in_cr_stride] + 3 * in_cr[(cy + 1) * in_cr_stride] + 2) / 4);
- }
-
- // bottom left corner
- if (height % 2 == 0) {
- out_cb[(height - 1) * out_cb_stride] = in_cb[(height / 2 - 1) * in_cb_stride];
- out_cr[(height - 1) * out_cr_stride] = in_cr[(height / 2 - 1) * in_cr_stride];
- }
-
- // right border
- if (width % 2 == 0) {
- for (uint32_t cy = 0; cy < (height - 1) / 2; cy++) {
- out_cb[(2 * cy + 1) * out_cb_stride + width - 1] = (Pixel) ((3 * in_cb[cy * in_cb_stride + width / 2 - 1] + 1 * in_cb[(cy + 1) * in_cb_stride + width / 2 - 1] + 2) / 4);
- out_cb[(2 * cy + 2) * out_cb_stride + width - 1] = (Pixel) ((1 * in_cb[cy * in_cb_stride + width / 2 - 1] + 3 * in_cb[(cy + 1) * in_cb_stride + width / 2 - 1] + 2) / 4);
- out_cr[(2 * cy + 1) * out_cr_stride + width - 1] = (Pixel) ((3 * in_cr[cy * in_cr_stride + width / 2 - 1] + 1 * in_cr[(cy + 1) * in_cr_stride + width / 2 - 1] + 2) / 4);
- out_cr[(2 * cy + 2) * out_cr_stride + width - 1] = (Pixel) ((1 * in_cr[cy * in_cr_stride + width / 2 - 1] + 3 * in_cr[(cy + 1) * in_cr_stride + width / 2 - 1] + 2) / 4);
- }
- }
-
- // bottom border
- if (height % 2 == 0) {
- for (uint32_t cx = 0; cx < (width - 1) / 2; cx++) {
- out_cb[(height - 1) * out_cb_stride + 2 * cx + 1] = (Pixel) ((3 * in_cb[(height / 2 - 1) * in_cb_stride + cx] + 1 * in_cb[(height / 2 - 1) * in_cb_stride + cx + 1] + 2) / 4);
- out_cb[(height - 1) * out_cb_stride + 2 * cx + 2] = (Pixel) ((1 * in_cb[(height / 2 - 1) * in_cb_stride + cx] + 3 * in_cb[(height / 2 - 1) * in_cb_stride + cx + 1] + 2) / 4);
- out_cr[(height - 1) * out_cr_stride + 2 * cx + 1] = (Pixel) ((3 * in_cr[(height / 2 - 1) * in_cr_stride + cx] + 1 * in_cr[(height / 2 - 1) * in_cr_stride + cx + 1] + 2) / 4);
- out_cr[(height - 1) * out_cr_stride + 2 * cx + 2] = (Pixel) ((1 * in_cr[(height / 2 - 1) * in_cr_stride + cx] + 3 * in_cr[(height / 2 - 1) * in_cr_stride + cx + 1] + 2) / 4);
- }
- }
-
- // bottom right corner
- if (width % 2 == 0 && height % 2 == 0) {
- out_cb[(height - 1) * out_cb_stride + width - 1] = in_cb[(height / 2 - 1) * in_cb_stride + width / 2 - 1];
- out_cr[(height - 1) * out_cr_stride + width - 1] = in_cr[(height / 2 - 1) * in_cr_stride + width / 2 - 1];
- }
-
-
- // --- bilinear filtering of inner part
-
- uint32_t x, y;
- for (y = 1; y < height - 1; y += 2) {
- for (x = 1; x < width - 1; x += 2) {
- uint32_t cx = x / 2;
- uint32_t cy = y / 2;
-
- Pixel cb00 = in_cb[cy * in_cb_stride + cx];
- Pixel cr00 = in_cr[cy * in_cr_stride + cx];
- Pixel cb01 = in_cb[cy * in_cb_stride + cx + 1];
- Pixel cr01 = in_cr[cy * in_cr_stride + cx + 1];
- Pixel cb10 = in_cb[(cy + 1) * in_cb_stride + cx];
- Pixel cr10 = in_cr[(cy + 1) * in_cr_stride + cx];
- Pixel cb11 = in_cb[(cy + 1) * in_cb_stride + cx + 1];
- Pixel cr11 = in_cr[(cy + 1) * in_cr_stride + cx + 1];
-
- out_cb[(y + 0) * out_cb_stride + x + 0] = (Pixel) ((cb00 * 3 * 3 + cb01 * 1 * 3 + cb10 * 3 * 1 + cb11 * 1 * 1 + 8) / 16);
- out_cb[(y + 0) * out_cb_stride + x + 1] = (Pixel) ((cb00 * 1 * 3 + cb01 * 3 * 3 + cb10 * 1 * 1 + cb11 * 3 * 1 + 8) / 16);
- out_cb[(y + 1) * out_cb_stride + x + 0] = (Pixel) ((cb00 * 3 * 1 + cb01 * 1 * 1 + cb10 * 3 * 3 + cb11 * 1 * 3 + 8) / 16);
- out_cb[(y + 1) * out_cb_stride + x + 1] = (Pixel) ((cb00 * 1 * 1 + cb01 * 3 * 1 + cb10 * 1 * 3 + cb11 * 3 * 3 + 8) / 16);
-
- out_cr[(y + 0) * out_cr_stride + x + 0] = (Pixel) ((cr00 * 3 * 3 + cr01 * 1 * 3 + cr10 * 3 * 1 + cr11 * 1 * 1 + 8) / 16);
- out_cr[(y + 0) * out_cr_stride + x + 1] = (Pixel) ((cr00 * 1 * 3 + cr01 * 3 * 3 + cr10 * 1 * 1 + cr11 * 3 * 1 + 8) / 16);
- out_cr[(y + 1) * out_cr_stride + x + 0] = (Pixel) ((cr00 * 3 * 1 + cr01 * 1 * 1 + cr10 * 3 * 3 + cr11 * 1 * 3 + 8) / 16);
- out_cr[(y + 1) * out_cr_stride + x + 1] = (Pixel) ((cr00 * 1 * 1 + cr01 * 3 * 1 + cr10 * 1 * 3 + cr11 * 3 * 3 + 8) / 16);
- }
- }
+ upsample_chroma_plane_bilinear(in_cb, in_cb_stride, out_cb, out_cb_stride, width, height);
+ upsample_chroma_plane_bilinear(in_cr, in_cr_stride, out_cr, out_cr_stride, width, height);
// TODO: check whether we can use HeifPixelImage::transfer_channel_from_image_as() instead of copying Y and Alpha
- for (y = 0; y < height; y++) {
+ for (uint32_t y = 0; y < height; y++) {
size_t copyWidth = static_cast<size_t>(width) * sizeof(Pixel);
memcpy(&out_y[y * out_y_stride], &in_y[y * in_y_stride], copyWidth);