gdk: optimize native big-endian RGB565 paths

Add native big-endian four-pixel converters for RGB565, grayscale RGB565, and dithered RGB565.

Select the block converters only where cross-target code generation is favorable: RGB on PowerPC, s390x, ARMv5+, and AArch64; grayscale on PowerPC, ARMv5+, AArch64, and m68k; and dither on PowerPC. Retain the scalar fallbacks everywhere else.

The little-endian machine code remains byte-for-byte unchanged. No runtime endian checks, allocations, or additional image buffers are introduced.

Validated with a complete build and test suite, deterministic and randomized big-endian formula tests, exact little-endian .text comparison, and cross-target code-generation checks.
This commit is contained in:
Daemonratte 2026-08-04 17:55:45 +02:00
commit be64a902a5

View file

@ -1464,19 +1464,33 @@ gdk_rgb_convert_gray8_gray (GdkRgbInfo *image_info, GdkImage *image,
}
}
#if G_BYTE_ORDER == G_LITTLE_ENDIAN
#define HAIRY_CONVERT_565
#if G_BYTE_ORDER == G_BIG_ENDIAN
#if defined(__powerpc__) || defined(__powerpc64__) || \
defined(__ppc__) || defined(__ppc64__) || \
defined(__PPC__) || defined(__PPC64__) || \
defined(__POWERPC__) || defined(_ARCH_PPC)
#define GDK_RGB_FAST_BE565_RGB
#define GDK_RGB_FAST_BE565_GRAY
#define GDK_RGB_FAST_BE565_DITHER
#elif defined(__s390x__)
#define GDK_RGB_FAST_BE565_RGB
#elif defined(__aarch64__)
#define GDK_RGB_FAST_BE565_RGB
#define GDK_RGB_FAST_BE565_GRAY
#elif defined(__arm__) && defined(__ARM_ARCH) && \
__ARM_ARCH >= 5
#define GDK_RGB_FAST_BE565_RGB
#define GDK_RGB_FAST_BE565_GRAY
#elif defined(__m68k__)
#define GDK_RGB_FAST_BE565_GRAY
#endif
#endif
#ifdef HAIRY_CONVERT_565
/* Render a 24-bit RGB image in buf into the GdkImage, without dithering.
This assumes native byte ordering - what should really be done is to
check whether the image byte_order is consistent with the _ENDIAN
config flag, and if not, use a different function.
This one is even faster than the one below - its inner loop loads 3
words (i.e. 4 24-bit pixels), does a lot of shifting and masking,
then writes 2 words. */
The four-pixel inner loop uses compile-time-specialized native word
layouts for little- and big-endian systems. */
#if G_BYTE_ORDER == G_LITTLE_ENDIAN || defined(GDK_RGB_FAST_BE565_RGB)
static void
gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
gint x0, gint y0, gint width, gint height,
@ -1520,6 +1534,7 @@ gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
r1b0g0r0 = ((guint32 *)bp2)[0];
g2r2b1g1 = ((guint32 *)bp2)[1];
b3g3r3b2 = ((guint32 *)bp2)[2];
#if G_BYTE_ORDER == G_LITTLE_ENDIAN
((guint32 *)obptr)[0] =
((r1b0g0r0 & 0xf8) << 8) |
((r1b0g0r0 & 0xfc00) >> 5) |
@ -1534,6 +1549,22 @@ gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
((b3g3r3b2 & 0xf800) << 16) |
((b3g3r3b2 & 0xfc0000) << 3) |
((b3g3r3b2 & 0xf8000000) >> 11);
#else
((guint32 *)obptr)[0] =
(r1b0g0r0 & 0xf8000000) |
((r1b0g0r0 & 0x00fc0000) << 3) |
((r1b0g0r0 & 0x0000f800) << 5) |
((r1b0g0r0 & 0x000000f8) << 8) |
((g2r2b1g1 & 0xfc000000) >> 21) |
((g2r2b1g1 & 0x00f80000) >> 19);
((guint32 *)obptr)[1] =
((g2r2b1g1 & 0x0000f800) << 16) |
((g2r2b1g1 & 0x000000fc) << 19) |
((b3g3r3b2 & 0xf8000000) >> 11) |
((b3g3r3b2 & 0x00f80000) >> 8) |
((b3g3r3b2 & 0x0000fc00) >> 5) |
((b3g3r3b2 & 0x000000f8) >> 3);
#endif
bp2 += 12;
obptr += 8;
}
@ -1553,27 +1584,6 @@ gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
}
}
#else
/* Render a 24-bit RGB image in buf into the GdkImage, without dithering.
This assumes native byte ordering - what should really be done is to
check whether the image byte_order is consistent with the _ENDIAN
config flag, and if not, use a different function.
This routine is faster than the one included with Gtk 1.0 for a number
of reasons:
1. Shifting instead of lookup tables (less memory traffic).
2. Much less register pressure, especially because shifts are
in the code.
3. A memcpy is avoided (i.e. the transfer function).
4. On big-endian architectures, byte swapping is avoided.
That said, it wouldn't be hard to make it even faster - just make an
inner loop that reads 3 words (i.e. 4 24-bit pixels), does a lot of
shifting and masking, then writes 2 words.
*/
static void
gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
gint x0, gint y0, gint width, gint height,
@ -1607,7 +1617,7 @@ gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
}
#endif
#ifdef HAIRY_CONVERT_565
#if G_BYTE_ORDER == G_LITTLE_ENDIAN || defined(GDK_RGB_FAST_BE565_GRAY)
static void
gdk_rgb_convert_565_gray (GdkRgbInfo *image_info, GdkImage *image,
gint x0, gint y0, gint width, gint height,
@ -1645,6 +1655,7 @@ gdk_rgb_convert_565_gray (GdkRgbInfo *image_info, GdkImage *image,
guint32 g3g2g1g0;
g3g2g1g0 = ((guint32 *)bp2)[0];
#if G_BYTE_ORDER == G_LITTLE_ENDIAN
((guint32 *)obptr)[0] =
((g3g2g1g0 & 0xf8) << 8) |
((g3g2g1g0 & 0xfc) << 3) |
@ -1659,6 +1670,22 @@ gdk_rgb_convert_565_gray (GdkRgbInfo *image_info, GdkImage *image,
(g3g2g1g0 & 0xf8000000) |
((g3g2g1g0 & 0xfc000000) >> 5) |
((g3g2g1g0 & 0xf8000000) >> 11);
#else
((guint32 *)obptr)[0] =
(g3g2g1g0 & 0xf8000000) |
((g3g2g1g0 & 0xfc000000) >> 5) |
((g3g2g1g0 & 0xf8000000) >> 11) |
((g3g2g1g0 & 0x00f80000) >> 8) |
((g3g2g1g0 & 0x00fc0000) >> 13) |
((g3g2g1g0 & 0x00f80000) >> 19);
((guint32 *)obptr)[1] =
((g3g2g1g0 & 0x0000f800) << 16) |
((g3g2g1g0 & 0x0000fc00) << 11) |
((g3g2g1g0 & 0x0000f800) << 5) |
((g3g2g1g0 & 0x000000f8) << 8) |
((g3g2g1g0 & 0x000000fc) << 3) |
((g3g2g1g0 & 0x000000f8) >> 3);
#endif
bp2 += 4;
obptr += 8;
}
@ -1745,7 +1772,7 @@ gdk_rgb_convert_565_br (GdkRgbInfo *image_info, GdkImage *image,
/* Thanks to Ray Lehtiniemi for a patch that resulted in a ~25% speedup
in this mode. */
#ifdef HAIRY_CONVERT_565
#if G_BYTE_ORDER == G_LITTLE_ENDIAN || defined(GDK_RGB_FAST_BE565_DITHER)
static void
gdk_rgb_convert_565_d (GdkRgbInfo *image_info, GdkImage *image,
gint x0, gint y0, gint width, gint height,
@ -1800,6 +1827,7 @@ gdk_rgb_convert_565_d (GdkRgbInfo *image_info, GdkImage *image,
r1b0g0r0 = ((guint32 *)bp2)[0];
g2r2b1g1 = ((guint32 *)bp2)[1];
b3g3r3b2 = ((guint32 *)bp2)[2];
#if G_BYTE_ORDER == G_LITTLE_ENDIAN
rgb02 =
((r1b0g0r0 & 0xff) << 20) +
((r1b0g0r0 & 0xff00) << 2) +
@ -1846,6 +1874,54 @@ gdk_rgb_convert_565_d (GdkRgbInfo *image_info, GdkImage *image,
((rgb13 & 0x0f800000) << 4) |
((rgb13 & 0x0003f000) << 9) |
((rgb13 & 0x000000f8) << 13);
#else
rgb02 =
((r1b0g0r0 & 0xff000000) >> 4) +
((r1b0g0r0 & 0x00ff0000) >> 6) +
((r1b0g0r0 & 0x0000ff00) >> 8) +
dmp[x & (DM_WIDTH - 1)];
rgb02 += 0x10040100
- ((rgb02 & 0x1e0001e0) >> 5)
- ((rgb02 & 0x00070000) >> 6);
rgb13 =
((r1b0g0r0 & 0x000000ff) << 20) +
((g2r2b1g1 & 0xff000000) >> 14) +
((g2r2b1g1 & 0x00ff0000) >> 16) +
dmp[(x + 1) & (DM_WIDTH - 1)];
rgb13 += 0x10040100
- ((rgb13 & 0x1e0001e0) >> 5)
- ((rgb13 & 0x00070000) >> 6);
((guint32 *)obptr)[0] =
((rgb02 & 0x0f800000) << 4) |
((rgb02 & 0x0003f000) << 9) |
((rgb02 & 0x000000f8) << 13) |
((rgb13 & 0x0f800000) >> 12) |
((rgb13 & 0x0003f000) >> 7) |
((rgb13 & 0x000000f8) >> 3);
rgb02 =
((g2r2b1g1 & 0x0000ff00) << 12) +
((g2r2b1g1 & 0x000000ff) << 10) +
((b3g3r3b2 & 0xff000000) >> 24) +
dmp[(x + 2) & (DM_WIDTH - 1)];
rgb02 += 0x10040100
- ((rgb02 & 0x1e0001e0) >> 5)
- ((rgb02 & 0x00070000) >> 6);
rgb13 =
((b3g3r3b2 & 0x00ff0000) << 4) +
((b3g3r3b2 & 0x0000ff00) << 2) +
(b3g3r3b2 & 0x000000ff) +
dmp[(x + 3) & (DM_WIDTH - 1)];
rgb13 += 0x10040100
- ((rgb13 & 0x1e0001e0) >> 5)
- ((rgb13 & 0x00070000) >> 6);
((guint32 *)obptr)[1] =
((rgb02 & 0x0f800000) << 4) |
((rgb02 & 0x0003f000) << 9) |
((rgb02 & 0x000000f8) << 13) |
((rgb13 & 0x0f800000) >> 12) |
((rgb13 & 0x0003f000) >> 7) |
((rgb13 & 0x000000f8) >> 3);
#endif
bp2 += 12;
obptr += 8;
}
@ -1916,6 +1992,10 @@ gdk_rgb_convert_565_d (GdkRgbInfo *image_info, GdkImage *image,
}
#endif
#undef GDK_RGB_FAST_BE565_RGB
#undef GDK_RGB_FAST_BE565_GRAY
#undef GDK_RGB_FAST_BE565_DITHER
static void
gdk_rgb_convert_555 (GdkRgbInfo *image_info, GdkImage *image,
gint x0, gint y0, gint width, gint height,