gdk: optimize native big-endian RGB565 paths
Add native big-endian four-pixel converters for RGB565, grayscale RGB565, and dithered RGB565. Select the block converters only where cross-target code generation is favorable: RGB on PowerPC, s390x, ARMv5+, and AArch64; grayscale on PowerPC, ARMv5+, AArch64, and m68k; and dither on PowerPC. Retain the scalar fallbacks everywhere else. The little-endian machine code remains byte-for-byte unchanged. No runtime endian checks, allocations, or additional image buffers are introduced. Validated with a complete build and test suite, deterministic and randomized big-endian formula tests, exact little-endian .text comparison, and cross-target code-generation checks.
This commit is contained in:
parent
71023e0333
commit
be64a902a5
1 changed files with 113 additions and 33 deletions
146
gdk/gdkrgb.c
146
gdk/gdkrgb.c
|
|
@ -1464,19 +1464,33 @@ gdk_rgb_convert_gray8_gray (GdkRgbInfo *image_info, GdkImage *image,
|
|||
}
|
||||
}
|
||||
|
||||
#if G_BYTE_ORDER == G_LITTLE_ENDIAN
|
||||
#define HAIRY_CONVERT_565
|
||||
|
||||
#if G_BYTE_ORDER == G_BIG_ENDIAN
|
||||
#if defined(__powerpc__) || defined(__powerpc64__) || \
|
||||
defined(__ppc__) || defined(__ppc64__) || \
|
||||
defined(__PPC__) || defined(__PPC64__) || \
|
||||
defined(__POWERPC__) || defined(_ARCH_PPC)
|
||||
#define GDK_RGB_FAST_BE565_RGB
|
||||
#define GDK_RGB_FAST_BE565_GRAY
|
||||
#define GDK_RGB_FAST_BE565_DITHER
|
||||
#elif defined(__s390x__)
|
||||
#define GDK_RGB_FAST_BE565_RGB
|
||||
#elif defined(__aarch64__)
|
||||
#define GDK_RGB_FAST_BE565_RGB
|
||||
#define GDK_RGB_FAST_BE565_GRAY
|
||||
#elif defined(__arm__) && defined(__ARM_ARCH) && \
|
||||
__ARM_ARCH >= 5
|
||||
#define GDK_RGB_FAST_BE565_RGB
|
||||
#define GDK_RGB_FAST_BE565_GRAY
|
||||
#elif defined(__m68k__)
|
||||
#define GDK_RGB_FAST_BE565_GRAY
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef HAIRY_CONVERT_565
|
||||
/* Render a 24-bit RGB image in buf into the GdkImage, without dithering.
|
||||
This assumes native byte ordering - what should really be done is to
|
||||
check whether the image byte_order is consistent with the _ENDIAN
|
||||
config flag, and if not, use a different function.
|
||||
|
||||
This one is even faster than the one below - its inner loop loads 3
|
||||
words (i.e. 4 24-bit pixels), does a lot of shifting and masking,
|
||||
then writes 2 words. */
|
||||
The four-pixel inner loop uses compile-time-specialized native word
|
||||
layouts for little- and big-endian systems. */
|
||||
#if G_BYTE_ORDER == G_LITTLE_ENDIAN || defined(GDK_RGB_FAST_BE565_RGB)
|
||||
static void
|
||||
gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
|
||||
gint x0, gint y0, gint width, gint height,
|
||||
|
|
@ -1520,6 +1534,7 @@ gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
|
|||
r1b0g0r0 = ((guint32 *)bp2)[0];
|
||||
g2r2b1g1 = ((guint32 *)bp2)[1];
|
||||
b3g3r3b2 = ((guint32 *)bp2)[2];
|
||||
#if G_BYTE_ORDER == G_LITTLE_ENDIAN
|
||||
((guint32 *)obptr)[0] =
|
||||
((r1b0g0r0 & 0xf8) << 8) |
|
||||
((r1b0g0r0 & 0xfc00) >> 5) |
|
||||
|
|
@ -1534,6 +1549,22 @@ gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
|
|||
((b3g3r3b2 & 0xf800) << 16) |
|
||||
((b3g3r3b2 & 0xfc0000) << 3) |
|
||||
((b3g3r3b2 & 0xf8000000) >> 11);
|
||||
#else
|
||||
((guint32 *)obptr)[0] =
|
||||
(r1b0g0r0 & 0xf8000000) |
|
||||
((r1b0g0r0 & 0x00fc0000) << 3) |
|
||||
((r1b0g0r0 & 0x0000f800) << 5) |
|
||||
((r1b0g0r0 & 0x000000f8) << 8) |
|
||||
((g2r2b1g1 & 0xfc000000) >> 21) |
|
||||
((g2r2b1g1 & 0x00f80000) >> 19);
|
||||
((guint32 *)obptr)[1] =
|
||||
((g2r2b1g1 & 0x0000f800) << 16) |
|
||||
((g2r2b1g1 & 0x000000fc) << 19) |
|
||||
((b3g3r3b2 & 0xf8000000) >> 11) |
|
||||
((b3g3r3b2 & 0x00f80000) >> 8) |
|
||||
((b3g3r3b2 & 0x0000fc00) >> 5) |
|
||||
((b3g3r3b2 & 0x000000f8) >> 3);
|
||||
#endif
|
||||
bp2 += 12;
|
||||
obptr += 8;
|
||||
}
|
||||
|
|
@ -1553,27 +1584,6 @@ gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
|
|||
}
|
||||
}
|
||||
#else
|
||||
/* Render a 24-bit RGB image in buf into the GdkImage, without dithering.
|
||||
This assumes native byte ordering - what should really be done is to
|
||||
check whether the image byte_order is consistent with the _ENDIAN
|
||||
config flag, and if not, use a different function.
|
||||
|
||||
This routine is faster than the one included with Gtk 1.0 for a number
|
||||
of reasons:
|
||||
|
||||
1. Shifting instead of lookup tables (less memory traffic).
|
||||
|
||||
2. Much less register pressure, especially because shifts are
|
||||
in the code.
|
||||
|
||||
3. A memcpy is avoided (i.e. the transfer function).
|
||||
|
||||
4. On big-endian architectures, byte swapping is avoided.
|
||||
|
||||
That said, it wouldn't be hard to make it even faster - just make an
|
||||
inner loop that reads 3 words (i.e. 4 24-bit pixels), does a lot of
|
||||
shifting and masking, then writes 2 words.
|
||||
*/
|
||||
static void
|
||||
gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
|
||||
gint x0, gint y0, gint width, gint height,
|
||||
|
|
@ -1607,7 +1617,7 @@ gdk_rgb_convert_565 (GdkRgbInfo *image_info, GdkImage *image,
|
|||
}
|
||||
#endif
|
||||
|
||||
#ifdef HAIRY_CONVERT_565
|
||||
#if G_BYTE_ORDER == G_LITTLE_ENDIAN || defined(GDK_RGB_FAST_BE565_GRAY)
|
||||
static void
|
||||
gdk_rgb_convert_565_gray (GdkRgbInfo *image_info, GdkImage *image,
|
||||
gint x0, gint y0, gint width, gint height,
|
||||
|
|
@ -1645,6 +1655,7 @@ gdk_rgb_convert_565_gray (GdkRgbInfo *image_info, GdkImage *image,
|
|||
guint32 g3g2g1g0;
|
||||
|
||||
g3g2g1g0 = ((guint32 *)bp2)[0];
|
||||
#if G_BYTE_ORDER == G_LITTLE_ENDIAN
|
||||
((guint32 *)obptr)[0] =
|
||||
((g3g2g1g0 & 0xf8) << 8) |
|
||||
((g3g2g1g0 & 0xfc) << 3) |
|
||||
|
|
@ -1659,6 +1670,22 @@ gdk_rgb_convert_565_gray (GdkRgbInfo *image_info, GdkImage *image,
|
|||
(g3g2g1g0 & 0xf8000000) |
|
||||
((g3g2g1g0 & 0xfc000000) >> 5) |
|
||||
((g3g2g1g0 & 0xf8000000) >> 11);
|
||||
#else
|
||||
((guint32 *)obptr)[0] =
|
||||
(g3g2g1g0 & 0xf8000000) |
|
||||
((g3g2g1g0 & 0xfc000000) >> 5) |
|
||||
((g3g2g1g0 & 0xf8000000) >> 11) |
|
||||
((g3g2g1g0 & 0x00f80000) >> 8) |
|
||||
((g3g2g1g0 & 0x00fc0000) >> 13) |
|
||||
((g3g2g1g0 & 0x00f80000) >> 19);
|
||||
((guint32 *)obptr)[1] =
|
||||
((g3g2g1g0 & 0x0000f800) << 16) |
|
||||
((g3g2g1g0 & 0x0000fc00) << 11) |
|
||||
((g3g2g1g0 & 0x0000f800) << 5) |
|
||||
((g3g2g1g0 & 0x000000f8) << 8) |
|
||||
((g3g2g1g0 & 0x000000fc) << 3) |
|
||||
((g3g2g1g0 & 0x000000f8) >> 3);
|
||||
#endif
|
||||
bp2 += 4;
|
||||
obptr += 8;
|
||||
}
|
||||
|
|
@ -1745,7 +1772,7 @@ gdk_rgb_convert_565_br (GdkRgbInfo *image_info, GdkImage *image,
|
|||
|
||||
/* Thanks to Ray Lehtiniemi for a patch that resulted in a ~25% speedup
|
||||
in this mode. */
|
||||
#ifdef HAIRY_CONVERT_565
|
||||
#if G_BYTE_ORDER == G_LITTLE_ENDIAN || defined(GDK_RGB_FAST_BE565_DITHER)
|
||||
static void
|
||||
gdk_rgb_convert_565_d (GdkRgbInfo *image_info, GdkImage *image,
|
||||
gint x0, gint y0, gint width, gint height,
|
||||
|
|
@ -1800,6 +1827,7 @@ gdk_rgb_convert_565_d (GdkRgbInfo *image_info, GdkImage *image,
|
|||
r1b0g0r0 = ((guint32 *)bp2)[0];
|
||||
g2r2b1g1 = ((guint32 *)bp2)[1];
|
||||
b3g3r3b2 = ((guint32 *)bp2)[2];
|
||||
#if G_BYTE_ORDER == G_LITTLE_ENDIAN
|
||||
rgb02 =
|
||||
((r1b0g0r0 & 0xff) << 20) +
|
||||
((r1b0g0r0 & 0xff00) << 2) +
|
||||
|
|
@ -1846,6 +1874,54 @@ gdk_rgb_convert_565_d (GdkRgbInfo *image_info, GdkImage *image,
|
|||
((rgb13 & 0x0f800000) << 4) |
|
||||
((rgb13 & 0x0003f000) << 9) |
|
||||
((rgb13 & 0x000000f8) << 13);
|
||||
#else
|
||||
rgb02 =
|
||||
((r1b0g0r0 & 0xff000000) >> 4) +
|
||||
((r1b0g0r0 & 0x00ff0000) >> 6) +
|
||||
((r1b0g0r0 & 0x0000ff00) >> 8) +
|
||||
dmp[x & (DM_WIDTH - 1)];
|
||||
rgb02 += 0x10040100
|
||||
- ((rgb02 & 0x1e0001e0) >> 5)
|
||||
- ((rgb02 & 0x00070000) >> 6);
|
||||
rgb13 =
|
||||
((r1b0g0r0 & 0x000000ff) << 20) +
|
||||
((g2r2b1g1 & 0xff000000) >> 14) +
|
||||
((g2r2b1g1 & 0x00ff0000) >> 16) +
|
||||
dmp[(x + 1) & (DM_WIDTH - 1)];
|
||||
rgb13 += 0x10040100
|
||||
- ((rgb13 & 0x1e0001e0) >> 5)
|
||||
- ((rgb13 & 0x00070000) >> 6);
|
||||
((guint32 *)obptr)[0] =
|
||||
((rgb02 & 0x0f800000) << 4) |
|
||||
((rgb02 & 0x0003f000) << 9) |
|
||||
((rgb02 & 0x000000f8) << 13) |
|
||||
((rgb13 & 0x0f800000) >> 12) |
|
||||
((rgb13 & 0x0003f000) >> 7) |
|
||||
((rgb13 & 0x000000f8) >> 3);
|
||||
rgb02 =
|
||||
((g2r2b1g1 & 0x0000ff00) << 12) +
|
||||
((g2r2b1g1 & 0x000000ff) << 10) +
|
||||
((b3g3r3b2 & 0xff000000) >> 24) +
|
||||
dmp[(x + 2) & (DM_WIDTH - 1)];
|
||||
rgb02 += 0x10040100
|
||||
- ((rgb02 & 0x1e0001e0) >> 5)
|
||||
- ((rgb02 & 0x00070000) >> 6);
|
||||
rgb13 =
|
||||
((b3g3r3b2 & 0x00ff0000) << 4) +
|
||||
((b3g3r3b2 & 0x0000ff00) << 2) +
|
||||
(b3g3r3b2 & 0x000000ff) +
|
||||
dmp[(x + 3) & (DM_WIDTH - 1)];
|
||||
rgb13 += 0x10040100
|
||||
- ((rgb13 & 0x1e0001e0) >> 5)
|
||||
- ((rgb13 & 0x00070000) >> 6);
|
||||
((guint32 *)obptr)[1] =
|
||||
((rgb02 & 0x0f800000) << 4) |
|
||||
((rgb02 & 0x0003f000) << 9) |
|
||||
((rgb02 & 0x000000f8) << 13) |
|
||||
((rgb13 & 0x0f800000) >> 12) |
|
||||
((rgb13 & 0x0003f000) >> 7) |
|
||||
((rgb13 & 0x000000f8) >> 3);
|
||||
#endif
|
||||
bp2 += 12;
|
||||
obptr += 8;
|
||||
}
|
||||
|
|
@ -1916,6 +1992,10 @@ gdk_rgb_convert_565_d (GdkRgbInfo *image_info, GdkImage *image,
|
|||
}
|
||||
#endif
|
||||
|
||||
#undef GDK_RGB_FAST_BE565_RGB
|
||||
#undef GDK_RGB_FAST_BE565_GRAY
|
||||
#undef GDK_RGB_FAST_BE565_DITHER
|
||||
|
||||
static void
|
||||
gdk_rgb_convert_555 (GdkRgbInfo *image_info, GdkImage *image,
|
||||
gint x0, gint y0, gint width, gint height,
|
||||
|
|
|
|||
Loading…
Reference in a new issue