[Scummvm-git-logs] scummvm master -> f62eb48fb324a99300de0ff5a207b2161268a70f
mikrosk
noreply at scummvm.org
Sun Oct 4 04:42:15 UTC 2026
This automated email contains information about 1 new commit which have been
pushed to the 'scummvm' repo located at https://api.github.com/repos/scummvm/scummvm .
Summary:
f62eb48fb3 BACKENDS: ATARI: Optimize the blit copy routines
Commit: f62eb48fb324a99300de0ff5a207b2161268a70f
https://github.com/scummvm/scummvm/commit/f62eb48fb324a99300de0ff5a207b2161268a70f
Author: Miro Kropacek (miro.kropacek at gmail.com)
Date: 2026-10-04T14:42:04+10:00
Commit Message:
BACKENDS: ATARI: Optimize the blit copy routines
Changed paths:
graphics/blit/blit-atari.cpp
diff --git a/graphics/blit/blit-atari.cpp b/graphics/blit/blit-atari.cpp
index 5f301051ff8..1d1bcbdc51d 100644
--- a/graphics/blit/blit-atari.cpp
+++ b/graphics/blit/blit-atari.cpp
@@ -35,12 +35,72 @@ static inline bool hasMove16() {
return hasMove16;
}
-template<typename T>
-constexpr bool isAligned(T val) {
- return (reinterpret_cast<uintptr>(val) & (MALLOC_ALIGNMENT - 1)) == 0;
+// move16 ignores the lowest 4 bits so only the offset within 16 bytes has to match
+static inline bool haveSameAlignment(uintptr a, uintptr b) {
+ return ((a ^ b) & (MALLOC_ALIGNMENT - 1)) == 0;
}
#endif
+// writes are made long-aligned by copying a byte/word head first, reads from src may be misaligned
+static inline void copyLong(byte *dst, const byte *src, uint w, uint h, uint dstSkip, uint srcSkip) {
+ int loopCount = h - 1;
+ __asm__ volatile(
+ "0:\n"
+ " move.l %3,%%d0\n"
+ // copy the head up to the next 4-byte boundary of dst
+ " move.l %1,%%d1\n"
+ " neg.l %%d1\n"
+ " and.l #3,%%d1\n"
+ " sub.l %%d1,%%d0\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 1f\n"
+
+ " move.b (%0)+,(%1)+\n"
+ "1:\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 2f\n"
+
+ " move.w (%0)+,(%1)+\n"
+ "2:\n"
+ // 256-byte blocks
+ " move.l %%d0,%%d1\n"
+ " lsr.l #8,%%d1\n"
+ " bra.w 4f\n"
+ "3:\n"
+ " .rept 64\n"
+ " move.l (%0)+,(%1)+\n"
+ " .endr\n"
+ "4:\n"
+ " dbra %%d1,3b\n"
+
+ " move.w %%d0,%%d1\n"
+ " and.w #0xff,%%d1\n"
+ " lsr.w #2,%%d1\n"
+ " bra.b 6f\n"
+ "5:\n"
+ " move.l (%0)+,(%1)+\n"
+ "6:\n"
+ " dbra %%d1,5b\n"
+ // copy the tail after the last long
+ " btst #1,%%d0\n"
+ " beq.b 7f\n"
+
+ " move.w (%0)+,(%1)+\n"
+ "7:\n"
+ " btst #0,%%d0\n"
+ " beq.b 8f\n"
+
+ " move.b (%0)+,(%1)+\n"
+ "8:\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ : "+a"(src), "+a"(dst), "+d"(loopCount) // outputs
+ : "g"(w), "r"(dstSkip), "r"(srcSkip) // inputs
+ : "d0", "d1", "cc" AND_MEMORY
+ );
+}
+
namespace Graphics {
// Function to blit a rect with a transparent color key
@@ -132,9 +192,39 @@ void copyBlit(byte *dst, const byte *src,
#endif
if (dstPitch == srcPitch && dstPitch == (w * bytesPerPixel)) {
#ifdef USE_MOVE16
- if (hasMove16() && isAligned(src) && isAligned(dst)) {
+ if (hasMove16() && dstPitch * h >= 16 && haveSameAlignment((uintptr)src, (uintptr)dst)) {
__asm__ volatile(
" move.l %2,%%d0\n"
+ // copy the head up to the next 16-byte boundary
+ " move.l %1,%%d1\n"
+ " neg.l %%d1\n"
+ " moveq #0x0f,%%d2\n"
+ " and.l %%d2,%%d1\n"
+ " beq.b 8f\n"
+
+ " sub.l %%d1,%%d0\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 5f\n"
+
+ " move.b (%0)+,(%1)+\n"
+ "5:\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 6f\n"
+
+ " move.w (%0)+,(%1)+\n"
+ "6:\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 7f\n"
+
+ " move.l (%0)+,(%1)+\n"
+ "7:\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 8f\n"
+
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ "8:\n"
+ " and.l %%d0,%%d2\n"
" lsr.l #4,%%d0\n"
" beq.b 3f\n"
@@ -162,66 +252,171 @@ void copyBlit(byte *dst, const byte *src,
" move16 (%0)+,(%1)+\n"
"2:\n"
" dbra %%d0,1b\n"
- // handle also the case when 'dstPitch' is not
- // divisible by 16 but 'src' and 'dst' are
+ // handle the tail after the last 16-byte boundary
"3:\n"
- " moveq #0x0f,%%d0\n"
- " and.l %2,%%d0\n"
- " neg.l %%d0\n"
- " jmp (4f,%%pc,%%d0.l*2)\n"
- // only 15x move.b as 16 would be handled above
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
+ " lsl.l #4,%%d2\n"
+ " jmp (9f,%%pc,%%d2.l)\n"
+ // 16-byte entries, one for each tail length
+ "9:\n"
+ // 0 bytes
+ " bra.w 4f\n"
+
+ // 1 byte
+ " .org 9b+16\n"
" move.b (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 2 bytes
+ " .org 9b+32\n"
+ " move.w (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 3 bytes
+ " .org 9b+48\n"
+ " move.w (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 4 bytes
+ " .org 9b+64\n"
+ " move.l (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 5 bytes
+ " .org 9b+80\n"
+ " move.l (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 6 bytes
+ " .org 9b+96\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 7 bytes
+ " .org 9b+112\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 8 bytes
+ " .org 9b+128\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 9 bytes
+ " .org 9b+144\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 10 bytes
+ " .org 9b+160\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 11 bytes
+ " .org 9b+176\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 12 bytes
+ " .org 9b+192\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 13 bytes
+ " .org 9b+208\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 14 bytes
+ " .org 9b+224\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
+ " bra.w 4f\n"
+
+ // 15 bytes
+ " .org 9b+240\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
"4:\n"
: "+a"(src), "+a"(dst) // outputs
: "g"(dstPitch * h) // inputs
- : "d0", "d1", "cc" AND_MEMORY
+ : "d0", "d1", "d2", "cc" AND_MEMORY
);
// WARNING: src and dst are modified by the asm code
- } else {
-#else
- {
+ } else
#endif
+ if (dstPitch * h >= 4) {
+ copyLong(dst, src, dstPitch * h, 1, 0, 0);
+ } else {
memcpy(dst, src, dstPitch * h);
}
} else {
#ifdef USE_MOVE16
- if (hasMove16() && isAligned(src) && isAligned(dst) && isAligned(srcPitch) && isAligned(dstPitch)) {
+ if (hasMove16() && w >= 16 && haveSameAlignment((uintptr)src, (uintptr)dst) && haveSameAlignment(srcPitch, dstPitch)) {
int loopCount = h - 1;
__asm__ volatile(
+ "0:\n"
" move.l %3,%%d0\n"
-
- " moveq #0x0f,%%d1\n"
- " and.l %%d0,%%d1\n"
+ // copy the head up to the next 16-byte boundary
+ " move.l %1,%%d1\n"
" neg.l %%d1\n"
- " lea (4f,%%pc,%%d1.l*2),%%a0\n"
- " move.l %%a0,%%a1\n"
+ " moveq #0x0f,%%d2\n"
+ " and.l %%d2,%%d1\n"
+ " beq.b 8f\n"
+
+ " sub.l %%d1,%%d0\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 5f\n"
+
+ " move.b (%0)+,(%1)+\n"
+ "5:\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 6f\n"
+ " move.w (%0)+,(%1)+\n"
+ "6:\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 7f\n"
+
+ " move.l (%0)+,(%1)+\n"
+ "7:\n"
+ " lsr.l #1,%%d1\n"
+ " bcc.b 8f\n"
+
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ "8:\n"
+ " and.l %%d0,%%d2\n"
" lsr.l #4,%%d0\n"
" beq.b 3f\n"
" moveq #0x0f,%%d1\n"
" and.l %%d0,%%d1\n"
" neg.l %%d1\n"
- " lea (2f,%%pc,%%d1.l*4),%%a0\n"
" lsr.l #4,%%d0\n"
- " move.l %%d0,%%d1\n"
- "0:\n"
- " move.l %%d1,%%d0\n"
- " jmp (%%a0)\n"
+ " jmp (2f,%%pc,%%d1.l*4)\n"
"1:\n"
" move16 (%0)+,(%1)+\n"
" move16 (%0)+,(%1)+\n"
@@ -241,39 +436,173 @@ void copyBlit(byte *dst, const byte *src,
" move16 (%0)+,(%1)+\n"
"2:\n"
" dbra %%d0,1b\n"
- // handle (w * bytesPerPixel) % 16
+ // handle the tail after the last 16-byte boundary
"3:\n"
- " jmp (%%a1)\n"
- // only 15x move.b as 16 would be handled above
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
- " move.b (%0)+,(%1)+\n"
+ " lsl.l #5,%%d2\n"
+ " jmp (9f,%%pc,%%d2.l)\n"
+ // 32-byte entries, one for each tail length
+ "9:\n"
+ // 0 bytes
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 1 byte
+ " .org 9b+32\n"
" move.b (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 2 bytes
+ " .org 9b+64\n"
+ " move.w (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 3 bytes
+ " .org 9b+96\n"
+ " move.w (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 4 bytes
+ " .org 9b+128\n"
+ " move.l (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 5 bytes
+ " .org 9b+160\n"
+ " move.l (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 6 bytes
+ " .org 9b+192\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 7 bytes
+ " .org 9b+224\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 8 bytes
+ " .org 9b+256\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 9 bytes
+ " .org 9b+288\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 10 bytes
+ " .org 9b+320\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 11 bytes
+ " .org 9b+352\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 12 bytes
+ " .org 9b+384\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 13 bytes
+ " .org 9b+416\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 14 bytes
+ " .org 9b+448\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
+ " add.l %4,%1\n"
+ " add.l %5,%0\n"
+ " dbra %2,0b\n"
+ " bra.w 4f\n"
+
+ // 15 bytes
+ " .org 9b+480\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.l (%0)+,(%1)+\n"
+ " move.w (%0)+,(%1)+\n"
" move.b (%0)+,(%1)+\n"
- "4:\n"
" add.l %4,%1\n"
" add.l %5,%0\n"
" dbra %2,0b\n"
+ "4:\n"
: "+a"(src), "+a"(dst), "+d"(loopCount) // outputs
: "g"(w * bytesPerPixel),
- "g"(dstPitch - w * bytesPerPixel), "g"(srcPitch - w * bytesPerPixel) // inputs
- : "d0", "d1", "a0", "a1", "cc" AND_MEMORY
+ "r"(dstPitch - w * bytesPerPixel), "r"(srcPitch - w * bytesPerPixel) // inputs
+ : "d0", "d1", "d2", "cc" AND_MEMORY
);
// WARNING: src and dst are modified by the asm code
- } else {
-#else
- {
+ } else
#endif
+ if (w * bytesPerPixel >= 4) {
+ copyLong(dst, src, w * bytesPerPixel, h, dstPitch - w * bytesPerPixel, srcPitch - w * bytesPerPixel);
+ } else {
for (uint i = 0; i < h; ++i) {
memcpy(dst, src, w * bytesPerPixel);
dst += dstPitch;
More information about the Scummvm-git-logs
mailing list