[Scummvm-git-logs] scummvm master -> f62eb48fb324a99300de0ff5a207b2161268a70f

mikrosk noreply at scummvm.org
Sun Oct 4 04:42:15 UTC 2026


This automated email contains information about 1 new commit which have been
pushed to the 'scummvm' repo located at https://api.github.com/repos/scummvm/scummvm .

Summary:
f62eb48fb3 BACKENDS: ATARI: Optimize the blit copy routines


Commit: f62eb48fb324a99300de0ff5a207b2161268a70f
    https://github.com/scummvm/scummvm/commit/f62eb48fb324a99300de0ff5a207b2161268a70f
Author: Miro Kropacek (miro.kropacek at gmail.com)
Date: 2026-10-04T14:42:04+10:00

Commit Message:
BACKENDS: ATARI: Optimize the blit copy routines

Changed paths:
    graphics/blit/blit-atari.cpp


diff --git a/graphics/blit/blit-atari.cpp b/graphics/blit/blit-atari.cpp
index 5f301051ff8..1d1bcbdc51d 100644
--- a/graphics/blit/blit-atari.cpp
+++ b/graphics/blit/blit-atari.cpp
@@ -35,12 +35,72 @@ static inline bool hasMove16() {
 	return hasMove16;
 }
 
-template<typename T>
-constexpr bool isAligned(T val) {
-	return (reinterpret_cast<uintptr>(val) & (MALLOC_ALIGNMENT - 1)) == 0;
+// move16 ignores the lowest 4 bits so only the offset within 16 bytes has to match
+static inline bool haveSameAlignment(uintptr a, uintptr b) {
+	return ((a ^ b) & (MALLOC_ALIGNMENT - 1)) == 0;
 }
 #endif
 
+// writes are made long-aligned by copying a byte/word head first, reads from src may be misaligned
+static inline void copyLong(byte *dst, const byte *src, uint w, uint h, uint dstSkip, uint srcSkip) {
+	int loopCount = h - 1;
+	__asm__ volatile(
+	"0:\n"
+	"	move.l	%3,%%d0\n"
+	// copy the head up to the next 4-byte boundary of dst
+	"	move.l	%1,%%d1\n"
+	"	neg.l	%%d1\n"
+	"	and.l	#3,%%d1\n"
+	"	sub.l	%%d1,%%d0\n"
+	"	lsr.l	#1,%%d1\n"
+	"	bcc.b	1f\n"
+
+	"	move.b	(%0)+,(%1)+\n"
+	"1:\n"
+	"	lsr.l	#1,%%d1\n"
+	"	bcc.b	2f\n"
+
+	"	move.w	(%0)+,(%1)+\n"
+	"2:\n"
+	// 256-byte blocks
+	"	move.l	%%d0,%%d1\n"
+	"	lsr.l	#8,%%d1\n"
+	"	bra.w	4f\n"
+	"3:\n"
+	"	.rept	64\n"
+	"	move.l	(%0)+,(%1)+\n"
+	"	.endr\n"
+	"4:\n"
+	"	dbra	%%d1,3b\n"
+
+	"	move.w	%%d0,%%d1\n"
+	"	and.w	#0xff,%%d1\n"
+	"	lsr.w	#2,%%d1\n"
+	"	bra.b	6f\n"
+	"5:\n"
+	"	move.l	(%0)+,(%1)+\n"
+	"6:\n"
+	"	dbra	%%d1,5b\n"
+	// copy the tail after the last long
+	"	btst	#1,%%d0\n"
+	"	beq.b	7f\n"
+
+	"	move.w	(%0)+,(%1)+\n"
+	"7:\n"
+	"	btst	#0,%%d0\n"
+	"	beq.b	8f\n"
+
+	"	move.b	(%0)+,(%1)+\n"
+	"8:\n"
+	"	add.l	%4,%1\n"
+	"	add.l	%5,%0\n"
+	"	dbra	%2,0b\n"
+		: "+a"(src), "+a"(dst), "+d"(loopCount) // outputs
+		: "g"(w), "r"(dstSkip), "r"(srcSkip) // inputs
+		: "d0", "d1", "cc" AND_MEMORY
+	);
+}
+
 namespace Graphics {
 
 // Function to blit a rect with a transparent color key
@@ -132,9 +192,39 @@ void copyBlit(byte *dst, const byte *src,
 #endif
 	if (dstPitch == srcPitch && dstPitch == (w * bytesPerPixel)) {
 #ifdef USE_MOVE16
-		if (hasMove16() && isAligned(src) && isAligned(dst)) {
+		if (hasMove16() && dstPitch * h >= 16 && haveSameAlignment((uintptr)src, (uintptr)dst)) {
 			__asm__ volatile(
 			"	move.l	%2,%%d0\n"
+			// copy the head up to the next 16-byte boundary
+			"	move.l	%1,%%d1\n"
+			"	neg.l	%%d1\n"
+			"	moveq	#0x0f,%%d2\n"
+			"	and.l	%%d2,%%d1\n"
+			"	beq.b	8f\n"
+
+			"	sub.l	%%d1,%%d0\n"
+			"	lsr.l	#1,%%d1\n"
+			"	bcc.b	5f\n"
+
+			"	move.b	(%0)+,(%1)+\n"
+			"5:\n"
+			"	lsr.l	#1,%%d1\n"
+			"	bcc.b	6f\n"
+
+			"	move.w	(%0)+,(%1)+\n"
+			"6:\n"
+			"	lsr.l	#1,%%d1\n"
+			"	bcc.b	7f\n"
+
+			"	move.l	(%0)+,(%1)+\n"
+			"7:\n"
+			"	lsr.l	#1,%%d1\n"
+			"	bcc.b	8f\n"
+
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"8:\n"
+			"	and.l	%%d0,%%d2\n"
 			"	lsr.l	#4,%%d0\n"
 			"	beq.b	3f\n"
 
@@ -162,66 +252,171 @@ void copyBlit(byte *dst, const byte *src,
 			"	move16	(%0)+,(%1)+\n"
 			"2:\n"
 			"	dbra	%%d0,1b\n"
-			// handle also the case when 'dstPitch' is not
-			// divisible by 16 but 'src' and 'dst' are
+			// handle the tail after the last 16-byte boundary
 			"3:\n"
-			"	moveq	#0x0f,%%d0\n"
-			"	and.l	%2,%%d0\n"
-			"	neg.l	%%d0\n"
-			"	jmp		(4f,%%pc,%%d0.l*2)\n"
-			// only 15x move.b as 16 would be handled above
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
+			"	lsl.l	#4,%%d2\n"
+			"	jmp		(9f,%%pc,%%d2.l)\n"
+			// 16-byte entries, one for each tail length
+			"9:\n"
+			// 0 bytes
+			"	bra.w	4f\n"
+
+			// 1 byte
+			"	.org	9b+16\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 2 bytes
+			"	.org	9b+32\n"
+			"	move.w	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 3 bytes
+			"	.org	9b+48\n"
+			"	move.w	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 4 bytes
+			"	.org	9b+64\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 5 bytes
+			"	.org	9b+80\n"
+			"	move.l	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 6 bytes
+			"	.org	9b+96\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 7 bytes
+			"	.org	9b+112\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 8 bytes
+			"	.org	9b+128\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 9 bytes
+			"	.org	9b+144\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 10 bytes
+			"	.org	9b+160\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 11 bytes
+			"	.org	9b+176\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 12 bytes
+			"	.org	9b+192\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 13 bytes
+			"	.org	9b+208\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 14 bytes
+			"	.org	9b+224\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
+			"	bra.w	4f\n"
+
+			// 15 bytes
+			"	.org	9b+240\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
 			"4:\n"
 				: "+a"(src), "+a"(dst) // outputs
 				: "g"(dstPitch * h) // inputs
-				: "d0", "d1", "cc" AND_MEMORY
+				: "d0", "d1", "d2", "cc" AND_MEMORY
 			);
 			// WARNING: src and dst are modified by the asm code
-		} else {
-#else
-		{
+		} else
 #endif
+		if (dstPitch * h >= 4) {
+			copyLong(dst, src, dstPitch * h, 1, 0, 0);
+		} else {
 			memcpy(dst, src, dstPitch * h);
 		}
 	} else {
 #ifdef USE_MOVE16
-		if (hasMove16() && isAligned(src) && isAligned(dst) && isAligned(srcPitch) && isAligned(dstPitch)) {
+		if (hasMove16() && w >= 16 && haveSameAlignment((uintptr)src, (uintptr)dst) && haveSameAlignment(srcPitch, dstPitch)) {
 			int loopCount = h - 1;
 			__asm__ volatile(
+			"0:\n"
 			"	move.l	%3,%%d0\n"
-
-			"	moveq	#0x0f,%%d1\n"
-			"	and.l	%%d0,%%d1\n"
+			// copy the head up to the next 16-byte boundary
+			"	move.l	%1,%%d1\n"
 			"	neg.l	%%d1\n"
-			"	lea		(4f,%%pc,%%d1.l*2),%%a0\n"
-			"	move.l	%%a0,%%a1\n"
+			"	moveq	#0x0f,%%d2\n"
+			"	and.l	%%d2,%%d1\n"
+			"	beq.b	8f\n"
+
+			"	sub.l	%%d1,%%d0\n"
+			"	lsr.l	#1,%%d1\n"
+			"	bcc.b	5f\n"
+
+			"	move.b	(%0)+,(%1)+\n"
+			"5:\n"
+			"	lsr.l	#1,%%d1\n"
+			"	bcc.b	6f\n"
 
+			"	move.w	(%0)+,(%1)+\n"
+			"6:\n"
+			"	lsr.l	#1,%%d1\n"
+			"	bcc.b	7f\n"
+
+			"	move.l	(%0)+,(%1)+\n"
+			"7:\n"
+			"	lsr.l	#1,%%d1\n"
+			"	bcc.b	8f\n"
+
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"8:\n"
+			"	and.l	%%d0,%%d2\n"
 			"	lsr.l	#4,%%d0\n"
 			"	beq.b	3f\n"
 
 			"	moveq	#0x0f,%%d1\n"
 			"	and.l	%%d0,%%d1\n"
 			"	neg.l	%%d1\n"
-			"	lea		(2f,%%pc,%%d1.l*4),%%a0\n"
 			"	lsr.l	#4,%%d0\n"
-			"	move.l	%%d0,%%d1\n"
-			"0:\n"
-			"	move.l	%%d1,%%d0\n"
-			"	jmp		(%%a0)\n"
+			"	jmp		(2f,%%pc,%%d1.l*4)\n"
 			"1:\n"
 			"	move16	(%0)+,(%1)+\n"
 			"	move16	(%0)+,(%1)+\n"
@@ -241,39 +436,173 @@ void copyBlit(byte *dst, const byte *src,
 			"	move16	(%0)+,(%1)+\n"
 			"2:\n"
 			"	dbra	%%d0,1b\n"
-			// handle (w * bytesPerPixel) % 16
+			// handle the tail after the last 16-byte boundary
 			"3:\n"
-			"	jmp		(%%a1)\n"
-			// only 15x move.b as 16 would be handled above
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
-			"	move.b	(%0)+,(%1)+\n"
+			"	lsl.l	#5,%%d2\n"
+			"	jmp		(9f,%%pc,%%d2.l)\n"
+			// 32-byte entries, one for each tail length
+			"9:\n"
+			// 0 bytes
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 1 byte
+			"	.org	9b+32\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 2 bytes
+			"	.org	9b+64\n"
+			"	move.w	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 3 bytes
+			"	.org	9b+96\n"
+			"	move.w	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 4 bytes
+			"	.org	9b+128\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 5 bytes
+			"	.org	9b+160\n"
+			"	move.l	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 6 bytes
+			"	.org	9b+192\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 7 bytes
+			"	.org	9b+224\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 8 bytes
+			"	.org	9b+256\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 9 bytes
+			"	.org	9b+288\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 10 bytes
+			"	.org	9b+320\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 11 bytes
+			"	.org	9b+352\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 12 bytes
+			"	.org	9b+384\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 13 bytes
+			"	.org	9b+416\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 14 bytes
+			"	.org	9b+448\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
+			"	add.l	%4,%1\n"
+			"	add.l	%5,%0\n"
+			"	dbra	%2,0b\n"
+			"	bra.w	4f\n"
+
+			// 15 bytes
+			"	.org	9b+480\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.l	(%0)+,(%1)+\n"
+			"	move.w	(%0)+,(%1)+\n"
 			"	move.b	(%0)+,(%1)+\n"
-			"4:\n"
 			"	add.l	%4,%1\n"
 			"	add.l	%5,%0\n"
 			"	dbra	%2,0b\n"
+			"4:\n"
 				: "+a"(src), "+a"(dst), "+d"(loopCount) // outputs
 				: "g"(w * bytesPerPixel),
-				  "g"(dstPitch - w * bytesPerPixel), "g"(srcPitch - w * bytesPerPixel) // inputs
-				: "d0", "d1", "a0", "a1", "cc" AND_MEMORY
+				  "r"(dstPitch - w * bytesPerPixel), "r"(srcPitch - w * bytesPerPixel) // inputs
+				: "d0", "d1", "d2", "cc" AND_MEMORY
 			);
 			// WARNING: src and dst are modified by the asm code
-		} else {
-#else
-		{
+		} else
 #endif
+		if (w * bytesPerPixel >= 4) {
+			copyLong(dst, src, w * bytesPerPixel, h, dstPitch - w * bytesPerPixel, srcPitch - w * bytesPerPixel);
+		} else {
 			for (uint i = 0; i < h; ++i) {
 				memcpy(dst, src, w * bytesPerPixel);
 				dst += dstPitch;




More information about the Scummvm-git-logs mailing list