00db10
# commit 3be87c77d24c4456ccca4034363b6d1814cd0c84
00db10
# Author: Alan Modra <amodra@gmail.com>
00db10
# Date:   Sat Aug 17 18:47:59 2013 +0930
00db10
# 
00db10
#     PowerPC LE memset
00db10
#     http://sourceware.org/ml/libc-alpha/2013-08/msg00104.html
00db10
#     
00db10
#     One of the things I noticed when looking at power7 timing is that rlwimi
00db10
#     is cracked and the two resulting insns have a register dependency.
00db10
#     That makes it a little slower than the equivalent rldimi.
00db10
#     
00db10
#         * sysdeps/powerpc/powerpc64/memset.S: Replace rlwimi with
00db10
#             insrdi.  Formatting.
00db10
#         * sysdeps/powerpc/powerpc64/power4/memset.S: Likewise.
00db10
#         * sysdeps/powerpc/powerpc64/power6/memset.S: Likewise.
00db10
#         * sysdeps/powerpc/powerpc64/power7/memset.S: Likewise.
00db10
#         * sysdeps/powerpc/powerpc32/power4/memset.S: Likewise.
00db10
#         * sysdeps/powerpc/powerpc32/power6/memset.S: Likewise.
00db10
#         * sysdeps/powerpc/powerpc32/power7/memset.S: Likewise.
00db10
# 
00db10
diff -urN glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power4/memset.S glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power4/memset.S
00db10
--- glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power4/memset.S	2014-05-29 13:07:41.000000000 -0500
00db10
+++ glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power4/memset.S	2014-05-29 13:07:46.000000000 -0500
00db10
@@ -52,7 +52,7 @@
00db10
 
00db10
 /* Align to word boundary.  */
00db10
 	cmplwi	cr5, rLEN, 31
00db10
-	rlwimi	rCHR, rCHR, 8, 16, 23 /* Replicate byte to halfword.  */
00db10
+	insrdi	rCHR, rCHR, 8, 48     /* Replicate byte to halfword.  */
00db10
 	beq+	L(aligned)
00db10
 	mtcrf	0x01, rMEMP0
00db10
 	subfic	rALIGN, rALIGN, 4
00db10
@@ -67,7 +67,7 @@
00db10
 /* Handle the case of size < 31.  */
00db10
 L(aligned):
00db10
 	mtcrf	0x01, rLEN
00db10
-	rlwimi	rCHR, rCHR, 16, 0, 15 /* Replicate halfword to word.  */
00db10
+	insrdi	rCHR, rCHR, 16, 32    /* Replicate halfword to word.  */
00db10
 	ble	cr5, L(medium)
00db10
 /* Align to 32-byte boundary.  */
00db10
 	andi.	rALIGN, rMEMP, 0x1C
00db10
diff -urN glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power6/memset.S glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power6/memset.S
00db10
--- glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power6/memset.S	2014-05-29 13:07:41.000000000 -0500
00db10
+++ glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power6/memset.S	2014-05-29 13:07:46.000000000 -0500
00db10
@@ -50,7 +50,7 @@
00db10
 	ble-	cr1, L(small)
00db10
 /* Align to word boundary.  */
00db10
 	cmplwi	cr5, rLEN, 31
00db10
-	rlwimi	rCHR, rCHR, 8, 16, 23 /* Replicate byte to halfword.  */
00db10
+	insrdi	rCHR, rCHR, 8, 48	/* Replicate byte to halfword.  */
00db10
 	beq+	L(aligned)
00db10
 	mtcrf	0x01, rMEMP0
00db10
 	subfic	rALIGN, rALIGN, 4
00db10
@@ -66,7 +66,7 @@
00db10
 /* Handle the case of size < 31.  */
00db10
 L(aligned):
00db10
 	mtcrf	0x01, rLEN
00db10
-	rlwimi	rCHR, rCHR, 16, 0, 15 /* Replicate halfword to word.  */
00db10
+	insrdi	rCHR, rCHR, 16, 32	/* Replicate halfword to word.  */
00db10
 	ble	cr5, L(medium)
00db10
 /* Align to 32-byte boundary.  */
00db10
 	andi.	rALIGN, rMEMP, 0x1C
00db10
diff -urN glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power7/memset.S glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power7/memset.S
00db10
--- glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power7/memset.S	2014-05-29 13:07:41.000000000 -0500
00db10
+++ glibc-2.17-c758a686/sysdeps/powerpc/powerpc32/power7/memset.S	2014-05-29 13:07:46.000000000 -0500
00db10
@@ -37,8 +37,8 @@
00db10
 	cfi_offset(31,-8)
00db10
 
00db10
 	/* Replicate byte to word.  */
00db10
-	rlwimi	4,4,8,16,23
00db10
-	rlwimi	4,4,16,0,15
00db10
+	insrdi	4,4,8,48
00db10
+	insrdi	4,4,16,32
00db10
 
00db10
 	ble	cr6,L(small)	/* If length <= 8, use short copy code.  */
00db10
 
00db10
diff -urN glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/memset.S glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/memset.S
00db10
--- glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/memset.S	2014-05-29 13:07:41.000000000 -0500
00db10
+++ glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/memset.S	2014-05-29 13:07:46.000000000 -0500
00db10
@@ -73,14 +73,14 @@
00db10
 
00db10
 /* Align to doubleword boundary.  */
00db10
 	cmpldi	cr5, rLEN, 31
00db10
-	rlwimi	rCHR, rCHR, 8, 16, 23 /* Replicate byte to halfword.  */
00db10
+	insrdi	rCHR, rCHR, 8, 48	/* Replicate byte to halfword.  */
00db10
 	beq+	L(aligned2)
00db10
 	mtcrf	0x01, rMEMP0
00db10
 	subfic	rALIGN, rALIGN, 8
00db10
 	cror	28,30,31		/* Detect odd word aligned.  */
00db10
 	add	rMEMP, rMEMP, rALIGN
00db10
 	sub	rLEN, rLEN, rALIGN
00db10
-	rlwimi	rCHR, rCHR, 16, 0, 15 /* Replicate halfword to word.  */
00db10
+	insrdi	rCHR, rCHR, 16, 32	/* Replicate halfword to word.  */
00db10
 	bt	29, L(g4)
00db10
 /* Process the even word of doubleword.  */
00db10
 	bf+	31, L(g2)
00db10
@@ -102,14 +102,14 @@
00db10
 
00db10
 /* Handle the case of size < 31.  */
00db10
 L(aligned2):
00db10
-	rlwimi	rCHR, rCHR, 16, 0, 15 /* Replicate halfword to word.  */
00db10
+	insrdi	rCHR, rCHR, 16, 32	/* Replicate halfword to word.  */
00db10
 L(aligned):
00db10
 	mtcrf	0x01, rLEN
00db10
 	ble	cr5, L(medium)
00db10
 /* Align to 32-byte boundary.  */
00db10
 	andi.	rALIGN, rMEMP, 0x18
00db10
 	subfic	rALIGN, rALIGN, 0x20
00db10
-	insrdi	rCHR,rCHR,32,0 /* Replicate word to double word. */
00db10
+	insrdi	rCHR, rCHR, 32, 0	/* Replicate word to double word. */
00db10
 	beq	L(caligned)
00db10
 	mtcrf	0x01, rALIGN
00db10
 	add	rMEMP, rMEMP, rALIGN
00db10
@@ -230,7 +230,7 @@
00db10
 /* Memset of 0-31 bytes.  */
00db10
 	.align 5
00db10
 L(medium):
00db10
-	insrdi	rCHR,rCHR,32,0 /* Replicate word to double word.  */
00db10
+	insrdi	rCHR, rCHR, 32, 0	/* Replicate word to double word.  */
00db10
 	cmpldi	cr1, rLEN, 16
00db10
 L(medium_tail2):
00db10
 	add	rMEMP, rMEMP, rLEN
00db10
diff -urN glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power4/memset.S glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power4/memset.S
00db10
--- glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power4/memset.S	2014-05-29 13:07:41.000000000 -0500
00db10
+++ glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power4/memset.S	2014-05-29 13:07:46.000000000 -0500
00db10
@@ -68,14 +68,14 @@
00db10
 
00db10
 /* Align to doubleword boundary.  */
00db10
 	cmpldi	cr5, rLEN, 31
00db10
-	rlwimi	rCHR, rCHR, 8, 16, 23 /* Replicate byte to halfword.  */
00db10
+	insrdi	rCHR, rCHR, 8, 48	/* Replicate byte to halfword.  */
00db10
 	beq+	L(aligned2)
00db10
 	mtcrf	0x01, rMEMP0
00db10
 	subfic	rALIGN, rALIGN, 8
00db10
 	cror	28,30,31		/* Detect odd word aligned.  */
00db10
 	add	rMEMP, rMEMP, rALIGN
00db10
 	sub	rLEN, rLEN, rALIGN
00db10
-	rlwimi	rCHR, rCHR, 16, 0, 15 /* Replicate halfword to word.  */
00db10
+	insrdi	rCHR, rCHR, 16, 32	/* Replicate halfword to word.  */
00db10
 	bt	29, L(g4)
00db10
 /* Process the even word of doubleword.  */
00db10
 	bf+	31, L(g2)
00db10
@@ -97,14 +97,14 @@
00db10
 
00db10
 /* Handle the case of size < 31.  */
00db10
 L(aligned2):
00db10
-	rlwimi	rCHR, rCHR, 16, 0, 15 /* Replicate halfword to word.  */
00db10
+	insrdi	rCHR, rCHR, 16, 32	/* Replicate halfword to word.  */
00db10
 L(aligned):
00db10
 	mtcrf	0x01, rLEN
00db10
 	ble	cr5, L(medium)
00db10
 /* Align to 32-byte boundary.  */
00db10
 	andi.	rALIGN, rMEMP, 0x18
00db10
 	subfic	rALIGN, rALIGN, 0x20
00db10
-	insrdi	rCHR,rCHR,32,0 /* Replicate word to double word. */
00db10
+	insrdi	rCHR, rCHR, 32, 0	/* Replicate word to double word. */
00db10
 	beq	L(caligned)
00db10
 	mtcrf	0x01, rALIGN
00db10
 	add	rMEMP, rMEMP, rALIGN
00db10
@@ -164,24 +164,24 @@
00db10
 L(getCacheAligned):
00db10
 	cmpldi	cr1,rLEN,32
00db10
 	andi.	rTMP,rMEMP,127
00db10
-	blt		cr1,L(handletail32)
00db10
-	beq		L(cacheAligned)
00db10
+	blt	cr1,L(handletail32)
00db10
+	beq	L(cacheAligned)
00db10
 	addi	rMEMP,rMEMP,32
00db10
 	addi	rLEN,rLEN,-32
00db10
-	std		rCHR,-32(rMEMP)
00db10
-	std		rCHR,-24(rMEMP)
00db10
-	std		rCHR,-16(rMEMP)
00db10
-	std		rCHR,-8(rMEMP)
00db10
-	b		L(getCacheAligned)
00db10
+	std	rCHR,-32(rMEMP)
00db10
+	std	rCHR,-24(rMEMP)
00db10
+	std	rCHR,-16(rMEMP)
00db10
+	std	rCHR,-8(rMEMP)
00db10
+	b	L(getCacheAligned)
00db10
 
00db10
 /* Now we are aligned to the cache line and can use dcbz.  */
00db10
 L(cacheAligned):
00db10
 	cmpld	cr1,rLEN,rCLS
00db10
-	blt		cr1,L(handletail32)
00db10
+	blt	cr1,L(handletail32)
00db10
 	dcbz	0,rMEMP
00db10
 	subf	rLEN,rCLS,rLEN
00db10
-	add		rMEMP,rMEMP,rCLS
00db10
-	b		L(cacheAligned)
00db10
+	add	rMEMP,rMEMP,rCLS
00db10
+	b	L(cacheAligned)
00db10
 
00db10
 /* We are here because the cache line size was set and was not 32-bytes
00db10
    and the remainder (rLEN) is less than the actual cache line size.
00db10
@@ -218,7 +218,7 @@
00db10
 /* Memset of 0-31 bytes.  */
00db10
 	.align 5
00db10
 L(medium):
00db10
-	insrdi	rCHR,rCHR,32,0 /* Replicate word to double word.  */
00db10
+	insrdi	rCHR, rCHR, 32, 0	/* Replicate word to double word.  */
00db10
 	cmpldi	cr1, rLEN, 16
00db10
 L(medium_tail2):
00db10
 	add	rMEMP, rMEMP, rLEN
00db10
diff -urN glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power6/memset.S glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power6/memset.S
00db10
--- glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power6/memset.S	2014-05-29 13:07:41.000000000 -0500
00db10
+++ glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power6/memset.S	2014-05-29 13:07:46.000000000 -0500
00db10
@@ -65,14 +65,14 @@
00db10
 
00db10
 /* Align to doubleword boundary.  */
00db10
 	cmpldi	cr5, rLEN, 31
00db10
-	rlwimi	rCHR, rCHR, 8, 16, 23 /* Replicate byte to halfword.  */
00db10
+	insrdi	rCHR, rCHR, 8, 48	/* Replicate byte to halfword.  */
00db10
 	beq+	L(aligned2)
00db10
 	mtcrf	0x01, rMEMP0
00db10
 	subfic	rALIGN, rALIGN, 8
00db10
 	cror	28,30,31		/* Detect odd word aligned.  */
00db10
 	add	rMEMP, rMEMP, rALIGN
00db10
 	sub	rLEN, rLEN, rALIGN
00db10
-	rlwimi	rCHR, rCHR, 16, 0, 15 /* Replicate halfword to word.  */
00db10
+	insrdi	rCHR, rCHR, 16, 32	/* Replicate halfword to word.  */
00db10
 	bt	29, L(g4)
00db10
 /* Process the even word of doubleword.  */
00db10
 	bf+	31, L(g2)
00db10
@@ -94,14 +94,14 @@
00db10
 
00db10
 /* Handle the case of size < 31.  */
00db10
 L(aligned2):
00db10
-	rlwimi	rCHR, rCHR, 16, 0, 15 /* Replicate halfword to word.  */
00db10
+	insrdi	rCHR, rCHR, 16, 32	/* Replicate halfword to word.  */
00db10
 L(aligned):
00db10
 	mtcrf	0x01, rLEN
00db10
 	ble	cr5, L(medium)
00db10
 /* Align to 32-byte boundary.  */
00db10
 	andi.	rALIGN, rMEMP, 0x18
00db10
 	subfic	rALIGN, rALIGN, 0x20
00db10
-	insrdi	rCHR,rCHR,32,0 /* Replicate word to double word. */
00db10
+	insrdi	rCHR, rCHR, 32, 0	/* Replicate word to double word. */
00db10
 	beq	L(caligned)
00db10
 	mtcrf	0x01, rALIGN
00db10
 	add	rMEMP, rMEMP, rALIGN
00db10
@@ -362,7 +362,7 @@
00db10
 /* Memset of 0-31 bytes.  */
00db10
 	.align 5
00db10
 L(medium):
00db10
-	insrdi	rCHR,rCHR,32,0 /* Replicate word to double word.  */
00db10
+	insrdi	rCHR, rCHR, 32, 0	/* Replicate word to double word.  */
00db10
 	cmpldi	cr1, rLEN, 16
00db10
 L(medium_tail2):
00db10
 	add	rMEMP, rMEMP, rLEN
00db10
diff -urN glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power7/memset.S glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power7/memset.S
00db10
--- glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power7/memset.S	2014-05-29 13:07:41.000000000 -0500
00db10
+++ glibc-2.17-c758a686/sysdeps/powerpc/powerpc64/power7/memset.S	2014-05-29 13:07:46.000000000 -0500
00db10
@@ -34,8 +34,8 @@
00db10
 	mr	10,3
00db10
 
00db10
 	/* Replicate byte to word.  */
00db10
-	rlwimi	4,4,8,16,23
00db10
-	rlwimi	4,4,16,0,15
00db10
+	insrdi	4,4,8,48
00db10
+	insrdi	4,4,16,32
00db10
 	ble	cr6,L(small)	/* If length <= 8, use short copy code.  */
00db10
 
00db10
 	neg	0,3
00db10
@@ -323,7 +323,7 @@
00db10
 	clrldi	0,0,62
00db10
 	beq	L(medium_aligned)
00db10
 
00db10
-	/* Force 4-bytes alignment for SRC.  */
00db10
+	/* Force 4-bytes alignment for DST.  */
00db10
 	mtocrf	0x01,0
00db10
 	subf	5,0,5
00db10
 1:	/* Copy 1 byte.  */