Groups | Search | Server Info | Keyboard shortcuts | Login | Register [http] [https] [nntp] [nntps]


Groups > linux.kernel > #1587299

[PATCH v4 15/24] openrisc: Add optimized memcpy routine

From Stafford Horne <shorne@gmail.com>
Newsgroups linux.kernel
Subject [PATCH v4 15/24] openrisc: Add optimized memcpy routine
Date 2017-02-24 05:50 +0100
Message-ID <tefZn-3l4-3@gated-at.bofh.it> (permalink)
References <tefPH-3gU-5@gated-at.bofh.it>
Organization linux.* mail to news gateway

Show all headers | View raw


The generic memcpy routine provided in kernel does only byte copies.
Using word copies we can lower boot time and cycles spend in memcpy
quite significantly.

Booting on my de0 nano I see boot times go from 7.2 to 5.6 seconds.
The avg cycles in memcpy during boot go from 6467 to 1887.

I tested several algorithms (see code in previous patch mails)

The implementations I tested and avg cycles:
  - Word Copies + Loop Unrolls + Non Aligned    1882
  - Word Copies + Loop Unrolls                  1887
  - Word Copies                                 2441
  - Byte Copies + Loop Unrolls                  6467
  - Byte Copies                                 7600

In the end I ended up going with Word Copies + Loop Unrolls as it
provides best tradeoff between simplicity and boot speedups.

Signed-off-by: Stafford Horne <shorne@gmail.com>
---
 arch/openrisc/TODO.openrisc        |   1 -
 arch/openrisc/include/asm/string.h |   3 +
 arch/openrisc/lib/Makefile         |   2 +-
 arch/openrisc/lib/memcpy.c         | 124 +++++++++++++++++++++++++++++++++++++
 4 files changed, 128 insertions(+), 2 deletions(-)
 create mode 100644 arch/openrisc/lib/memcpy.c

diff --git a/arch/openrisc/TODO.openrisc b/arch/openrisc/TODO.openrisc
index 0eb04c8..c43d4e1 100644
--- a/arch/openrisc/TODO.openrisc
+++ b/arch/openrisc/TODO.openrisc
@@ -10,4 +10,3 @@ that are due for investigation shortly, i.e. our TODO list:
    or1k and this change is slowly trickling through the stack.  For the time
    being, or32 is equivalent to or1k.
 
--- Implement optimized version of memcpy and memset
diff --git a/arch/openrisc/include/asm/string.h b/arch/openrisc/include/asm/string.h
index 33470d4..64939cc 100644
--- a/arch/openrisc/include/asm/string.h
+++ b/arch/openrisc/include/asm/string.h
@@ -4,4 +4,7 @@
 #define __HAVE_ARCH_MEMSET
 extern void *memset(void *s, int c, __kernel_size_t n);
 
+#define __HAVE_ARCH_MEMCPY
+extern void *memcpy(void *dest, __const void *src, __kernel_size_t n);
+
 #endif /* __ASM_OPENRISC_STRING_H */
diff --git a/arch/openrisc/lib/Makefile b/arch/openrisc/lib/Makefile
index 67c583e..17d9d37 100644
--- a/arch/openrisc/lib/Makefile
+++ b/arch/openrisc/lib/Makefile
@@ -2,4 +2,4 @@
 # Makefile for or32 specific library files..
 #
 
-obj-y  = memset.o string.o delay.o
+obj-y	:= delay.o string.o memset.o memcpy.o
diff --git a/arch/openrisc/lib/memcpy.c b/arch/openrisc/lib/memcpy.c
new file mode 100644
index 0000000..4706f01
--- /dev/null
+++ b/arch/openrisc/lib/memcpy.c
@@ -0,0 +1,124 @@
+/*
+ * arch/openrisc/lib/memcpy.c
+ *
+ * Optimized memory copy routines for openrisc.  These are mostly copied
+ * from ohter sources but slightly entended based on ideas discuassed in
+ * #openrisc.
+ *
+ * The word unroll implementation is an extension to the arm byte
+ * unrolled implementation, but using word copies (if things are
+ * properly aligned)
+ *
+ * The great arm loop unroll algorithm can be found at:
+ *  arch/arm/boot/compressed/string.c
+ */
+
+#include <linux/export.h>
+
+#include <linux/string.h>
+
+#ifdef CONFIG_OR1200
+/*
+ * Do memcpy with word copies and loop unrolling. This gives the
+ * best performance on the OR1200 and MOR1KX archirectures
+ */
+void *memcpy(void *dest, __const void *src, __kernel_size_t n)
+{
+	int i = 0;
+	unsigned char *d, *s;
+	uint32_t *dest_w = (uint32_t *)dest, *src_w = (uint32_t *)src;
+
+	/* If both source and dest are word aligned copy words */
+	if (!((unsigned int)dest_w & 3) && !((unsigned int)src_w & 3)) {
+		/* Copy 32 bytes per loop */
+		for (i = n >> 5; i > 0; i--) {
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+		}
+
+		if (n & 1 << 4) {
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+		}
+
+		if (n & 1 << 3) {
+			*dest_w++ = *src_w++;
+			*dest_w++ = *src_w++;
+		}
+
+		if (n & 1 << 2)
+			*dest_w++ = *src_w++;
+
+		d = (unsigned char *)dest_w;
+		s = (unsigned char *)src_w;
+
+	} else {
+		d = (unsigned char *)dest_w;
+		s = (unsigned char *)src_w;
+
+		for (i = n >> 3; i > 0; i--) {
+			*d++ = *s++;
+			*d++ = *s++;
+			*d++ = *s++;
+			*d++ = *s++;
+			*d++ = *s++;
+			*d++ = *s++;
+			*d++ = *s++;
+			*d++ = *s++;
+		}
+
+		if (n & 1 << 2) {
+			*d++ = *s++;
+			*d++ = *s++;
+			*d++ = *s++;
+			*d++ = *s++;
+		}
+	}
+
+	if (n & 1 << 1) {
+		*d++ = *s++;
+		*d++ = *s++;
+	}
+
+	if (n & 1)
+		*d++ = *s++;
+
+	return dest;
+}
+#else
+/*
+ * Use word copies but no loop unrolling as we cannot assume there
+ * will be benefits on the archirecture
+ */
+void *memcpy(void *dest, __const void *src, __kernel_size_t n)
+{
+	unsigned char *d = (unsigned char *)dest, *s = (unsigned char *)src;
+	uint32_t *dest_w = (uint32_t *)dest, *src_w = (uint32_t *)src;
+
+	/* If both source and dest are word aligned copy words */
+	if (!((unsigned int)dest_w & 3) && !((unsigned int)src_w & 3)) {
+		for (; n >= 4; n -= 4)
+			*dest_w++ = *src_w++;
+	}
+
+	d = (unsigned char *)dest_w;
+	s = (unsigned char *)src_w;
+
+	/* For remaining or if not aligned, copy bytes */
+	for (; n >= 1; n -= 1)
+		*d++ = *s++;
+
+	return dest;
+
+}
+#endif
+
+EXPORT_SYMBOL(memcpy);
-- 
2.9.3

Back to linux.kernel | Previous | NextPrevious in thread | Next in thread | Find similar | Unroll thread


Thread

[PATCH v4 00/24] OpenRISC patches for 4.11 Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 23/24] openrisc: Export ioremap symbols used by modules Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 13/24] openrisc: Initial support for the idle state Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 18/24] scripts/checkstack.pl: Add openrisc support Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 18/24] scripts/checkstack.pl: Add openrisc support Tobias Klauser <tklauser@distanz.ch> - 2017-02-24 16:00 +0100
      Re: [PATCH v4 18/24] scripts/checkstack.pl: Add openrisc support Stafford Horne <shorne@gmail.com> - 2017-02-24 20:40 +0100
  [PATCH v4 22/24] arch/openrisc/lib/memcpy.c: use correct OR1200 option Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 22/24] arch/openrisc/lib/memcpy.c: use correct OR1200  option Jonas Bonn <jonas@southpole.se> - 2017-02-24 10:50 +0100
      Re: [PATCH v4 22/24] arch/openrisc/lib/memcpy.c: use correct OR1200  option Stafford Horne <shorne@gmail.com> - 2017-02-24 21:00 +0100
  [PATCH v4 08/24] openrisc: add cmpxchg and xchg implementations Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 20/24] openrisc: entry: Fix delay slot detection Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 20/24] openrisc: entry: Fix delay slot detection Jonas Bonn <jonas@southpole.se> - 2017-02-24 11:00 +0100
      Re: [PATCH v4 20/24] openrisc: entry: Fix delay slot detection Stafford Horne <shorne@gmail.com> - 2017-02-24 21:20 +0100
  [PATCH v4 09/24] openrisc: add optimized atomic operations Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 04/24] openrisc: head: use THREAD_SIZE instead of magic constant Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 07/24] openrisc: add atomic bitops Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 07/24] openrisc: add atomic bitops Peter Zijlstra <peterz@infradead.org> - 2017-02-24 12:00 +0100
      Re: [PATCH v4 07/24] openrisc: add atomic bitops Stafford Horne <shorne@gmail.com> - 2017-02-24 20:30 +0100
  [PATCH v4 11/24] openrisc: remove unnecessary stddef.h include Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 01/24] openrisc: use SPARSE_IRQ Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 21/24] openrisc: head: Move init strings to rodata section Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 21/24] openrisc: head: Move init strings to rodata  section Jonas Bonn <jonas@southpole.se> - 2017-02-24 11:00 +0100
      Re: [PATCH v4 21/24] openrisc: head: Move init strings to rodata  section Stafford Horne <shorne@gmail.com> - 2017-02-24 21:20 +0100
  [PATCH v4 05/24] openrisc: head: refactor out tlb flush into it's own function Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 05/24] openrisc: head: refactor out tlb flush into it's  own function Jonas Bonn <jonas@southpole.se> - 2017-02-24 11:00 +0100
      Re: [PATCH v4 05/24] openrisc: head: refactor out tlb flush into  it's own function Stefan Kristiansson <stefan.kristiansson@saunalahti.fi> - 2017-02-24 12:10 +0100
        Re: [PATCH v4 05/24] openrisc: head: refactor out tlb flush into it's  own function Jonas Bonn <jonas@southpole.se> - 2017-02-24 13:50 +0100
          Re: [PATCH v4 05/24] openrisc: head: refactor out tlb flush into  it's own function Stefan Kristiansson <stefan.kristiansson@saunalahti.fi> - 2017-02-24 15:00 +0100
            Re: [PATCH v4 05/24] openrisc: head: refactor out tlb flush into  it's own function Stafford Horne <shorne@gmail.com> - 2017-02-24 20:30 +0100
  [PATCH v4 24/24] openrisc: head: Init r0 to 0 on start Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 24/24] openrisc: head: Init r0 to 0 on start Jonas Bonn <jonas@southpole.se> - 2017-02-24 11:10 +0100
      Re: [PATCH v4 24/24] openrisc: head: Init r0 to 0 on start Stafford Horne <shorne@gmail.com> - 2017-02-24 20:40 +0100
  [PATCH v4 17/24] MAINTAINERS: Add the openrisc official repository Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 12/24] openrisc: Fix the bitmask for the unit present register Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 06/24] openrisc: add l.lwa/l.swa emulation Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 06/24] openrisc: add l.lwa/l.swa emulation Jonas Bonn <jonas@southpole.se> - 2017-02-24 10:50 +0100
      Re: [PATCH v4 06/24] openrisc: add l.lwa/l.swa emulation Stafford Horne <shorne@gmail.com> - 2017-02-24 21:00 +0100
  [PATCH v4 10/24] openrisc: add futex_atomic_* implementations Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 03/24] openrisc: tlb miss handler optimizations Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
  [PATCH v4 19/24] openrisc: entry: Whitespace and comment cleanups Stafford Horne <shorne@gmail.com> - 2017-02-24 05:40 +0100
    Re: [PATCH v4 19/24] openrisc: entry: Whitespace and comment cleanups Jonas Bonn <jonas@southpole.se> - 2017-02-24 10:50 +0100
      Re: [PATCH v4 19/24] openrisc: entry: Whitespace and comment cleanups Stafford Horne <shorne@gmail.com> - 2017-02-24 21:40 +0100
  [PATCH v4 15/24] openrisc: Add optimized memcpy routine Stafford Horne <shorne@gmail.com> - 2017-02-24 05:50 +0100
  [PATCH v4 16/24] openrisc: Add .gitignore Stafford Horne <shorne@gmail.com> - 2017-02-24 05:50 +0100
  [PATCH v4 14/24] openrisc: Add optimized memset Stafford Horne <shorne@gmail.com> - 2017-02-24 05:50 +0100

csiph-web