libs/libc/risc-v: Refresh memcpy and memset with XLEN-adaptive loops.

Rewrite arch_memcpy.S and arch_memset.S to be register-width aware on
both RV32 and RV64 using REG_L/REG_S/SZREG macros from asm.h.

memcpy gains:
 - 16xSZREG unrolled main loop (128B/iter on RV64, 64B on RV32).
 - Shift-merge path for misaligned src: reads two aligned words
   straddling each output word and shifts them together, so no load
   or store is ever misaligned.
 - Single SZREG and byte loops for remainder and small copies.

memset gains:
 - 32xSZREG unrolled main loop (256B/iter on RV64, 128B on RV32)
   using Duff's device for non-power-of-two remainders.
 - .option norvc ensures fixed 4-byte instruction width for correct
   jump offset calculation in the Duff's device entry.
 - Zero-length input handled correctly (branch to guarded tail).

The old memcpy always used lw/sw even on RV64, wasting half the
memory bandwidth. The old memset unrolled only 16 bytes per iteration.

Signed-off-by: ganjing <ganjing@xiaomi.com>
This commit is contained in:
ganjing 2026-08-11 11:59:21 +08:00 committed by Xiang Xiao
parent 931d5f50d4
commit a0fcbb7957
2 changed files with 277 additions and 169 deletions

View file

@ -28,6 +28,20 @@
#ifdef LIBC_BUILD_MEMCPY
#include "asm.h"
/************************************************************************************
* Pre-processor Definitions
************************************************************************************/
#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
# define SHL_H srl
# define SHL_L sll
#else
# define SHL_H sll
# define SHL_L srl
#endif
/************************************************************************************
* Public Symbols
************************************************************************************/
@ -38,109 +52,187 @@
/************************************************************************************
* Name: memcpy
*
* void *memcpy(void *dst, const void *src, size_t n)
*
* Optimized for RISC-V using XLEN-sized load/store with 16×SZREG unrolling.
* Handles unaligned src via shift-merge technique.
************************************************************************************/
.text
.align 2
ARCH_LIBCFUN(memcpy):
.cfi_sections .debug_frame
.cfi_startproc
move t6, a0 /* Preserve return value */
/* Defer to byte-oriented copy for small sizes */
sltiu a3, a2, 128
bnez a3, 4f
/* Use word-oriented copy only if low-order bits match */
andi a3, t6, 3
andi a4, a1, 3
bne a3, a4, 4f
move t6, a0 /* Preserve return value (dst) */
beqz a3, 2f /* Skip if already aligned */
/*
* Round to nearest double word-aligned address
* greater than or equal to start address
*/
andi a3, a1, ~3
addi a3, a3, 4
/* Handle initial misalignment */
sub a4, a3, a1
1:
lb a5, 0(a1)
addi a1, a1, 1
sb a5, 0(t6)
addi t6, t6, 1
bltu a1, a3, 1b
sub a2, a2, a4 /* Update count */
/* Small copy: size < 3*SZREG → byte-by-byte */
2:
andi a4, a2, ~63
beqz a4, 4f
add a3, a1, a4
3:
lw a4, 0(a1)
lw a5, 4(a1)
lw a6, 2*4(a1)
lw a7, 3*4(a1)
lw t0, 4*4(a1)
lw t1, 5*4(a1)
lw t2, 6*4(a1)
lw t3, 7*4(a1)
lw t4, 8*4(a1)
lw t5, 9*4(a1)
sw a4, 0(t6)
sw a5, 4(t6)
sw a6, 2*4(t6)
sw a7, 3*4(t6)
sw t0, 4*4(t6)
sw t1, 5*4(t6)
sw t2, 6*4(t6)
sw t3, 7*4(t6)
sw t4, 8*4(t6)
sw t5, 9*4(t6)
lw a4, 10*4(a1)
lw a5, 11*4(a1)
lw a6, 12*4(a1)
lw a7, 13*4(a1)
lw t0, 14*4(a1)
lw t1, 15*4(a1)
addi a1, a1, 16*4
sw a4, 10*4(t6)
sw a5, 11*4(t6)
sw a6, 12*4(t6)
sw a7, 13*4(t6)
sw t0, 14*4(t6)
sw t1, 15*4(t6)
addi t6, t6, 16*4
bltu a1, a3, 3b
andi a2, a2, 63 /* Update count */
li a3, 3*SZREG
bltu a2, a3, .Lbyte_copy
4:
/* Handle trailing misalignment */
beqz a2, 6f
add a3, a1, a2
/* Align dst to SZREG boundary */
/* Use word-oriented copy if co-aligned to word boundary */
or a5, a1, t6
or a5, a5, a3
andi a5, a5, 3
bnez a5, 5f
7:
lw a4, 0(a1)
addi a1, a1, 4
sw a4, 0(t6)
addi t6, t6, 4
bltu a1, a3, 7b
andi a3, a0, SZREG-1
beqz a3, .Ldst_aligned
/* Copy head bytes to align dst */
sub a3, zero, a3
addi a3, a3, SZREG /* a3 = bytes to copy = SZREG - misalignment */
sub a2, a2, a3 /* Update remaining count */
.Lalign_head:
lbu a4, 0(a1)
addi a1, a1, 1
sb a4, 0(t6)
addi t6, t6, 1
addi a3, a3, -1
bnez a3, .Lalign_head
.Ldst_aligned:
/* Now dst (t6) is SZREG-aligned. Check if src is also aligned */
andi a3, a1, SZREG-1
bnez a3, .Lunaligned
/* === Aligned path: both src and dst are SZREG-aligned === */
/* Main loop: 16×SZREG per iteration */
andi a4, a2, ~(16*SZREG-1)
beqz a4, .Laligned_tail
add a3, a1, a4
.align 3
.Laligned_loop:
REG_L a4, 0*SZREG(a1)
REG_L a5, 1*SZREG(a1)
REG_L a6, 2*SZREG(a1)
REG_L a7, 3*SZREG(a1)
REG_L t0, 4*SZREG(a1)
REG_L t1, 5*SZREG(a1)
REG_L t2, 6*SZREG(a1)
REG_L t3, 7*SZREG(a1)
REG_L t4, 8*SZREG(a1)
REG_L t5, 9*SZREG(a1)
REG_S a4, 0*SZREG(t6)
REG_S a5, 1*SZREG(t6)
REG_S a6, 2*SZREG(t6)
REG_S a7, 3*SZREG(t6)
REG_S t0, 4*SZREG(t6)
REG_S t1, 5*SZREG(t6)
REG_S t2, 6*SZREG(t6)
REG_S t3, 7*SZREG(t6)
REG_S t4, 8*SZREG(t6)
REG_S t5, 9*SZREG(t6)
REG_L a4, 10*SZREG(a1)
REG_L a5, 11*SZREG(a1)
REG_L a6, 12*SZREG(a1)
REG_L a7, 13*SZREG(a1)
REG_L t0, 14*SZREG(a1)
REG_L t1, 15*SZREG(a1)
addi a1, a1, 16*SZREG
REG_S a4, 10*SZREG(t6)
REG_S a5, 11*SZREG(t6)
REG_S a6, 12*SZREG(t6)
REG_S a7, 13*SZREG(t6)
REG_S t0, 14*SZREG(t6)
REG_S t1, 15*SZREG(t6)
addi t6, t6, 16*SZREG
bltu a1, a3, .Laligned_loop
andi a2, a2, 16*SZREG-1 /* Update remaining count */
.Laligned_tail:
/* Single-word copy for remainder */
andi a4, a2, ~(SZREG-1)
beqz a4, .Lbyte_copy_update
add a3, a1, a4
.Lword_loop:
REG_L a4, 0(a1)
addi a1, a1, SZREG
REG_S a4, 0(t6)
addi t6, t6, SZREG
bltu a1, a3, .Lword_loop
andi a2, a2, SZREG-1 /* Update remaining count */
.Lbyte_copy_update:
/* Fall through to byte copy with updated a2 */
.Lbyte_copy:
/* Byte-by-byte copy for small/tail */
beqz a2, .Ldone
add a3, a1, a2
.Lbyte_loop:
lbu a4, 0(a1)
addi a1, a1, 1
sb a4, 0(t6)
addi t6, t6, 1
bltu a1, a3, .Lbyte_loop
.Ldone:
ret
5:
lb a4, 0(a1)
addi a1, a1, 1
sb a4, 0(t6)
addi t6, t6, 1
bltu a1, a3, 5b
6:
ret
/* === Unaligned path: dst aligned, src not aligned === */
/* Uses shift-merge to combine two aligned loads into one store */
.Lunaligned:
/* a3 = src misalignment (already computed above) */
slli a6, a3, 3 /* a6 = shift_h = misalign * 8 bits */
sub a7, zero, a6
addi a7, a7, SZREG*8 /* a7 = shift_l = XLEN - shift_h */
/* Save src misalignment for later restore */
mv t4, a3 /* t4 = original misalignment bytes */
/* Align src down to SZREG boundary */
andi a1, a1, ~(SZREG-1)
/* Preload first aligned word from src */
REG_L a5, 0(a1)
/* Calculate loop count: process 2×SZREG per iteration */
andi a4, a2, ~(2*SZREG-1)
beqz a4, .Lunaligned_tail
add a3, t6, a4 /* a3 = end address for dst */
.align 3
.Lunaligned_loop:
REG_L a4, SZREG(a1) /* Load next aligned word */
SHL_H t0, a5, a6 /* High part from previous word */
SHL_L t1, a4, a7 /* Low part from current word */
or t0, t0, t1 /* Combine */
REG_S t0, 0(t6) /* Store to dst */
REG_L a5, 2*SZREG(a1) /* Load next aligned word */
SHL_H t0, a4, a6 /* High part */
SHL_L t1, a5, a7 /* Low part */
or t0, t0, t1 /* Combine */
REG_S t0, SZREG(t6) /* Store to dst */
addi a1, a1, 2*SZREG
addi t6, t6, 2*SZREG
bltu t6, a3, .Lunaligned_loop
andi a2, a2, 2*SZREG-1 /* Update remaining */
.Lunaligned_tail:
/* Restore real src pointer: aligned_src + misalignment */
add a1, a1, t4
j .Lbyte_copy
.cfi_endproc
#endif
.size ARCH_LIBCFUN(memcpy), .-ARCH_LIBCFUN(memcpy)
#endif /* LIBC_BUILD_MEMCPY */

View file

@ -18,98 +18,114 @@
#ifdef LIBC_BUILD_MEMSET
#include "asm.h"
.text
.global ARCH_LIBCFUN(memset)
.type ARCH_LIBCFUN(memset), @function
.align 2
/************************************************************************************
* Name: memset
*
* void *memset(void *s, int c, size_t n)
*
* Optimized with 32xSZREG unrolled loop and .option norvc.
************************************************************************************/
ARCH_LIBCFUN(memset):
.cfi_sections .debug_frame
.cfi_startproc
li t1, 15
move a4, a0
bleu a2, t1, .Ltiny
and a5, a4, 15
bnez a5, .Lmisaligned
mv t0, a0
li t1, 2*SZREG
bleu a2, t1, .Lbyte_tail
andi a1, a1, 0xff
slli t1, a1, 8
or a1, a1, t1
slli t1, a1, 16
or a1, a1, t1
#ifdef CONFIG_ARCH_RV64
slli t1, a1, 32
or a1, a1, t1
#endif
andi t1, t0, SZREG-1
beqz t1, .Laligned
sub t2, zero, t1
addi t2, t2, SZREG
sub a2, a2, t2
.Lalign_head:
sb a1, 0(t0)
addi t0, t0, 1
addi t2, t2, -1
bnez t2, .Lalign_head
.Laligned:
bnez a1, .Lwordify
li t3, 32*SZREG
bltu a2, t3, .Lword_tail
.Lwordified:
and a3, a2, ~15
and a2, a2, 15
add a3, a3, a4
.align 3
.Lblock_loop:
.option push
.option norvc
REG_S a1, 0*SZREG(t0)
REG_S a1, 1*SZREG(t0)
REG_S a1, 2*SZREG(t0)
REG_S a1, 3*SZREG(t0)
REG_S a1, 4*SZREG(t0)
REG_S a1, 5*SZREG(t0)
REG_S a1, 6*SZREG(t0)
REG_S a1, 7*SZREG(t0)
REG_S a1, 8*SZREG(t0)
REG_S a1, 9*SZREG(t0)
REG_S a1, 10*SZREG(t0)
REG_S a1, 11*SZREG(t0)
REG_S a1, 12*SZREG(t0)
REG_S a1, 13*SZREG(t0)
REG_S a1, 14*SZREG(t0)
REG_S a1, 15*SZREG(t0)
REG_S a1, 16*SZREG(t0)
REG_S a1, 17*SZREG(t0)
REG_S a1, 18*SZREG(t0)
REG_S a1, 19*SZREG(t0)
REG_S a1, 20*SZREG(t0)
REG_S a1, 21*SZREG(t0)
REG_S a1, 22*SZREG(t0)
REG_S a1, 23*SZREG(t0)
REG_S a1, 24*SZREG(t0)
REG_S a1, 25*SZREG(t0)
REG_S a1, 26*SZREG(t0)
REG_S a1, 27*SZREG(t0)
REG_S a1, 28*SZREG(t0)
REG_S a1, 29*SZREG(t0)
REG_S a1, 30*SZREG(t0)
REG_S a1, 31*SZREG(t0)
.option pop
addi t0, t0, 32*SZREG
sub a2, a2, t3
bgeu a2, t3, .Lblock_loop
#if __riscv_xlen == 64
1:
sd a1, 0(a4)
sd a1, 8(a4)
#else
1:
sw a1, 0(a4)
sw a1, 4(a4)
sw a1, 8(a4)
sw a1, 12(a4)
#endif
add a4, a4, 16
bltu a4, a3, 1b
.Lword_tail:
andi t1, a2, ~(SZREG-1)
beqz t1, .Lbyte_tail
add t2, t0, t1
.Lword_loop:
REG_S a1, 0(t0)
addi t0, t0, SZREG
bltu t0, t2, .Lword_loop
andi a2, a2, SZREG-1
bnez a2, .Ltiny
.Lbyte_tail:
beqz a2, .Ldone
.Lbyte_fill:
sb a1, 0(t0)
addi t0, t0, 1
addi a2, a2, -1
bnez a2, .Lbyte_fill
.Ldone:
ret
.Ltiny:
sub a3, t1, a2
sll a3, a3, 2
1:
auipc t0, %pcrel_hi(.Ltable)
add a3, a3, t0
.option push
.option norvc
.Ltable_misaligned:
jr a3, %pcrel_lo(1b)
.Ltable:
sb a1,14(a4)
sb a1,13(a4)
sb a1,12(a4)
sb a1,11(a4)
sb a1,10(a4)
sb a1, 9(a4)
sb a1, 8(a4)
sb a1, 7(a4)
sb a1, 6(a4)
sb a1, 5(a4)
sb a1, 4(a4)
sb a1, 3(a4)
sb a1, 2(a4)
sb a1, 1(a4)
sb a1, 0(a4)
.option pop
ret
.Lwordify:
and a1, a1, 0xFF
sll a3, a1, 8
or a1, a1, a3
sll a3, a1, 16
or a1, a1, a3
#if __riscv_xlen == 64
sll a3, a1, 32
or a1, a1, a3
#endif
j .Lwordified
.Lmisaligned:
sll a3, a5, 2
1:
auipc t0, %pcrel_hi(.Ltable_misaligned)
add a3, a3, t0
mv t0, ra
jalr a3, %pcrel_lo(1b)
mv ra, t0
add a5, a5, -16
sub a4, a4, a5
add a2, a2, a5
bleu a2, t1, .Ltiny
j .Laligned
.cfi_endproc
.size ARCH_LIBCFUN(memset), .-ARCH_LIBCFUN(memset)