nuttx/libs/libc/machine/risc-v/arch_memcpy.S
ganjing a0fcbb7957 libs/libc/risc-v: Refresh memcpy and memset with XLEN-adaptive loops.
Rewrite arch_memcpy.S and arch_memset.S to be register-width aware on
both RV32 and RV64 using REG_L/REG_S/SZREG macros from asm.h.

memcpy gains:
 - 16xSZREG unrolled main loop (128B/iter on RV64, 64B on RV32).
 - Shift-merge path for misaligned src: reads two aligned words
   straddling each output word and shifts them together, so no load
   or store is ever misaligned.
 - Single SZREG and byte loops for remainder and small copies.

memset gains:
 - 32xSZREG unrolled main loop (256B/iter on RV64, 128B on RV32)
   using Duff's device for non-power-of-two remainders.
 - .option norvc ensures fixed 4-byte instruction width for correct
   jump offset calculation in the Duff's device entry.
 - Zero-length input handled correctly (branch to guarded tail).

The old memcpy always used lw/sw even on RV64, wasting half the
memory bandwidth. The old memset unrolled only 16 bytes per iteration.

Signed-off-by: ganjing <ganjing@xiaomi.com>
2026-08-12 10:24:01 +08:00

238 lines
6.1 KiB
ArmAsm
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

/****************************************************************************
* libs/libc/machine/risc-v/arch_memcpy.S
*
* SPDX-License-Identifier: Apache-2.0
*
* Licensed to the Apache Software Foundation (ASF) under one or more
* contributor license agreements. See the NOTICE file distributed with
* this work for additional information regarding copyright ownership. The
* ASF licenses this file to you under the Apache License, Version 2.0 (the
* "License"); you may not use this file except in compliance with the
* License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
* WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
* License for the specific language governing permissions and limitations
* under the License.
*
****************************************************************************/
/************************************************************************************
* Included Files
************************************************************************************/
#include "libc.h"
#ifdef LIBC_BUILD_MEMCPY
#include "asm.h"
/************************************************************************************
* Pre-processor Definitions
************************************************************************************/
#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__
# define SHL_H srl
# define SHL_L sll
#else
# define SHL_H sll
# define SHL_L srl
#endif
/************************************************************************************
* Public Symbols
************************************************************************************/
.global ARCH_LIBCFUN(memcpy)
.type ARCH_LIBCFUN(memcpy), @function
.file "arch_memcpy.S"
/************************************************************************************
* Name: memcpy
*
* void *memcpy(void *dst, const void *src, size_t n)
*
* Optimized for RISC-V using XLEN-sized load/store with 16×SZREG unrolling.
* Handles unaligned src via shift-merge technique.
************************************************************************************/
.text
.align 2
ARCH_LIBCFUN(memcpy):
.cfi_sections .debug_frame
.cfi_startproc
move t6, a0 /* Preserve return value (dst) */
/* Small copy: size < 3*SZREG → byte-by-byte */
li a3, 3*SZREG
bltu a2, a3, .Lbyte_copy
/* Align dst to SZREG boundary */
andi a3, a0, SZREG-1
beqz a3, .Ldst_aligned
/* Copy head bytes to align dst */
sub a3, zero, a3
addi a3, a3, SZREG /* a3 = bytes to copy = SZREG - misalignment */
sub a2, a2, a3 /* Update remaining count */
.Lalign_head:
lbu a4, 0(a1)
addi a1, a1, 1
sb a4, 0(t6)
addi t6, t6, 1
addi a3, a3, -1
bnez a3, .Lalign_head
.Ldst_aligned:
/* Now dst (t6) is SZREG-aligned. Check if src is also aligned */
andi a3, a1, SZREG-1
bnez a3, .Lunaligned
/* === Aligned path: both src and dst are SZREG-aligned === */
/* Main loop: 16×SZREG per iteration */
andi a4, a2, ~(16*SZREG-1)
beqz a4, .Laligned_tail
add a3, a1, a4
.align 3
.Laligned_loop:
REG_L a4, 0*SZREG(a1)
REG_L a5, 1*SZREG(a1)
REG_L a6, 2*SZREG(a1)
REG_L a7, 3*SZREG(a1)
REG_L t0, 4*SZREG(a1)
REG_L t1, 5*SZREG(a1)
REG_L t2, 6*SZREG(a1)
REG_L t3, 7*SZREG(a1)
REG_L t4, 8*SZREG(a1)
REG_L t5, 9*SZREG(a1)
REG_S a4, 0*SZREG(t6)
REG_S a5, 1*SZREG(t6)
REG_S a6, 2*SZREG(t6)
REG_S a7, 3*SZREG(t6)
REG_S t0, 4*SZREG(t6)
REG_S t1, 5*SZREG(t6)
REG_S t2, 6*SZREG(t6)
REG_S t3, 7*SZREG(t6)
REG_S t4, 8*SZREG(t6)
REG_S t5, 9*SZREG(t6)
REG_L a4, 10*SZREG(a1)
REG_L a5, 11*SZREG(a1)
REG_L a6, 12*SZREG(a1)
REG_L a7, 13*SZREG(a1)
REG_L t0, 14*SZREG(a1)
REG_L t1, 15*SZREG(a1)
addi a1, a1, 16*SZREG
REG_S a4, 10*SZREG(t6)
REG_S a5, 11*SZREG(t6)
REG_S a6, 12*SZREG(t6)
REG_S a7, 13*SZREG(t6)
REG_S t0, 14*SZREG(t6)
REG_S t1, 15*SZREG(t6)
addi t6, t6, 16*SZREG
bltu a1, a3, .Laligned_loop
andi a2, a2, 16*SZREG-1 /* Update remaining count */
.Laligned_tail:
/* Single-word copy for remainder */
andi a4, a2, ~(SZREG-1)
beqz a4, .Lbyte_copy_update
add a3, a1, a4
.Lword_loop:
REG_L a4, 0(a1)
addi a1, a1, SZREG
REG_S a4, 0(t6)
addi t6, t6, SZREG
bltu a1, a3, .Lword_loop
andi a2, a2, SZREG-1 /* Update remaining count */
.Lbyte_copy_update:
/* Fall through to byte copy with updated a2 */
.Lbyte_copy:
/* Byte-by-byte copy for small/tail */
beqz a2, .Ldone
add a3, a1, a2
.Lbyte_loop:
lbu a4, 0(a1)
addi a1, a1, 1
sb a4, 0(t6)
addi t6, t6, 1
bltu a1, a3, .Lbyte_loop
.Ldone:
ret
/* === Unaligned path: dst aligned, src not aligned === */
/* Uses shift-merge to combine two aligned loads into one store */
.Lunaligned:
/* a3 = src misalignment (already computed above) */
slli a6, a3, 3 /* a6 = shift_h = misalign * 8 bits */
sub a7, zero, a6
addi a7, a7, SZREG*8 /* a7 = shift_l = XLEN - shift_h */
/* Save src misalignment for later restore */
mv t4, a3 /* t4 = original misalignment bytes */
/* Align src down to SZREG boundary */
andi a1, a1, ~(SZREG-1)
/* Preload first aligned word from src */
REG_L a5, 0(a1)
/* Calculate loop count: process 2×SZREG per iteration */
andi a4, a2, ~(2*SZREG-1)
beqz a4, .Lunaligned_tail
add a3, t6, a4 /* a3 = end address for dst */
.align 3
.Lunaligned_loop:
REG_L a4, SZREG(a1) /* Load next aligned word */
SHL_H t0, a5, a6 /* High part from previous word */
SHL_L t1, a4, a7 /* Low part from current word */
or t0, t0, t1 /* Combine */
REG_S t0, 0(t6) /* Store to dst */
REG_L a5, 2*SZREG(a1) /* Load next aligned word */
SHL_H t0, a4, a6 /* High part */
SHL_L t1, a5, a7 /* Low part */
or t0, t0, t1 /* Combine */
REG_S t0, SZREG(t6) /* Store to dst */
addi a1, a1, 2*SZREG
addi t6, t6, 2*SZREG
bltu t6, a3, .Lunaligned_loop
andi a2, a2, 2*SZREG-1 /* Update remaining */
.Lunaligned_tail:
/* Restore real src pointer: aligned_src + misalignment */
add a1, a1, t4
j .Lbyte_copy
.cfi_endproc
.size ARCH_LIBCFUN(memcpy), .-ARCH_LIBCFUN(memcpy)
#endif /* LIBC_BUILD_MEMCPY */