mirror of
https://github.com/apache/nuttx.git
synced 2026-08-25 15:37:23 +00:00
The BSD string functions take a word path only when both pointers are
aligned, and a byte path otherwise. A pair at the same offset from a
boundary takes the byte path even though copying or comparing a few leading
bytes aligns both at once, since aligning one aligns the other.
Add MISALIGNED(), which asks whether two pointers disagree about where a
boundary falls, and walk an agreeing pair up to the boundary before the
existing path selection. MISALIGNED4() does the same for the 4-byte path,
so a pair that is 4-byte but not 8-byte aligned reaches the wide path
instead of the middle one. No existing line changes: the walk is a new step
ahead of the current decisions. A pair at differing offsets still takes the
byte path, since no single boundary serves both.
Measured on an EIC7700 EVB (EIC7700X, RV64GC, 1.4GHz) with the BSD string
functions selected and the RISC-V assembly ones disabled, using the
benchmark in apps#3706, medians of 3 runs in MB/s at its largest size:
equal offset aligned
memcpy 414 -> 4148 10.0x 4214 -> 4208
memcmp 41 -> 361 8.8x 362 -> 360
strncmp 28 -> 202 7.4x 207 -> 207
strcmp 42 -> 273 6.5x 278 -> 276
strncpy 377 -> 1676 4.5x 1824 -> 1748
stpncpy 376 -> 1654 4.4x 1843 -> 1724
stpcpy 551 -> 1833 3.3x 1970 -> 1939
memccpy 650 -> 2012 3.1x 2478 -> 2016
strcpy 636 -> 1837 2.9x 1678 -> 1965
Cases the walk never runs for move in both directions by up to a third, the
largest being memccpy at differing offsets, 648 -> 414. Their code is
unchanged, so that is code placement rather than an effect of the change.
The change is architecture independent but has only been measured on
RV64GC. Word size, alignment cost and byte loop codegen all differ
elsewhere, so the balance wants measuring on other architectures.
Assisted-by: Claude:claude-opus-5
Signed-off-by: Justin Hammond <justin@dynam.ac>
154 lines
4.8 KiB
C
154 lines
4.8 KiB
C
/****************************************************************************
|
|
* libs/libc/string/lib_bsdmemccpy.c
|
|
*
|
|
* SPDX-License-Identifier: BSD
|
|
* SPDX-FileCopyrightText: 1994-2009 Red Hat, Inc. All rights reserved
|
|
*
|
|
* Copyright (c) 1994-2009 Red Hat, Inc. All rights reserved.
|
|
*
|
|
* This copyrighted material is made available to anyone wishing to use,
|
|
* modify, copy, or redistribute it subject to the terms and conditions
|
|
* of the BSD License. This program is distributed in the hope that
|
|
* it will be useful, but WITHOUT ANY WARRANTY expressed or implied,
|
|
* including the implied warranties of MERCHANTABILITY or FITNESS FOR
|
|
* A PARTICULAR PURPOSE. A copy of this license is available at
|
|
* http://www.opensource.org/licenses. Any Red Hat trademarks that are
|
|
* incorporated in the source code or documentation are not subject to
|
|
* the BSD License and may only be used or replicated with the express
|
|
* permission of Red Hat, Inc.
|
|
*
|
|
****************************************************************************/
|
|
|
|
/****************************************************************************
|
|
* Included Files
|
|
****************************************************************************/
|
|
|
|
#include <nuttx/config.h>
|
|
#include <sys/types.h>
|
|
#include <string.h>
|
|
|
|
#include "libc.h"
|
|
|
|
/****************************************************************************
|
|
* Pre-processor Definitions
|
|
****************************************************************************/
|
|
|
|
/****************************************************************************
|
|
* Public Functions
|
|
****************************************************************************/
|
|
|
|
/****************************************************************************
|
|
* Name: memccpy
|
|
*
|
|
* Description:
|
|
* The memccpy() function copies bytes from memory area s2 into s1,
|
|
* stopping after the first occurrence of byte c (converted to an unsigned
|
|
* char) is copied, or after n bytes are copied, whichever comes first. If
|
|
* copying takes place between objects that overlap, the behavior is
|
|
* undefined.
|
|
*
|
|
* Returned Value:
|
|
* The memccpy() function returns a pointer to the byte after the copy of c
|
|
* in s1, or a null pointer if c was not found in the first n bytes of s2.
|
|
*
|
|
****************************************************************************/
|
|
|
|
#undef memccpy
|
|
FAR void *memccpy(FAR void *s1, FAR const void *s2, int c, size_t n)
|
|
{
|
|
FAR unsigned char *pout = (FAR unsigned char *)s1;
|
|
FAR const unsigned char *pin = (FAR const unsigned char *)s2;
|
|
unsigned char endchar = c & 0xff;
|
|
|
|
/* Walk a pair that agrees about where a boundary falls up to it, so that
|
|
* the word path below is reached even when the caller aligned neither
|
|
* pointer. The end character is left for the byte loop, which copies it
|
|
* and reports where it landed. Fewer than LITTLEBLOCKSIZE bytes are
|
|
* copied and the size was tested against that first, so this cannot run
|
|
* past the end.
|
|
*/
|
|
|
|
if (!TOO_SMALL(n) && !MISALIGNED(pin, pout) && UNALIGNED_X(pin))
|
|
{
|
|
while (*pin != endchar)
|
|
{
|
|
*pout++ = *pin++;
|
|
n--;
|
|
if (!UNALIGNED_X(pin))
|
|
{
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
/* If the size is small, or either pin or pout is unaligned,
|
|
* then punt into the byte copy loop. This should be rare.
|
|
*/
|
|
|
|
if (!TOO_SMALL(n) && !UNALIGNED(pin, pout))
|
|
{
|
|
FAR libc_data_t *paligned_out = (FAR libc_data_t *)pout;
|
|
FAR const libc_data_t *paligned_in = (FAR libc_data_t *)pin;
|
|
libc_data_t mask = 0;
|
|
unsigned int i;
|
|
|
|
for (i = 0; i < LITTLEBLOCKSIZE; i++)
|
|
{
|
|
mask = (mask << 8) + endchar;
|
|
}
|
|
|
|
while (n >= LITTLEBLOCKSIZE)
|
|
{
|
|
libc_data_t buffer = (libc_data_t)(*paligned_in);
|
|
buffer ^= mask;
|
|
if (DETECTNULL(buffer))
|
|
{
|
|
break;
|
|
}
|
|
|
|
*paligned_out++ = *paligned_in++;
|
|
n -= LITTLEBLOCKSIZE;
|
|
}
|
|
|
|
pout = (FAR unsigned char *)paligned_out;
|
|
pin = (FAR unsigned char *)paligned_in;
|
|
}
|
|
else if (!TOO_SMALL4(n) && !UNALIGNED4(pin, pout))
|
|
{
|
|
FAR uint32_t *paligned_out = (FAR uint32_t *)pout;
|
|
FAR const uint32_t *paligned_in = (FAR uint32_t *)pin;
|
|
uint32_t mask = 0;
|
|
unsigned int i;
|
|
|
|
for (i = 0; i < LITTLEBLOCKSIZE4; i++)
|
|
{
|
|
mask = (mask << 8) + endchar;
|
|
}
|
|
|
|
while (n >= LITTLEBLOCKSIZE4)
|
|
{
|
|
uint32_t buffer = *paligned_in;
|
|
buffer ^= mask;
|
|
if (DETECTNULL32(buffer))
|
|
{
|
|
break;
|
|
}
|
|
|
|
*paligned_out++ = *paligned_in++;
|
|
n -= LITTLEBLOCKSIZE4;
|
|
}
|
|
|
|
pout = (FAR unsigned char *)paligned_out;
|
|
pin = (FAR unsigned char *)paligned_in;
|
|
}
|
|
|
|
while (n--)
|
|
{
|
|
if ((*pout++ = *pin++) == endchar)
|
|
{
|
|
return pout;
|
|
}
|
|
}
|
|
|
|
return NULL;
|
|
}
|