mirror of
https://github.com/RfidResearchGroup/proxmark3.git
synced 2026-10-02 12:48:43 +00:00
armsrc: copy words in memcpy, not just bytes
The device ships its own string.c because there is no libc linked, and memcpy was a byte at a time loop: six instructions per byte, 17 cycles per byte measured, 221us for a 624 byte frame. Every copy on the device paid that. Take words when source and destination share their offset within a word, which is the only case ARM7TDMI can do at all, and unroll the byte tail four ways since at -Os the loop bookkeeping otherwise costs more than the copy. 624 bytes aligned goes 221us -> 34us, misaligned 221 -> 125. 192 bytes, was 24. The bootrom does not link string.c. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
89305b2f8b
commit
b66bb2659e
+56
-4
@@ -17,13 +17,65 @@
|
||||
//-----------------------------------------------------------------------------
|
||||
#include "string.h"
|
||||
|
||||
// The word view of the buffers has to be declared as aliasing, the tree builds
|
||||
// with -fstrict-aliasing.
|
||||
typedef uint32_t __attribute__((may_alias)) aliasing_u32;
|
||||
|
||||
void *memcpy(void *dest, const void *src, int len) {
|
||||
uint8_t *d = dest;
|
||||
const uint8_t *s = src;
|
||||
while ((len--) > 0) {
|
||||
*d = *s;
|
||||
d++;
|
||||
s++;
|
||||
|
||||
if (len <= 0) {
|
||||
return dest;
|
||||
}
|
||||
|
||||
// ARM7TDMI cannot do misaligned word access, so words are only usable when
|
||||
// both sides sit at the same offset within a word
|
||||
if ((((uintptr_t)d ^ (uintptr_t)s) & 3) == 0) {
|
||||
|
||||
while ((((uintptr_t)d & 3) != 0) && (len > 0)) {
|
||||
*d++ = *s++;
|
||||
len--;
|
||||
}
|
||||
|
||||
aliasing_u32 *dw = (aliasing_u32 *)d;
|
||||
const aliasing_u32 *sw = (const aliasing_u32 *)s;
|
||||
|
||||
while (len >= 16) {
|
||||
dw[0] = sw[0];
|
||||
dw[1] = sw[1];
|
||||
dw[2] = sw[2];
|
||||
dw[3] = sw[3];
|
||||
dw += 4;
|
||||
sw += 4;
|
||||
len -= 16;
|
||||
}
|
||||
|
||||
while (len >= 4) {
|
||||
*dw++ = *sw++;
|
||||
len -= 4;
|
||||
}
|
||||
|
||||
d = (uint8_t *)dw;
|
||||
s = (const uint8_t *)sw;
|
||||
}
|
||||
|
||||
// Byte tail, and the whole copy when the two sides are misaligned against
|
||||
// each other. Unrolled because at -Os the compiler keeps a counter and
|
||||
// indexes off it, so loop overhead otherwise costs more than the copy.
|
||||
while (len >= 4) {
|
||||
d[0] = s[0];
|
||||
d[1] = s[1];
|
||||
d[2] = s[2];
|
||||
d[3] = s[3];
|
||||
d += 4;
|
||||
s += 4;
|
||||
len -= 4;
|
||||
}
|
||||
|
||||
while (len > 0) {
|
||||
*d++ = *s++;
|
||||
len--;
|
||||
}
|
||||
return dest;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user