#include <machine/asm.h>
ENTRY(memcpy)
push {r0, lr}
subs r2, r2, #4
blo .Lmemcpy_l4
ands r12, r0, #3
bne .Lmemcpy_destul
ands r12, r1, #3
bne .Lmemcpy_srcul
.Lmemcpy_t8:
subs r2, r2, #8
blo .Lmemcpy_l12
subs r2, r2, #0x14
blo .Lmemcpy_l32
push {r4}
.Lmemcpy_loop32:
ldmia r1!, {r3, r4, r12, lr}
stmia r0!, {r3, r4, r12, lr}
ldmia r1!, {r3, r4, r12, lr}
stmia r0!, {r3, r4, r12, lr}
subs r2, r2, #0x20
bhs .Lmemcpy_loop32
cmn r2, #0x10
ldmiahs r1!, {r3, r4, r12, lr}
stmiahs r0!, {r3, r4, r12, lr}
subhs r2, r2, #0x10
pop {r4}
.Lmemcpy_l32:
adds r2, r2, #0x14
.Lmemcpy_loop12:
ldmiahs r1!, {r3, r12, lr}
stmiahs r0!, {r3, r12, lr}
subshs r2, r2, #0x0c
bhs .Lmemcpy_loop12
.Lmemcpy_l12:
adds r2, r2, #8
blo .Lmemcpy_l4
subs r2, r2, #4
ldrlo r3, [r1], #4
strlo r3, [r0], #4
ldmiahs r1!, {r3, r12}
stmiahs r0!, {r3, r12}
subhs r2, r2, #4
.Lmemcpy_l4:
adds r2, r2, #4
#ifdef __APCS_26_
ldmiaeq sp!, {r0, pc}^
#else
popeq {r0, pc}
#endif
cmp r2, #2
ldrb r3, [r1], #1
strb r3, [r0], #1
ldrbhs r3, [r1], #1
strbhs r3, [r0], #1
ldrbhi r3, [r1], #1
strbhi r3, [r0], #1
pop {r0, pc}
.Lmemcpy_destul:
rsb r12, r12, #4
cmp r12, #2
ldrb r3, [r1], #1
strb r3, [r0], #1
ldrbhs r3, [r1], #1
strbhs r3, [r0], #1
ldrbhi r3, [r1], #1
strbhi r3, [r0], #1
subs r2, r2, r12
blo .Lmemcpy_l4
ands r12, r1, #3
beq .Lmemcpy_t8
.Lmemcpy_srcul:
bic r1, r1, #3
ldr lr, [r1], #4
cmp r12, #2
bhi .Lmemcpy_srcul3
beq .Lmemcpy_srcul2
cmp r2, #0x0c
blo .Lmemcpy_srcul1loop4
sub r2, r2, #0x0c
push {r4, r5}
.Lmemcpy_srcul1loop16:
#ifdef __ARMEB__
mov r3, lr, lsl #8
#else
mov r3, lr, lsr #8
#endif
ldmia r1!, {r4, r5, r12, lr}
#ifdef __ARMEB__
orr r3, r3, r4, lsr #24
mov r4, r4, lsl #8
orr r4, r4, r5, lsr #24
mov r5, r5, lsl #8
orr r5, r5, r12, lsr #24
mov r12, r12, lsl #8
orr r12, r12, lr, lsr #24
#else
orr r3, r3, r4, lsl #24
mov r4, r4, lsr #8
orr r4, r4, r5, lsl #24
mov r5, r5, lsr #8
orr r5, r5, r12, lsl #24
mov r12, r12, lsr #8
orr r12, r12, lr, lsl #24
#endif
stmia r0!, {r3-r5, r12}
subs r2, r2, #0x10
bhs .Lmemcpy_srcul1loop16
pop {r4, r5}
adds r2, r2, #0x0c
blo .Lmemcpy_srcul1l4
.Lmemcpy_srcul1loop4:
#ifdef __ARMEB__
mov r12, lr, lsl #8
#else
mov r12, lr, lsr #8
#endif
ldr lr, [r1], #4
#ifdef __ARMEB__
orr r12, r12, lr, lsr #24
#else
orr r12, r12, lr, lsl #24
#endif
str r12, [r0], #4
subs r2, r2, #4
bhs .Lmemcpy_srcul1loop4
.Lmemcpy_srcul1l4:
sub r1, r1, #3
b .Lmemcpy_l4
.Lmemcpy_srcul2:
cmp r2, #0x0c
blo .Lmemcpy_srcul2loop4
sub r2, r2, #0x0c
push {r4, r5}
.Lmemcpy_srcul2loop16:
#ifdef __ARMEB__
mov r3, lr, lsl #16
#else
mov r3, lr, lsr #16
#endif
ldmia r1!, {r4, r5, r12, lr}
#ifdef __ARMEB__
orr r3, r3, r4, lsr #16
mov r4, r4, lsl #16
orr r4, r4, r5, lsr #16
mov r5, r5, lsl #16
orr r5, r5, r12, lsr #16
mov r12, r12, lsl #16
orr r12, r12, lr, lsr #16
#else
orr r3, r3, r4, lsl #16
mov r4, r4, lsr #16
orr r4, r4, r5, lsl #16
mov r5, r5, lsr #16
orr r5, r5, r12, lsl #16
mov r12, r12, lsr #16
orr r12, r12, lr, lsl #16
#endif
stmia r0!, {r3-r5, r12}
subs r2, r2, #0x10
bhs .Lmemcpy_srcul2loop16
pop {r4, r5}
adds r2, r2, #0x0c
blo .Lmemcpy_srcul2l4
.Lmemcpy_srcul2loop4:
#ifdef __ARMEB__
mov r12, lr, lsl #16
#else
mov r12, lr, lsr #16
#endif
ldr lr, [r1], #4
#ifdef __ARMEB__
orr r12, r12, lr, lsr #16
#else
orr r12, r12, lr, lsl #16
#endif
str r12, [r0], #4
subs r2, r2, #4
bhs .Lmemcpy_srcul2loop4
.Lmemcpy_srcul2l4:
sub r1, r1, #2
b .Lmemcpy_l4
.Lmemcpy_srcul3:
cmp r2, #0x0c
blo .Lmemcpy_srcul3loop4
sub r2, r2, #0x0c
push {r4, r5}
.Lmemcpy_srcul3loop16:
#ifdef __ARMEB__
mov r3, lr, lsl #24
#else
mov r3, lr, lsr #24
#endif
ldmia r1!, {r4, r5, r12, lr}
#ifdef __ARMEB__
orr r3, r3, r4, lsr #8
mov r4, r4, lsl #24
orr r4, r4, r5, lsr #8
mov r5, r5, lsl #24
orr r5, r5, r12, lsr #8
mov r12, r12, lsl #24
orr r12, r12, lr, lsr #8
#else
orr r3, r3, r4, lsl #8
mov r4, r4, lsr #24
orr r4, r4, r5, lsl #8
mov r5, r5, lsr #24
orr r5, r5, r12, lsl #8
mov r12, r12, lsr #24
orr r12, r12, lr, lsl #8
#endif
stmia r0!, {r3-r5, r12}
subs r2, r2, #0x10
bhs .Lmemcpy_srcul3loop16
pop {r4, r5}
adds r2, r2, #0x0c
blo .Lmemcpy_srcul3l4
.Lmemcpy_srcul3loop4:
#ifdef __ARMEB__
mov r12, lr, lsl #24
#else
mov r12, lr, lsr #24
#endif
ldr lr, [r1], #4
#ifdef __ARMEB__
orr r12, r12, lr, lsr #8
#else
orr r12, r12, lr, lsl #8
#endif
str r12, [r0], #4
subs r2, r2, #4
bhs .Lmemcpy_srcul3loop4
.Lmemcpy_srcul3l4:
sub r1, r1, #1
b .Lmemcpy_l4
END(memcpy)