mirror of
https://github.com/espressif/esp-nimble.git
synced 2026-09-28 11:47:23 +00:00
nimble/phy/nrf52: Optimize PDU copy
This optimizes ble_phy_rxpdu_copy by (1) using uint32_t instead of
uint16_t for calculations and (2) adding inline assembly to perform
actual copying:
(1) means there is no need to sign-extend values all the time,
(2) means there are no e.g. redundant cmp instructions which GCC has
tendency to use.
Overall, optimized copying is over 2 times faster than old routine
when copying long PDUs.
This commit is contained in:
@@ -397,69 +397,80 @@ ble_phy_get_cur_phy(void)
|
|||||||
void
|
void
|
||||||
ble_phy_rxpdu_copy(uint8_t *dptr, struct os_mbuf *rxpdu)
|
ble_phy_rxpdu_copy(uint8_t *dptr, struct os_mbuf *rxpdu)
|
||||||
{
|
{
|
||||||
uint16_t rem_bytes;
|
uint32_t rem_len;
|
||||||
uint16_t mb_bytes;
|
uint32_t copy_len;
|
||||||
uint16_t copylen;
|
uint32_t block_len;
|
||||||
uint32_t *dst;
|
void *dst;
|
||||||
uint32_t *src;
|
void *src;
|
||||||
struct os_mbuf *m;
|
struct os_mbuf * om;
|
||||||
struct ble_mbuf_hdr *ble_hdr;
|
|
||||||
struct os_mbuf_pkthdr *pkthdr;
|
|
||||||
|
|
||||||
/* Better be aligned */
|
/* Better be aligned */
|
||||||
assert(((uint32_t)dptr & 3) == 0);
|
assert(((uint32_t)dptr & 3) == 0);
|
||||||
|
|
||||||
pkthdr = OS_MBUF_PKTHDR(rxpdu);
|
block_len = rxpdu->om_omp->omp_databuf_len;
|
||||||
rem_bytes = pkthdr->omp_len;
|
rem_len = OS_MBUF_PKTHDR(rxpdu)->omp_len;
|
||||||
|
src = dptr;
|
||||||
|
|
||||||
/* Fill in the mbuf pkthdr first. */
|
/*
|
||||||
dst = (uint32_t *)(rxpdu->om_data);
|
* Setup for copying from first mbuf which is shorter due to packet header
|
||||||
src = (uint32_t *)dptr;
|
* and extra leading space
|
||||||
|
*/
|
||||||
|
copy_len = block_len - rxpdu->om_pkthdr_len - 4;
|
||||||
|
om = rxpdu;
|
||||||
|
dst = om->om_data;
|
||||||
|
|
||||||
mb_bytes = (rxpdu->om_omp->omp_databuf_len - rxpdu->om_pkthdr_len - 4);
|
while (om) {
|
||||||
copylen = min(mb_bytes, rem_bytes);
|
/*
|
||||||
copylen &= 0xFFFC;
|
* Always copy blocks of length aligned to word size, only last mbuf
|
||||||
rem_bytes -= copylen;
|
* will have remaining non-word size bytes appended.
|
||||||
mb_bytes -= copylen;
|
*/
|
||||||
rxpdu->om_len = copylen;
|
copy_len = min(copy_len, rem_len);
|
||||||
while (copylen > 0) {
|
copy_len &= ~3;
|
||||||
*dst = *src;
|
|
||||||
++dst;
|
|
||||||
++src;
|
|
||||||
copylen -= 4;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Copy remaining bytes */
|
dst = om->om_data;
|
||||||
m = rxpdu;
|
om->om_len = copy_len;
|
||||||
while (rem_bytes > 0) {
|
rem_len -= copy_len;
|
||||||
/* If there are enough bytes in the mbuf, copy them and leave */
|
|
||||||
if (rem_bytes <= mb_bytes) {
|
__asm__ volatile (".syntax unified \n"
|
||||||
memcpy(m->om_data + m->om_len, src, rem_bytes);
|
" mov r4, %[len] \n"
|
||||||
m->om_len += rem_bytes;
|
" b 2f \n"
|
||||||
|
"1: ldr r3, [%[src], %[len]] \n"
|
||||||
|
" str r3, [%[dst], %[len]] \n"
|
||||||
|
"2: subs %[len], #4 \n"
|
||||||
|
" bpl 1b \n"
|
||||||
|
" adds %[src], %[src], r4 \n"
|
||||||
|
" adds %[dst], %[dst], r4 \n"
|
||||||
|
: [dst] "+r" (dst), [src] "+r" (src),
|
||||||
|
[len] "+r" (copy_len)
|
||||||
|
:
|
||||||
|
: "r3", "r4", "memory"
|
||||||
|
);
|
||||||
|
|
||||||
|
if (rem_len < 4) {
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
m = SLIST_NEXT(m, om_next);
|
/* Move to next mbuf */
|
||||||
assert(m != NULL);
|
om = SLIST_NEXT(om, om_next);
|
||||||
|
copy_len = block_len;
|
||||||
mb_bytes = m->om_omp->omp_databuf_len;
|
|
||||||
copylen = min(mb_bytes, rem_bytes);
|
|
||||||
copylen &= 0xFFFC;
|
|
||||||
rem_bytes -= copylen;
|
|
||||||
mb_bytes -= copylen;
|
|
||||||
m->om_len = copylen;
|
|
||||||
dst = (uint32_t *)m->om_data;
|
|
||||||
while (copylen > 0) {
|
|
||||||
*dst = *src;
|
|
||||||
++dst;
|
|
||||||
++src;
|
|
||||||
copylen -= 4;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Copy ble header */
|
/* Copy remaining bytes, if any, to last mbuf */
|
||||||
ble_hdr = BLE_MBUF_HDR_PTR(rxpdu);
|
om->om_len += rem_len;
|
||||||
memcpy(ble_hdr, &g_ble_phy_data.rxhdr, sizeof(struct ble_mbuf_hdr));
|
__asm__ volatile (".syntax unified \n"
|
||||||
|
" b 2f \n"
|
||||||
|
"1: ldrb r3, [%[src], %[len]] \n"
|
||||||
|
" strb r3, [%[dst], %[len]] \n"
|
||||||
|
"2: subs %[len], #1 \n"
|
||||||
|
" bpl 1b \n"
|
||||||
|
: [len] "+r" (rem_len)
|
||||||
|
: [dst] "r" (dst), [src] "r" (src)
|
||||||
|
: "r3", "memory"
|
||||||
|
);
|
||||||
|
|
||||||
|
/* Copy header */
|
||||||
|
memcpy(BLE_MBUF_HDR_PTR(rxpdu), &g_ble_phy_data.rxhdr,
|
||||||
|
sizeof(struct ble_mbuf_hdr));
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
Reference in New Issue
Block a user