Move NEON version to a separate function and fix indentation
This commit is contained in:
+51
-42
@@ -954,6 +954,29 @@ void ZSTD_row_update(ZSTD_matchState_t* const ms, const BYTE* ip) {
|
||||
ZSTD_row_update_internal(ms, ip, mls, rowLog, rowMask, 0 /* don't use cache */);
|
||||
}
|
||||
|
||||
/* Returns the mask width of bits group of which will be set to 1. Given not all
|
||||
* architectures have easy movemask instruction, this helps to iterate over
|
||||
* groups of bits easier and faster.
|
||||
*/
|
||||
FORCE_INLINE_TEMPLATE U32
|
||||
ZSTD_row_matchMaskGroupWidth(const U32 rowEntries)
|
||||
{
|
||||
assert((rowEntries == 16) || (rowEntries == 32) || rowEntries == 64);
|
||||
assert(rowEntries <= ZSTD_ROW_HASH_MAX_ENTRIES);
|
||||
#if defined(ZSTD_ARCH_ARM_NEON)
|
||||
if (rowEntries == 16) {
|
||||
return 4;
|
||||
}
|
||||
if (rowEntries == 32) {
|
||||
return 2;
|
||||
}
|
||||
if (rowEntries == 64) {
|
||||
return 1;
|
||||
}
|
||||
#endif
|
||||
return 1;
|
||||
}
|
||||
|
||||
#if defined(ZSTD_ARCH_X86_SSE2)
|
||||
FORCE_INLINE_TEMPLATE ZSTD_VecMask
|
||||
ZSTD_row_getSSEMask(int nbChunks, const BYTE* const src, const BYTE tag, const U32 head)
|
||||
@@ -974,52 +997,11 @@ ZSTD_row_getSSEMask(int nbChunks, const BYTE* const src, const BYTE tag, const U
|
||||
}
|
||||
#endif
|
||||
|
||||
/* Returns the mask width of bits group of which will be set to 1. Given not all
|
||||
* architectures have easy movemask instruction, this helps to iterate over
|
||||
* groups of bits easier and faster.
|
||||
*/
|
||||
FORCE_INLINE_TEMPLATE U32
|
||||
ZSTD_row_matchMaskGroupWidth(const U32 rowEntries) {
|
||||
assert((rowEntries == 16) || (rowEntries == 32) || rowEntries == 64);
|
||||
assert(rowEntries <= ZSTD_ROW_HASH_MAX_ENTRIES);
|
||||
(void)rowEntries;
|
||||
#if defined(ZSTD_ARCH_ARM_NEON)
|
||||
if (rowEntries == 16) {
|
||||
return 4;
|
||||
}
|
||||
if (rowEntries == 32) {
|
||||
return 2;
|
||||
}
|
||||
if (rowEntries == 64) {
|
||||
return 1;
|
||||
}
|
||||
#endif
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* Returns a ZSTD_VecMask (U64) that has the nth group (determined by
|
||||
* ZSTD_row_matchMaskGroupWidth) of bits set to 1 if the newly-computed "tag"
|
||||
* matches the hash at the nth position in a row of the tagTable.
|
||||
* Each row is a circular buffer beginning at the value of "headGrouped". So we
|
||||
* must rotate the "matches" bitfield to match up with the actual layout of the
|
||||
* entries within the hashTable */
|
||||
FORCE_INLINE_TEMPLATE ZSTD_VecMask
|
||||
ZSTD_row_getMatchMask(const BYTE* const tagRow, const BYTE tag, const U32 headGrouped, const U32 rowEntries)
|
||||
ZSTD_row_getNEONMask(const U32 rowEntries, const BYTE* const src, const BYTE tag, const U32 headGrouped)
|
||||
{
|
||||
const BYTE* const src = tagRow + ZSTD_ROW_HASH_TAG_OFFSET;
|
||||
assert((rowEntries == 16) || (rowEntries == 32) || rowEntries == 64);
|
||||
assert(rowEntries <= ZSTD_ROW_HASH_MAX_ENTRIES);
|
||||
assert(ZSTD_row_matchMaskGroupWidth(rowEntries) * rowEntries <= sizeof(ZSTD_VecMask) * 8);
|
||||
|
||||
#if defined(ZSTD_ARCH_X86_SSE2)
|
||||
|
||||
return ZSTD_row_getSSEMask(rowEntries / 16, src, tag, headGrouped);
|
||||
|
||||
#else /* SW or NEON-LE */
|
||||
|
||||
# if defined(ZSTD_ARCH_ARM_NEON)
|
||||
/* This NEON path only works for little endian - otherwise use SWAR below */
|
||||
if (MEM_isLittleEndian()) {
|
||||
if (rowEntries == 16) {
|
||||
/* vshrn_n_u16 shifts by 4 every u16 and narrows to 8 lower bits.
|
||||
* After that groups of 4 bits represent the equalMask. We lower
|
||||
@@ -1061,6 +1043,33 @@ ZSTD_row_getMatchMask(const BYTE* const tagRow, const BYTE tag, const U32 headGr
|
||||
return ZSTD_rotateRight_U64(matches, headGrouped);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
/* Returns a ZSTD_VecMask (U64) that has the nth group (determined by
|
||||
* ZSTD_row_matchMaskGroupWidth) of bits set to 1 if the newly-computed "tag"
|
||||
* matches the hash at the nth position in a row of the tagTable.
|
||||
* Each row is a circular buffer beginning at the value of "headGrouped". So we
|
||||
* must rotate the "matches" bitfield to match up with the actual layout of the
|
||||
* entries within the hashTable */
|
||||
FORCE_INLINE_TEMPLATE ZSTD_VecMask
|
||||
ZSTD_row_getMatchMask(const BYTE* const tagRow, const BYTE tag, const U32 headGrouped, const U32 rowEntries)
|
||||
{
|
||||
const BYTE* const src = tagRow + ZSTD_ROW_HASH_TAG_OFFSET;
|
||||
assert((rowEntries == 16) || (rowEntries == 32) || rowEntries == 64);
|
||||
assert(rowEntries <= ZSTD_ROW_HASH_MAX_ENTRIES);
|
||||
assert(ZSTD_row_matchMaskGroupWidth(rowEntries) * rowEntries <= sizeof(ZSTD_VecMask) * 8);
|
||||
|
||||
#if defined(ZSTD_ARCH_X86_SSE2)
|
||||
|
||||
return ZSTD_row_getSSEMask(rowEntries / 16, src, tag, headGrouped);
|
||||
|
||||
#else /* SW or NEON-LE */
|
||||
|
||||
# if defined(ZSTD_ARCH_ARM_NEON)
|
||||
/* This NEON path only works for little endian - otherwise use SWAR below */
|
||||
if (MEM_isLittleEndian()) {
|
||||
return ZSTD_row_getNEONMask(rowEntries, src, tag, headGrouped);
|
||||
}
|
||||
# endif /* ZSTD_ARCH_ARM_NEON */
|
||||
/* SWAR */
|
||||
{ const size_t chunkSize = sizeof(size_t);
|
||||
|
||||
Reference in New Issue
Block a user