minor refactor zstd_fast

make hot variables more local
This commit is contained in:
Yann Collet
2024-10-07 11:22:40 -07:00
parent e8fce38954
commit 1e7fa242f4
2 changed files with 36 additions and 39 deletions
+3 -1
View File
@@ -560,7 +560,9 @@ MEM_STATIC int ZSTD_cParam_withinBounds(ZSTD_cParameter cParam, int value)
/* ZSTD_selectAddr: /* ZSTD_selectAddr:
* @return a >= b ? trueAddr : falseAddr, * @return a >= b ? trueAddr : falseAddr,
* tries to force branchless codegen. */ * tries to force branchless codegen. */
MEM_STATIC const BYTE* ZSTD_selectAddr(U32 a, U32 b, const BYTE* trueAddr, const BYTE* falseAddr) { MEM_STATIC const BYTE*
ZSTD_selectAddr(U32 a, U32 b, const BYTE* trueAddr, const BYTE* falseAddr)
{
#if defined(__GNUC__) && defined(__x86_64__) #if defined(__GNUC__) && defined(__x86_64__)
__asm__ ( __asm__ (
"cmp %1, %2\n" "cmp %1, %2\n"
+32 -37
View File
@@ -166,7 +166,6 @@ size_t ZSTD_compressBlock_fast_noDict_generic(
* we load from here instead of from tables, if the index is invalid. * we load from here instead of from tables, if the index is invalid.
* Used to avoid unpredictable branches. */ * Used to avoid unpredictable branches. */
const BYTE dummy[] = {0x12,0x34,0x56,0x78,0x9a,0xbc,0xde,0xf0,0xe2,0xb4}; const BYTE dummy[] = {0x12,0x34,0x56,0x78,0x9a,0xbc,0xde,0xf0,0xe2,0xb4};
const BYTE *mvalAddr;
const BYTE* anchor = istart; const BYTE* anchor = istart;
const BYTE* ip0 = istart; const BYTE* ip0 = istart;
@@ -182,7 +181,6 @@ size_t ZSTD_compressBlock_fast_noDict_generic(
size_t hash0; /* hash for ip0 */ size_t hash0; /* hash for ip0 */
size_t hash1; /* hash for ip1 */ size_t hash1; /* hash for ip1 */
U32 idx; /* match idx for ip0 */ U32 idx; /* match idx for ip0 */
U32 mval; /* src value at match idx */
U32 offcode; U32 offcode;
const BYTE* match0; const BYTE* match0;
@@ -255,22 +253,21 @@ _start: /* Requires: ip0 */
* However expression below complies into conditional move. Since * However expression below complies into conditional move. Since
* match is unlikely and we only *branch* on idxl0 > prefixLowestIndex * match is unlikely and we only *branch* on idxl0 > prefixLowestIndex
* if there is a match, all branches become predictable. */ * if there is a match, all branches become predictable. */
mvalAddr = base + idx; { const BYTE* mvalAddr = ZSTD_selectAddr(idx, prefixStartIndex, base + idx, &dummy[0]);
mvalAddr = ZSTD_selectAddr(idx, prefixStartIndex, mvalAddr, &dummy[0]); /* load match for ip[0] */
U32 const mval = MEM_read32(mvalAddr);
/* load match for ip[0] */ /* check match at ip[0] */
mval = MEM_read32(mvalAddr); if (MEM_read32(ip0) == mval && idx >= prefixStartIndex) {
/* found a match! */
/* check match at ip[0] */ /* Write next hash table entry (it's already calculated).
if (MEM_read32(ip0) == mval && idx >= prefixStartIndex) { * This write is known to be safe because the ip1 == ip0 + 1,
/* found a match! */ * so searching will resume after ip1 */
hashTable[hash1] = (U32)(ip1 - base);
/* First write next hash table entry; we've already calculated it. goto _offset;
* This write is known to be safe because the ip1 == ip0 + 1, so }
* we know we will resume searching after ip1 */
hashTable[hash1] = (U32)(ip1 - base);
goto _offset;
} }
/* lookup ip[1] */ /* lookup ip[1] */
@@ -289,32 +286,30 @@ _start: /* Requires: ip0 */
current0 = (U32)(ip0 - base); current0 = (U32)(ip0 - base);
hashTable[hash0] = current0; hashTable[hash0] = current0;
mvalAddr = base + idx; { const BYTE* mvalAddr = ZSTD_selectAddr(idx, prefixStartIndex, base + idx, &dummy[0]);
mvalAddr = ZSTD_selectAddr(idx, prefixStartIndex, mvalAddr, &dummy[0]); /* load match for ip[0] */
U32 const mval = MEM_read32(mvalAddr);
/* load match for ip[0] */ /* check match at ip[0] */
mval = MEM_read32(mvalAddr); if (MEM_read32(ip0) == mval && idx >= prefixStartIndex) {
/* found a match! */
/* first write next hash table entry; we've already calculated it */
if (step <= 4) {
/* We need to avoid writing an index into the hash table >= the
* position at which we will pick up our searching after we've
* taken this match.
*
* The minimum possible match has length 4, so the earliest ip0
* can be after we take this match will be the current ip0 + 4.
* ip1 is ip0 + step - 1. If ip1 is >= ip0 + 4, we can't safely
* write this position.
*/
hashTable[hash1] = (U32)(ip1 - base);
}
/* check match at ip[0] */ goto _offset;
if (MEM_read32(ip0) == mval && idx >= prefixStartIndex) {
/* found a match! */
/* first write next hash table entry; we've already calculated it */
if (step <= 4) {
/* We need to avoid writing an index into the hash table >= the
* position at which we will pick up our searching after we've
* taken this match.
*
* The minimum possible match has length 4, so the earliest ip0
* can be after we take this match will be the current ip0 + 4.
* ip1 is ip0 + step - 1. If ip1 is >= ip0 + 4, we can't safely
* write this position.
*/
hashTable[hash1] = (U32)(ip1 - base);
} }
goto _offset;
} }
/* lookup ip[1] */ /* lookup ip[1] */