From 0c66a44d1bcc0b7eae7f8ef52d6008541abdb7b1 Mon Sep 17 00:00:00 2001 From: Yann Collet Date: Tue, 28 Aug 2018 15:47:07 -0700 Subject: [PATCH] first working test program measures : - compression ratio with / without dictionary - create one dictionary per block - memory budget for dictionaries - decompression speed, using one different dictionary per block current limitations : - only one file - 4K blocks only - automatic dictionary built with 4K size dictionary can be selected on command line, with -D --- contrib/largeNbDicts/.gitignore | 2 + contrib/largeNbDicts/Makefile | 15 ++- contrib/largeNbDicts/largeNbDicts | Bin 14034 -> 0 bytes contrib/largeNbDicts/largeNbDicts.c | 156 ++++++++++++++++++++++++++-- programs/bench.c | 4 +- 5 files changed, 162 insertions(+), 15 deletions(-) create mode 100644 contrib/largeNbDicts/.gitignore delete mode 100755 contrib/largeNbDicts/largeNbDicts diff --git a/contrib/largeNbDicts/.gitignore b/contrib/largeNbDicts/.gitignore new file mode 100644 index 000000000..e77c4e496 --- /dev/null +++ b/contrib/largeNbDicts/.gitignore @@ -0,0 +1,2 @@ +# build artifacts +largeNbDicts diff --git a/contrib/largeNbDicts/Makefile b/contrib/largeNbDicts/Makefile index 026d76f12..f4b060ae0 100644 --- a/contrib/largeNbDicts/Makefile +++ b/contrib/largeNbDicts/Makefile @@ -7,8 +7,10 @@ # in the COPYING file in the root directory of this source tree). # ################################################################ +PROGDIR = ../../programs +LIBDIR = ../../lib -CPPFLAGS+= -I../../lib -I../../lib/common -I../../lib/dictBuilder -I../../programs +CPPFLAGS+= -I$(LIBDIR) -I$(LIBDIR)/common -I$(LIBDIR)/dictBuilder -I$(PROGDIR) CFLAGS ?= -O3 DEBUGFLAGS= -Wall -Wextra -Wcast-qual -Wcast-align -Wshadow \ @@ -24,9 +26,18 @@ default: largeNbDicts all : largeNbDicts largeNbDicts: LDFLAGS += -lzstd -largeNbDicts: largeNbDicts.c +largeNbDicts: bench.o datagen.o xxhash.o largeNbDicts.c $(CC) $(CPPFLAGS) $(CFLAGS) $^ $(LDFLAGS) -o $@ +bench.o : $(PROGDIR)/bench.c + $(CC) $(CPPFLAGS) $(CFLAGS) $^ -c + +datagen.o: $(PROGDIR)/datagen.c + $(CC) $(CPPFLAGS) $(CFLAGS) $^ -c + +xxhash.o : $(LIBDIR)/common/xxhash.c + $(CC) $(CPPFLAGS) $(CFLAGS) $^ -c clean: + $(RM) *.o $(RM) largeNbDicts diff --git a/contrib/largeNbDicts/largeNbDicts b/contrib/largeNbDicts/largeNbDicts deleted file mode 100755 index c057a2b78aa551a2de831b6e304f8747a6ea3d0f..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 14034 zcmX^A>+L^w1_nlE28ISE1_lNJ1_p)+tPBjT39lG3DNC=b(pEm9Ek>YyrMd?=TJ18N?^eIWDV zGg5O3Qj4&-k3||{-Xo|1AU-JEpiom1pLq#AoKE<%9XC?wcYHF)sn4 zodLv0Hv=jKragc@%1 zI6#U)Sb>27rWeEo#iuBU0mbq0c{%aLmAOgzIq?N0MGW!rsP5x{x(}3pKw3b2bo0bO z5>Ol;pOc8sJPD|IE1>E@d}Q-L{*{2rfhbV8LGr1Or=Pd0izh6P8K8xm0Z26i!xfMR z85kHq=791GM3jL+iGiU3ti*tU0TebI1`G_av;|UQV8FnzgOP#Zg&_k&1p@;EC@w+n zC@^ARkY`|E5HV$7IKarj0Lp#=APrD8Aax)sL1v0W#j#NdCJYR^SS0uu7{DbG4+8^( zera)$eokhReoAFd3RJB$0|Nud9d9k&nJ<5}jb4!?>6xKZt*!zs(-^oI7`Pa;U_7WU z4F-k=kggA);Dd^RD3C1``QgCH)Noua16I2VxS_Uqdf>B~L z1V%$(Gz3ONU^E0qLtr!nMniz25D4|?eCinP80Hx27!vH!{6@l~^Rq{1?FEl+R|$_! z*AqUyCCvW~FZpzS@c91Ov-7x5ugX!7g+86nUu@uJVDRib3Suq=GmrSTzAbV0ZGBSW zu$O zqXFd@fH($Fjs=Kg;lX&rqxlerNAs~CrTreQCrcDOdrhu*_J&+#nBdXL?0Cy8Qj&-C6VJ2go-*ofD16#^d3M0Hf}}rjGBBj2rRnj@ zw}68NC4K`uJI{M|Uh(Ms?AiIxqxH5&H|rlhP{>-8$a(adx^puy7#{HGy!B!qCn(&% zdvxA=v73Q`;dP8l=V6atQ#+6~p#G-effutm85s71*e_Ocf|CI#?%no*M0`5m`*c3? z>HO)~dEUnm?ad;kg6M!iH;L&`91J(OXV2_zL zf_=-&3bJ@D$bUj0<9cl$gT+mgz!dLKE(V5On?cf_Il$h>;{T9fkH$9>KUAFfj0MZw>kX|Gz532L=XP&^QTy>kN<{gzAG()s5Ixm*)9& zmu~QE{`bEm#iQ4>l?UYa`!6y%Kp}M8xAlLCuTSUq7v&rv@xvaiw@VcEf%;V*o%ek@ zKlyb2eBs5xz~E!~p@bLYGG`7323NyxhPQn>|G$s}8zQ6P+gYN*0U`umOL-o5Q2}QN z5CfbeJdd-0I(!TsjYmM91qCWNMPW@39*u8afHGp|H;>NWKHawOc^DWxyX`%CP4{!c zlEP+oP&01m8sFV_6~|KFp#6~c90;n8}4zhyNuSleg*y55Zc|Np1)C-F`N#RGo~ z?_4llbMh0v0P8|9TeBAw?w|N04|?>Pws0~qY+!uh53-}XwE%2~i;4iuf^M*R9-Tfa z93Gu4DjvOi75@MK4^Cq*oPUEI0uu1(bUgreh$Ay7G0!#l|NlS48=0I83_C$-%3sHL*)Oo;KwPlHo-u(U-1UM-_goE-alLyvz$(9a zbce3+=)CFE?YiMb;Gh5heY&@TWPDq{@wZ$6Yl=PL(^)&gr+cZu|NsAYfd)W)I=_SL zc)|)ca|hUh!%PeeKHYmAK!$fu^#D<@gwS1k0aWUChA!~v_C4U!?RvtaJ9dL-=gk)a zpy28}2r7147J@A6W_=6FP~8@39{hPHLGgH&je#MJKZf@zn65dQ#;?%|iN82d{4M2R zU;w*m1LKP>kbAmacYtFGY-#Nbkj32yi$Rvo23Z8Q)U)$Ge~Sl*+s#@EvbXcFN4GWH z>L9R{F}zV=8pG-a4p7o@>3sA;kd=YK@EfQWy34}A;K{%KDAbq(Y5Ym7++gi7tioWr z=D;WZNLFz+P?|QahZ^zg#W@xR2CrV5x1b{P#Z8bYoscwef{}p%RIqs(-cI9}cL5~` zaNy^H)v(+L2YwHzg!q5h$MSX21E0>{X~^ZTN4NEhAQlD&kLJT1p8wC5z6WLHY>+^A z?T!~lEMOB~L_q{DyaHe{93ou;kzNjw)&NQC zfTS%T(xDLP4v4e@NLn8xEeVk}gGgsWr1?S8Mj+{5pmfx0D-MzNf=K&-q(Nh;FK$Dm zQ$PhxucOqnUe+#H0GQ90-_}`4J7>%B<%%}PJ>9#f=JH;NxuO}Ycn!1fb$zTKEi^*^^>VT zsC4e;^XRqx#lpa_4>TgW3sil+07Zspx1C3?=?6%j=)Cpf7XxIRmLb|PF7_}iyujv# zJBEQ;yPcnXx{aIvGuEm4bXRk@8a`?M&r}lH{GX*%#G|`fz@yjnAPWP-i>nNv&?<58 z=$1VPa&9-bN3U%qssrYOihyowk6zP!umgNLKfS1AU|?|E19IPM0mE;vSv*KVu9NIFMB{NMn(pPmrWp+D2N5BO*1q>tUQpIIf#`4V!47?Q6N?*hy|*AGLk_o zH;`BXh-Cv})q_|@AXXoU1*&E<7K2!F|Ns97c?gCLoEaDlTwy#24H5&fVB#C>iVJcw zOPos6(o%~UauSQuQ~i=$GLuV+^^zI#lALn#le3Ez>}(Yb8S+xg!Q6n7A_W^;h2o-Q z*Sr*loczR;%)E4kl+5Ik%>2B>qDlqTVg)XSyEth5R&y5|9-o3dNaKsS2v4 z3Q3hEsc_|~Ad4Y}Fo4X1xJp6MR>3dS$A>{NDmcU?KC{@hs3^Zk1Ee50vno{+?gZ6h zJv}`IE{3GkyyOgq;-X}Te!YT{BDfh~XQifqT(6J>_B6;*5ZCA#rYZP3DR6{PIBo2a+&gNY2kKC`v6Z1_ga$i9$|lS!xc*edU=Y8Tq9p z$nj*Q5R_PwnQyDAXOyO(paF{wO;GH?%|E1NaTSW1&K5TJSL{6=A{-v zQiN(L!WEgR#X3+URiWXZpQezTnwwu#sldgckdz3rF)=+=!3xAxEmla&%t_5l%uQ8@ z)^$OcQyHtk#h~lLP?Q7@0dNQxGoXicUJ@i^ixq6ai5RR5Dx?ROWJpdfsQ~#MOUg?I zYXJqkCM21sx`4vpIX@>S6`TneQb1vXNG}j>G)T}dDHd)jXqJe9K?XW&2JXx-Fff2< z&^(kNHw)v%2v$Z0eg*~`1_%btNP}`Oh!3jmcog@=>RAl1*KDXdR=B1<-F@WY%8HywX z9pa;0Je@=0ONtUR^W2K^b0Nbd3=n6G{`Y9GdxgaU|<4|rzR(+#FrK)rl&GwFf)8m zWngAlV9dbG@WB{lCj$dRgWLn=2Bry23%DjQg9gs0GBPmmF)}bb5QL0#t%33dBp`f{ zCH#!wIeqX57HGvhKO+Of0!fIxB52tGBLl++AqZas%6}jL;afrZp!M9KfiBP*ZjgS^ zco~TA%K%mfVuD==T0PDP76Nm@E83Cy-Dv8UqVcz(@lT-fZ=vx&q47bZame<8rXi8} zpqU*;$Rc~N13|NF$b3tPVo{|x#GAL@BufPzGC7;A6!IOGCWG*2dBeTSVb4omNj>1t`#)D@q zP>V@aF>pzTDui4#LS+zzB8mVg12Z6tK#DJj2xbuo5l0k$2tKktSb-NGACFYjA=MkO z@-99NT+qd*K})#!G;q-dqEZ>+(;$UeJhT*xM=8GIi%Y=e7Dnv@4Qxm`6`xj=pX-*H Qlgbbe4?^%v3piB(0Dn~WwEzGB diff --git a/contrib/largeNbDicts/largeNbDicts.c b/contrib/largeNbDicts/largeNbDicts.c index 536e45ffe..168766018 100644 --- a/contrib/largeNbDicts/largeNbDicts.c +++ b/contrib/largeNbDicts/largeNbDicts.c @@ -12,7 +12,7 @@ * This is a benchmark test tool * dedicated to the specific case of dictionary decompression * using a very large nb of dictionaries - * thus generating many cache-misses. + * thus suffering latency from lots of cache misses. * It's created in a bid to investigate performance and find optimizations. */ @@ -24,6 +24,7 @@ #include /* assert */ #include "util.h" +#include "bench.h" #define ZSTD_STATIC_LINKING_ONLY #include "zstd.h" #include "zdict.h" @@ -158,6 +159,17 @@ buffer_collection_t splitBuffer(buffer_t srcBuffer, size_t blockSize) return result; } +/* shrinkSizes() : + * update sizes in buffer collection */ +void shrinkSizes(buffer_collection_t collection, + const size_t* sizes) /* presumed same size as collection */ +{ + size_t const nbBlocks = collection.nbBuffers; + for (size_t blockNb = 0; blockNb < nbBlocks; blockNb++) { + assert(sizes[blockNb] <= collection.capacities[blockNb]); + collection.capacities[blockNb] = sizes[blockNb]; + } +} /*--- dictionary creation ---*/ @@ -221,13 +233,30 @@ static ddict_collection_t createDDictCollection(const void* dictBuffer, size_t d } +/* mess with adresses, so that linear scanning dictionaries != linear address scanning */ +void shuffleDictionaries(ddict_collection_t dicts) +{ + size_t const nbDicts = dicts.nbDDict; + for (size_t r=0; rdctx, + dst, dstCapacity, + src, srcSize, + di->dictionaries.ddicts[di->blockNb]); + + di->blockNb = di->blockNb + 1; + if (di->blockNb >= di->nbBlocks) di->blockNb = 0; + + return result; +} + + +#define BENCH_TIME_DEFAULT_MS 6000 +#define RUN_TIME_DEFAULT_MS 1000 + +static int benchMem(buffer_collection_t dstBlocks, + buffer_collection_t srcBlocks, + ddict_collection_t dictionaries) +{ + assert(dstBlocks.nbBuffers == srcBlocks.nbBuffers); + assert(dstBlocks.nbBuffers == dictionaries.nbDDict); + + double bestSpeed = 0.; + + BMK_timedFnState_t* const benchState = + BMK_createTimedFnState(BENCH_TIME_DEFAULT_MS, RUN_TIME_DEFAULT_MS); + decompressInstructions di = createDecompressInstructions(dictionaries); + + for (;;) { + BMK_runOutcome_t const outcome = BMK_benchTimedFn(benchState, + decompress, &di, + NULL, NULL, + dstBlocks.nbBuffers, + (const void* const *)srcBlocks.buffers, srcBlocks.capacities, + dstBlocks.buffers, dstBlocks.capacities, + NULL); + + assert(BMK_isSuccessful_runOutcome(outcome)); + BMK_runTime_t const result = BMK_extract_runTime(outcome); + U64 const dTime_ns = result.nanoSecPerRun; + double const dTime_sec = (double)dTime_ns / 1000000000; + size_t const srcSize = result.sumOfReturn; + double const dSpeed_MBps = (double)srcSize / dTime_sec / (1 MB); + if (dSpeed_MBps > bestSpeed) bestSpeed = dSpeed_MBps; + DISPLAY("Decompression Speed : %.1f MB/s \r", bestSpeed); + if (BMK_isCompleted_TimedFn(benchState)) break; + } + DISPLAY("\n"); + + freeDecompressInstructions(di); + BMK_freeTimedFnState(benchState); + + return 0; /* success */ +} + /* bench() : * fileName : file to load for benchmarking purpose @@ -272,8 +387,9 @@ int bench(const char* fileName, const char* dictionary) DISPLAYLEVEL(3, "loading %s... \n", fileName); buffer_t const srcBuffer = createBuffer_fromFile(fileName); assert(srcBuffer.ptr != NULL); + size_t const srcSize = srcBuffer.size; DISPLAYLEVEL(3, "created src buffer of size %.1f MB \n", - (double)(srcBuffer.size) / (1 MB)); + (double)srcSize / (1 MB)); buffer_collection_t const srcBlockBuffers = splitBuffer(srcBuffer, BLOCKSIZE); assert(srcBlockBuffers.buffers != NULL); @@ -302,17 +418,20 @@ int bench(const char* fileName, const char* dictionary) ZSTD_CDict* const cdict = ZSTD_createCDict(dictBuffer.ptr, dictBuffer.size, COMP_LEVEL); assert(cdict != NULL); - size_t const cTotalSizeNoDict = compressBlocks(dstBlockBuffers, srcBlockBuffers, NULL, COMP_LEVEL); + size_t const cTotalSizeNoDict = compressBlocks(NULL, dstBlockBuffers, srcBlockBuffers, NULL, COMP_LEVEL); assert(cTotalSizeNoDict != 0); DISPLAYLEVEL(3, "compressing at level %u without dictionary : Ratio=%.2f (%u bytes) \n", COMP_LEVEL, - (double)srcBuffer.size / cTotalSizeNoDict, (unsigned)cTotalSizeNoDict); + (double)srcSize / cTotalSizeNoDict, (unsigned)cTotalSizeNoDict); - size_t const cTotalSize = compressBlocks(dstBlockBuffers, srcBlockBuffers, cdict, COMP_LEVEL); + size_t* const cSizes = malloc(nbBlocks * sizeof(size_t)); + assert(cSizes != NULL); + + size_t const cTotalSize = compressBlocks(cSizes, dstBlockBuffers, srcBlockBuffers, cdict, COMP_LEVEL); assert(cTotalSize != 0); DISPLAYLEVEL(3, "compressed using a %u bytes dictionary : Ratio=%.2f (%u bytes) \n", (unsigned)dictBuffer.size, - (double)srcBuffer.size / cTotalSize, (unsigned)cTotalSize); + (double)srcSize / cTotalSize, (unsigned)cTotalSize); size_t const dictMem = ZSTD_estimateDDictSize(dictBuffer.size, ZSTD_dlm_byCopy); size_t const allDictMem = dictMem * nbBlocks; @@ -322,13 +441,28 @@ int bench(const char* fileName, const char* dictionary) ddict_collection_t const dictionaries = createDDictCollection(dictBuffer.ptr, dictBuffer.size, nbBlocks); assert(dictionaries.ddicts != NULL); + shuffleDictionaries(dictionaries); + // for (size_t u = 0; u < dictionaries.nbDDict; u++) DISPLAY("dict address : %p \n", dictionaries.ddicts[u]); /* check dictionary addresses */ + void* const resultPtr = malloc(srcSize); + assert(resultPtr != NULL); + buffer_t resultBuffer; + resultBuffer.ptr = resultPtr; + resultBuffer.capacity = srcSize; + resultBuffer.size = srcSize; - //result = benchMem(srcBlockBuffers, dstBlockBuffers, dictionaries);; + buffer_collection_t const resultBlockBuffers = splitBuffer(resultBuffer, BLOCKSIZE); + assert(resultBlockBuffers.buffers != NULL); + shrinkSizes(dstBlockBuffers, cSizes); + result = benchMem(resultBlockBuffers, dstBlockBuffers, dictionaries); + /* free all heap objects in reverse order */ + freeCollection(resultBlockBuffers); + free(resultPtr); freeDDictCollection(dictionaries); + free(cSizes); ZSTD_freeCDict(cdict); freeBuffer(dictBuffer); freeCollection(dstBlockBuffers); @@ -342,7 +476,7 @@ int bench(const char* fileName, const char* dictionary) -/*--- Command Line ---*/ +/* --- Command Line --- */ int bad_usage(const char* exeName) { diff --git a/programs/bench.c b/programs/bench.c index b3a8222dd..5ff9afac5 100644 --- a/programs/bench.c +++ b/programs/bench.c @@ -253,7 +253,7 @@ static size_t local_defaultCompress( /* `addArgs` is the context */ static size_t local_defaultDecompress( const void* srcBuffer, size_t srcSize, - void* dstBuffer, size_t dstSize, + void* dstBuffer, size_t dstCapacity, void* addArgs) { size_t moreToFlush = 1; @@ -261,7 +261,7 @@ static size_t local_defaultDecompress( ZSTD_inBuffer in; ZSTD_outBuffer out; in.src = srcBuffer; in.size = srcSize; in.pos = 0; - out.dst = dstBuffer; out.size = dstSize; out.pos = 0; + out.dst = dstBuffer; out.size = dstCapacity; out.pos = 0; while (moreToFlush) { if(out.pos == out.size) { return (size_t)-ZSTD_error_dstSize_tooSmall;