Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
75ed1a815e | ||
|
|
722e14bb65 | ||
|
|
d4e103a04e | ||
|
|
bd10607063 | ||
|
|
c40ba718d7 | ||
|
|
c5fb5b7fcd | ||
|
|
ed3845d3fa | ||
|
|
26f681451f | ||
|
|
aa2628da30 | ||
|
|
19c27d27f1 | ||
|
|
e72efeb0a1 | ||
|
|
974f52fc5d | ||
|
|
e09d38e921 | ||
|
|
f323bf7d32 | ||
|
|
52c04fe58f | ||
|
|
f246cf5423 | ||
|
|
a3d03a3973 | ||
|
|
29652e2618 | ||
|
|
99b045b70a | ||
|
|
bcb5f77efa | ||
|
|
445d49d898 | ||
|
|
a295b3170f | ||
|
|
517e1ba623 | ||
|
|
fe07eaa972 | ||
|
|
e0ce5b094b | ||
|
|
cd25a91741 | ||
|
|
9ca73364e6 | ||
|
|
f9cac7a734 | ||
|
|
23f05ccc6b | ||
|
|
92c986b4e8 | ||
|
|
00d44abe71 | ||
|
|
d916c908e0 | ||
|
|
440bb637e2 | ||
|
|
041f1ad9a7 | ||
|
|
06ad6f1911 | ||
|
|
fb5c59fc89 | ||
|
|
2b61f74b1e | ||
|
|
5cc1882d45 | ||
|
|
698cb63305 | ||
|
|
d57dffbe76 | ||
|
|
c62cda9def | ||
|
|
302ff036f6 | ||
|
|
9634f67107 | ||
|
|
92d75667e4 | ||
|
|
b2b5309c6f | ||
|
|
5e734ad09b | ||
|
|
f4b7b747c7 | ||
|
|
d96e8f20b9 | ||
|
|
0d5bf8f06f | ||
|
|
ed7fb8413c | ||
|
|
b71adf45c1 | ||
|
|
b7f197633a | ||
|
|
a7a5d7736c | ||
|
|
cebab1d322 | ||
|
|
2fa9904844 | ||
|
|
406dcca98a | ||
|
|
c35cd5b1db | ||
|
|
c093208ab8 | ||
|
|
6c6e1751f6 | ||
|
|
3d2cd7f816 | ||
|
|
ec224d256d | ||
|
|
4c56f4a3cf | ||
|
|
529d9c7dee | ||
|
|
d4f4e58ee1 | ||
|
|
63b5e7a2ea |
@@ -168,6 +168,9 @@ bmix32test: clean
|
|||||||
|
|
||||||
bmi32test: clean
|
bmi32test: clean
|
||||||
CFLAGS="-O3 -mbmi -m32 -Werror" $(MAKE) -C $(PRGDIR) test
|
CFLAGS="-O3 -mbmi -m32 -Werror" $(MAKE) -C $(PRGDIR) test
|
||||||
|
|
||||||
|
staticAnalyze: clean
|
||||||
|
CPPFLAGS=-g scan-build --status-bugs -v $(MAKE) all
|
||||||
endif
|
endif
|
||||||
|
|
||||||
|
|
||||||
@@ -187,7 +190,7 @@ gcc5install:
|
|||||||
|
|
||||||
gcc6install:
|
gcc6install:
|
||||||
sudo add-apt-repository -y ppa:ubuntu-toolchain-r/test
|
sudo add-apt-repository -y ppa:ubuntu-toolchain-r/test
|
||||||
sudo apt-get update -y -qq
|
sudo apt-get update -y -qq
|
||||||
sudo apt-get install -y -qq gcc-6-multilib
|
sudo apt-get install -y -qq gcc-6-multilib
|
||||||
|
|
||||||
arminstall: clean
|
arminstall: clean
|
||||||
|
|||||||
@@ -1,3 +1,20 @@
|
|||||||
|
v0.7.3
|
||||||
|
New : compression format specification
|
||||||
|
New : `--` separator, stating that all following arguments are file names. Suggested by Chip Turner.
|
||||||
|
New : `ZSTD_getDecompressedSize()`
|
||||||
|
New : OpenBSD target, by Juan Francisco Cantero Hurtado
|
||||||
|
New : `examples` directory
|
||||||
|
fixed : dictBuilder using HC levels, reported by Bartosz Taudul
|
||||||
|
fixed : legacy support from ZSTD_decompress_usingDDict(), reported by Felix Handte
|
||||||
|
fixed : multi-blocks decoding with intermediate uncompressed blocks, reported by Greg Slazinski
|
||||||
|
modified : removed "mem.h" and "error_public.h" dependencies from "zstd.h" (experimental section)
|
||||||
|
modified : legacy functions no longer need magic number
|
||||||
|
|
||||||
|
v0.7.2
|
||||||
|
fixed : ZSTD_decompressBlock() using multiple consecutive blocks. Reported by Greg Slazinski.
|
||||||
|
fixed : potential segfault on very large files (many gigabytes). Reported by Chip Turner.
|
||||||
|
fixed : CLI displays system error message when destination file cannot be created (#231). Reported by Chip Turner.
|
||||||
|
|
||||||
v0.7.1
|
v0.7.1
|
||||||
fixed : ZBUFF_compressEnd() called multiple times with too small `dst` buffer, reported by Christophe Chevalier
|
fixed : ZBUFF_compressEnd() called multiple times with too small `dst` buffer, reported by Christophe Chevalier
|
||||||
fixed : dictBuilder fails if first sample is too small, reported by Руслан Ковалёв
|
fixed : dictBuilder fails if first sample is too small, reported by Руслан Ковалёв
|
||||||
|
|||||||
@@ -0,0 +1,7 @@
|
|||||||
|
Zstandard library : usage examples
|
||||||
|
==================================
|
||||||
|
|
||||||
|
- [Dictionary decompression](dictionary_decompression.c)
|
||||||
|
Decompress multiple files using the same dictionary.
|
||||||
|
Compatible with Legacy modes.
|
||||||
|
Introduces usage of : `ZSTD_createDDict()` and `ZSTD_decompress_usingDDict()`
|
||||||
@@ -0,0 +1,112 @@
|
|||||||
|
#include <stdlib.h> // exit
|
||||||
|
#include <stdio.h> // printf
|
||||||
|
#include <string.h> // strerror
|
||||||
|
#include <errno.h> // errno
|
||||||
|
#include <sys/stat.h> // stat
|
||||||
|
#include <zstd.h>
|
||||||
|
|
||||||
|
|
||||||
|
static off_t fsizeX(const char *filename)
|
||||||
|
{
|
||||||
|
struct stat st;
|
||||||
|
if (stat(filename, &st) == 0) return st.st_size;
|
||||||
|
/* error */
|
||||||
|
printf("stat: %s : %s \n", filename, strerror(errno));
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
|
|
||||||
|
static FILE* fopenX(const char *filename, const char *instruction)
|
||||||
|
{
|
||||||
|
FILE* const inFile = fopen(filename, instruction);
|
||||||
|
if (inFile) return inFile;
|
||||||
|
/* error */
|
||||||
|
printf("fopen: %s : %s \n", filename, strerror(errno));
|
||||||
|
exit(2);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void* mallocX(size_t size)
|
||||||
|
{
|
||||||
|
void* const buff = malloc(size);
|
||||||
|
if (buff) return buff;
|
||||||
|
/* error */
|
||||||
|
printf("malloc: %s \n", strerror(errno));
|
||||||
|
exit(3);
|
||||||
|
}
|
||||||
|
|
||||||
|
static void* loadFileX(const char* fileName, size_t* size)
|
||||||
|
{
|
||||||
|
off_t const buffSize = fsizeX(fileName);
|
||||||
|
FILE* const inFile = fopenX(fileName, "rb");
|
||||||
|
void* const buffer = mallocX(buffSize);
|
||||||
|
size_t const readSize = fread(buffer, 1, buffSize, inFile);
|
||||||
|
if (readSize != (size_t)buffSize) {
|
||||||
|
printf("fread: %s : %s \n", fileName, strerror(errno));
|
||||||
|
exit(4);
|
||||||
|
}
|
||||||
|
fclose(inFile);
|
||||||
|
*size = buffSize;
|
||||||
|
return buffer;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
static const ZSTD_DDict* createDict(const char* dictFileName)
|
||||||
|
{
|
||||||
|
size_t dictSize;
|
||||||
|
void* const dictBuffer = loadFileX(dictFileName, &dictSize);
|
||||||
|
const ZSTD_DDict* const ddict = ZSTD_createDDict(dictBuffer, dictSize);
|
||||||
|
free(dictBuffer);
|
||||||
|
return ddict;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
/* prototype declared here, as it currently is part of experimental section */
|
||||||
|
unsigned long long ZSTD_getDecompressedSize(const void* src, size_t srcSize);
|
||||||
|
|
||||||
|
static void decompress(const char* fname, const ZSTD_DDict* ddict)
|
||||||
|
{
|
||||||
|
size_t cSize;
|
||||||
|
void* const cBuff = loadFileX(fname, &cSize);
|
||||||
|
unsigned long long const rSize = ZSTD_getDecompressedSize(cBuff, cSize);
|
||||||
|
if (rSize==0) {
|
||||||
|
printf("%s : original size unknown \n", fname);
|
||||||
|
exit(5);
|
||||||
|
}
|
||||||
|
void* const rBuff = mallocX(rSize);
|
||||||
|
|
||||||
|
ZSTD_DCtx* const dctx = ZSTD_createDCtx();
|
||||||
|
size_t const dSize = ZSTD_decompress_usingDDict(dctx, rBuff, rSize, cBuff, cSize, ddict);
|
||||||
|
|
||||||
|
if (dSize != rSize) {
|
||||||
|
printf("error decoding %s : %s \n", fname, ZSTD_getErrorName(dSize));
|
||||||
|
exit(7);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* success */
|
||||||
|
printf("%25s : %6u -> %7u \n", fname, (unsigned)cSize, (unsigned)rSize);
|
||||||
|
|
||||||
|
ZSTD_freeDCtx(dctx);
|
||||||
|
free(rBuff);
|
||||||
|
free(cBuff);
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
int main(int argc, const char** argv)
|
||||||
|
{
|
||||||
|
const char* const exeName = argv[0];
|
||||||
|
|
||||||
|
if (argc<3) {
|
||||||
|
printf("wrong arguments\n");
|
||||||
|
printf("usage:\n");
|
||||||
|
printf("%s [FILES] dictionary\n", exeName);
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* load dictionary only once */
|
||||||
|
const char* const dictName = argv[argc-1];
|
||||||
|
const ZSTD_DDict* const dictPtr = createDict(dictName);
|
||||||
|
|
||||||
|
int u;
|
||||||
|
for (u=1; u<argc-1; u++) decompress(argv[u], dictPtr);
|
||||||
|
|
||||||
|
printf("All %u files decoded. \n", argc-2);
|
||||||
|
}
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
# make install artefact
|
||||||
|
libzstd.pc
|
||||||
+10
-5
@@ -45,14 +45,19 @@ It is used by `zstd` command line utility, and [7zip plugin](http://mcmilk.de/pr
|
|||||||
- compress/zbuff_compress.c
|
- compress/zbuff_compress.c
|
||||||
- decompress/zbuff_decompress.c
|
- decompress/zbuff_decompress.c
|
||||||
|
|
||||||
|
|
||||||
#### Dictionary builder
|
#### Dictionary builder
|
||||||
|
|
||||||
To create dictionaries from training sets :
|
In order to create dictionaries from some training sets,
|
||||||
|
it's needed to include all files from [dictBuilder directory](dictBuilder/)
|
||||||
|
|
||||||
|
|
||||||
|
#### Legacy support
|
||||||
|
|
||||||
|
Zstandard can decode previous formats, starting from v0.1.
|
||||||
|
Support for these format is provided in [folder legacy](legacy/).
|
||||||
|
It's also required to compile the library with `ZSTD_LEGACY_SUPPORT = 1`.
|
||||||
|
|
||||||
- dictBuilder/divsufsort.c
|
|
||||||
- dictBuilder/divsufsort.h
|
|
||||||
- dictBuilder/zdict.c
|
|
||||||
- dictBuilder/zdict.h
|
|
||||||
|
|
||||||
#### Miscellaneous
|
#### Miscellaneous
|
||||||
|
|
||||||
|
|||||||
@@ -63,7 +63,11 @@ typedef enum {
|
|||||||
ZSTD_error_maxCode
|
ZSTD_error_maxCode
|
||||||
} ZSTD_ErrorCode;
|
} ZSTD_ErrorCode;
|
||||||
|
|
||||||
/* note : compare with size_t function results using ZSTD_getError() */
|
/*! ZSTD_getErrorCode() :
|
||||||
|
convert a `size_t` function result into a `ZSTD_ErrorCode` enum type,
|
||||||
|
which can be used to compare directly with enum list published into "error_public.h" */
|
||||||
|
ZSTD_ErrorCode ZSTD_getErrorCode(size_t functionResult);
|
||||||
|
const char* ZSTD_getErrorString(ZSTD_ErrorCode code);
|
||||||
|
|
||||||
|
|
||||||
#if defined (__cplusplus)
|
#if defined (__cplusplus)
|
||||||
|
|||||||
+5
-7
@@ -44,10 +44,8 @@ extern "C" {
|
|||||||
/* ***************************************************************
|
/* ***************************************************************
|
||||||
* Compiler specifics
|
* Compiler specifics
|
||||||
*****************************************************************/
|
*****************************************************************/
|
||||||
/*!
|
/* ZSTD_DLL_EXPORT :
|
||||||
* ZSTD_DLL_EXPORT :
|
* Enable exporting of functions when building a Windows DLL */
|
||||||
* Enable exporting of functions when building a Windows DLL
|
|
||||||
*/
|
|
||||||
#if defined(_WIN32) && defined(ZSTD_DLL_EXPORT) && (ZSTD_DLL_EXPORT==1)
|
#if defined(_WIN32) && defined(ZSTD_DLL_EXPORT) && (ZSTD_DLL_EXPORT==1)
|
||||||
# define ZSTDLIB_API __declspec(dllexport)
|
# define ZSTDLIB_API __declspec(dllexport)
|
||||||
#else
|
#else
|
||||||
@@ -103,8 +101,8 @@ ZSTDLIB_API size_t ZBUFF_compressEnd(ZBUFF_CCtx* cctx, void* dst, size_t* dstCap
|
|||||||
* @return : nb of bytes still present into internal buffer (0 if it's empty)
|
* @return : nb of bytes still present into internal buffer (0 if it's empty)
|
||||||
* or an error code, which can be tested using ZBUFF_isError().
|
* or an error code, which can be tested using ZBUFF_isError().
|
||||||
*
|
*
|
||||||
* Hint : recommended buffer sizes (not compulsory) : ZBUFF_recommendedCInSize / ZBUFF_recommendedCOutSize
|
* Hint : _recommended buffer_ sizes (not compulsory) : ZBUFF_recommendedCInSize() / ZBUFF_recommendedCOutSize()
|
||||||
* input : ZBUFF_recommendedCInSize==128 KB block size is the internal unit, it improves latency to use this value (skipped buffering).
|
* input : ZBUFF_recommendedCInSize==128 KB block size is the internal unit, use this value to reduce intermediate stages (better latency)
|
||||||
* output : ZBUFF_recommendedCOutSize==ZSTD_compressBound(128 KB) + 3 + 3 : ensures it's always possible to write/flush/end a full block. Skip some buffering.
|
* output : ZBUFF_recommendedCOutSize==ZSTD_compressBound(128 KB) + 3 + 3 : ensures it's always possible to write/flush/end a full block. Skip some buffering.
|
||||||
* By using both, it ensures that input will be entirely consumed, and output will always contain the result, reducing intermediate buffering.
|
* By using both, it ensures that input will be entirely consumed, and output will always contain the result, reducing intermediate buffering.
|
||||||
* **************************************************/
|
* **************************************************/
|
||||||
@@ -187,7 +185,7 @@ ZSTDLIB_API ZBUFF_DCtx* ZBUFF_createDCtx_advanced(ZSTD_customMem customMem);
|
|||||||
/*--- Advanced Streaming function ---*/
|
/*--- Advanced Streaming function ---*/
|
||||||
ZSTDLIB_API size_t ZBUFF_compressInit_advanced(ZBUFF_CCtx* zbc,
|
ZSTDLIB_API size_t ZBUFF_compressInit_advanced(ZBUFF_CCtx* zbc,
|
||||||
const void* dict, size_t dictSize,
|
const void* dict, size_t dictSize,
|
||||||
ZSTD_parameters params, U64 pledgedSrcSize);
|
ZSTD_parameters params, unsigned long long pledgedSrcSize);
|
||||||
|
|
||||||
#endif /* ZBUFF_STATIC_LINKING_ONLY */
|
#endif /* ZBUFF_STATIC_LINKING_ONLY */
|
||||||
|
|
||||||
|
|||||||
+45
-37
@@ -61,7 +61,7 @@ extern "C" {
|
|||||||
***************************************/
|
***************************************/
|
||||||
#define ZSTD_VERSION_MAJOR 0
|
#define ZSTD_VERSION_MAJOR 0
|
||||||
#define ZSTD_VERSION_MINOR 7
|
#define ZSTD_VERSION_MINOR 7
|
||||||
#define ZSTD_VERSION_RELEASE 1
|
#define ZSTD_VERSION_RELEASE 3
|
||||||
|
|
||||||
#define ZSTD_LIB_VERSION ZSTD_VERSION_MAJOR.ZSTD_VERSION_MINOR.ZSTD_VERSION_RELEASE
|
#define ZSTD_LIB_VERSION ZSTD_VERSION_MAJOR.ZSTD_VERSION_MINOR.ZSTD_VERSION_RELEASE
|
||||||
#define ZSTD_QUOTE(str) #str
|
#define ZSTD_QUOTE(str) #str
|
||||||
@@ -197,15 +197,13 @@ ZSTDLIB_API size_t ZSTD_decompress_usingDDict(ZSTD_DCtx* dctx,
|
|||||||
* Use them only in association with static linking.
|
* Use them only in association with static linking.
|
||||||
* ==================================================================================== */
|
* ==================================================================================== */
|
||||||
|
|
||||||
/*--- Dependency ---*/
|
|
||||||
#include "mem.h" /* U32 */
|
|
||||||
|
|
||||||
|
|
||||||
/*--- Constants ---*/
|
/*--- Constants ---*/
|
||||||
#define ZSTD_MAGICNUMBER 0xFD2FB527 /* v0.7 */
|
#define ZSTD_MAGICNUMBER 0xFD2FB527 /* v0.7 */
|
||||||
#define ZSTD_MAGIC_SKIPPABLE_START 0x184D2A50U
|
#define ZSTD_MAGIC_SKIPPABLE_START 0x184D2A50U
|
||||||
|
|
||||||
#define ZSTD_WINDOWLOG_MAX ((U32)(MEM_32bits() ? 25 : 27))
|
#define ZSTD_WINDOWLOG_MAX_32 25
|
||||||
|
#define ZSTD_WINDOWLOG_MAX_64 27
|
||||||
|
#define ZSTD_WINDOWLOG_MAX ((U32)(MEM_32bits() ? ZSTD_WINDOWLOG_MAX_32 : ZSTD_WINDOWLOG_MAX_64))
|
||||||
#define ZSTD_WINDOWLOG_MIN 18
|
#define ZSTD_WINDOWLOG_MIN 18
|
||||||
#define ZSTD_CHAINLOG_MAX (ZSTD_WINDOWLOG_MAX+1)
|
#define ZSTD_CHAINLOG_MAX (ZSTD_WINDOWLOG_MAX+1)
|
||||||
#define ZSTD_CHAINLOG_MIN 4
|
#define ZSTD_CHAINLOG_MIN 4
|
||||||
@@ -230,19 +228,19 @@ static const size_t ZSTD_skippableHeaderSize = 8; /* magic number + skippable f
|
|||||||
typedef enum { ZSTD_fast, ZSTD_greedy, ZSTD_lazy, ZSTD_lazy2, ZSTD_btlazy2, ZSTD_btopt } ZSTD_strategy; /*< from faster to stronger */
|
typedef enum { ZSTD_fast, ZSTD_greedy, ZSTD_lazy, ZSTD_lazy2, ZSTD_btlazy2, ZSTD_btopt } ZSTD_strategy; /*< from faster to stronger */
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
U32 windowLog; /*< largest match distance : larger == more compression, more memory needed during decompression */
|
unsigned windowLog; /*< largest match distance : larger == more compression, more memory needed during decompression */
|
||||||
U32 chainLog; /*< fully searched segment : larger == more compression, slower, more memory (useless for fast) */
|
unsigned chainLog; /*< fully searched segment : larger == more compression, slower, more memory (useless for fast) */
|
||||||
U32 hashLog; /*< dispatch table : larger == faster, more memory */
|
unsigned hashLog; /*< dispatch table : larger == faster, more memory */
|
||||||
U32 searchLog; /*< nb of searches : larger == more compression, slower */
|
unsigned searchLog; /*< nb of searches : larger == more compression, slower */
|
||||||
U32 searchLength; /*< match length searched : larger == faster decompression, sometimes less compression */
|
unsigned searchLength; /*< match length searched : larger == faster decompression, sometimes less compression */
|
||||||
U32 targetLength; /*< acceptable match size for optimal parser (only) : larger == more compression, slower */
|
unsigned targetLength; /*< acceptable match size for optimal parser (only) : larger == more compression, slower */
|
||||||
ZSTD_strategy strategy;
|
ZSTD_strategy strategy;
|
||||||
} ZSTD_compressionParameters;
|
} ZSTD_compressionParameters;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
U32 contentSizeFlag; /*< 1: content size will be in frame header (if known). */
|
unsigned contentSizeFlag; /*< 1: content size will be in frame header (if known). */
|
||||||
U32 checksumFlag; /*< 1: will generate a 22-bits checksum at end of frame, to be used for error detection by decompressor */
|
unsigned checksumFlag; /*< 1: will generate a 22-bits checksum at end of frame, to be used for error detection by decompressor */
|
||||||
U32 noDictIDFlag; /*< 1: no dict ID will be saved into frame header (if dictionary compression) */
|
unsigned noDictIDFlag; /*< 1: no dict ID will be saved into frame header (if dictionary compression) */
|
||||||
} ZSTD_frameParameters;
|
} ZSTD_frameParameters;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
@@ -270,19 +268,24 @@ ZSTDLIB_API ZSTD_CDict* ZSTD_createCDict_advanced(const void* dict, size_t dictS
|
|||||||
|
|
||||||
ZSTDLIB_API unsigned ZSTD_maxCLevel (void);
|
ZSTDLIB_API unsigned ZSTD_maxCLevel (void);
|
||||||
|
|
||||||
|
/*! ZSTD_getParams() :
|
||||||
|
* same as ZSTD_getCParams(), but @return a full `ZSTD_parameters` object instead of a `ZSTD_compressionParameters`.
|
||||||
|
* All fields of `ZSTD_frameParameters` are set to default (0) */
|
||||||
|
ZSTD_parameters ZSTD_getParams(int compressionLevel, unsigned long long srcSize, size_t dictSize);
|
||||||
|
|
||||||
/*! ZSTD_getCParams() :
|
/*! ZSTD_getCParams() :
|
||||||
* @return ZSTD_compressionParameters structure for a selected compression level and srcSize.
|
* @return ZSTD_compressionParameters structure for a selected compression level and srcSize.
|
||||||
* `srcSize` value is optional, select 0 if not known */
|
* `srcSize` value is optional, select 0 if not known */
|
||||||
ZSTDLIB_API ZSTD_compressionParameters ZSTD_getCParams(int compressionLevel, U64 srcSize, size_t dictSize);
|
ZSTDLIB_API ZSTD_compressionParameters ZSTD_getCParams(int compressionLevel, unsigned long long srcSize, size_t dictSize);
|
||||||
|
|
||||||
/*! ZSTD_checkParams() :
|
/*! ZSTD_checkCParams() :
|
||||||
* Ensure param values remain within authorized range */
|
* Ensure param values remain within authorized range */
|
||||||
ZSTDLIB_API size_t ZSTD_checkCParams(ZSTD_compressionParameters params);
|
ZSTDLIB_API size_t ZSTD_checkCParams(ZSTD_compressionParameters params);
|
||||||
|
|
||||||
/*! ZSTD_adjustParams() :
|
/*! ZSTD_adjustCParams() :
|
||||||
* optimize params for a given `srcSize` and `dictSize`.
|
* optimize params for a given `srcSize` and `dictSize`.
|
||||||
* both values are optional, select `0` if unknown. */
|
* both values are optional, select `0` if unknown. */
|
||||||
ZSTDLIB_API ZSTD_compressionParameters ZSTD_adjustCParams(ZSTD_compressionParameters cPar, U64 srcSize, size_t dictSize);
|
ZSTDLIB_API ZSTD_compressionParameters ZSTD_adjustCParams(ZSTD_compressionParameters cPar, unsigned long long srcSize, size_t dictSize);
|
||||||
|
|
||||||
/*! ZSTD_compress_advanced() :
|
/*! ZSTD_compress_advanced() :
|
||||||
* Same as ZSTD_compress_usingDict(), with fine-tune control of each compression parameter */
|
* Same as ZSTD_compress_usingDict(), with fine-tune control of each compression parameter */
|
||||||
@@ -295,6 +298,15 @@ ZSTDLIB_API size_t ZSTD_compress_advanced (ZSTD_CCtx* ctx,
|
|||||||
|
|
||||||
/*--- Advanced Decompression functions ---*/
|
/*--- Advanced Decompression functions ---*/
|
||||||
|
|
||||||
|
/** ZSTD_getDecompressedSize() :
|
||||||
|
* compatible with legacy mode
|
||||||
|
* @return : decompressed size if known, 0 otherwise
|
||||||
|
note : 0 can mean any of the following :
|
||||||
|
- decompressed size is not provided within frame header
|
||||||
|
- frame header unknown / not supported
|
||||||
|
- frame header not completely provided (`srcSize` too small) */
|
||||||
|
unsigned long long ZSTD_getDecompressedSize(const void* src, size_t srcSize);
|
||||||
|
|
||||||
/*! ZSTD_createDCtx_advanced() :
|
/*! ZSTD_createDCtx_advanced() :
|
||||||
* Create a ZSTD decompression context using external alloc and free functions */
|
* Create a ZSTD decompression context using external alloc and free functions */
|
||||||
ZSTDLIB_API ZSTD_DCtx* ZSTD_createDCtx_advanced(ZSTD_customMem customMem);
|
ZSTDLIB_API ZSTD_DCtx* ZSTD_createDCtx_advanced(ZSTD_customMem customMem);
|
||||||
@@ -305,7 +317,7 @@ ZSTDLIB_API ZSTD_DCtx* ZSTD_createDCtx_advanced(ZSTD_customMem customMem);
|
|||||||
******************************************************************/
|
******************************************************************/
|
||||||
ZSTDLIB_API size_t ZSTD_compressBegin(ZSTD_CCtx* cctx, int compressionLevel);
|
ZSTDLIB_API size_t ZSTD_compressBegin(ZSTD_CCtx* cctx, int compressionLevel);
|
||||||
ZSTDLIB_API size_t ZSTD_compressBegin_usingDict(ZSTD_CCtx* cctx, const void* dict, size_t dictSize, int compressionLevel);
|
ZSTDLIB_API size_t ZSTD_compressBegin_usingDict(ZSTD_CCtx* cctx, const void* dict, size_t dictSize, int compressionLevel);
|
||||||
ZSTDLIB_API size_t ZSTD_compressBegin_advanced(ZSTD_CCtx* cctx, const void* dict, size_t dictSize, ZSTD_parameters params, U64 pledgedSrcSize);
|
ZSTDLIB_API size_t ZSTD_compressBegin_advanced(ZSTD_CCtx* cctx, const void* dict, size_t dictSize, ZSTD_parameters params, unsigned long long pledgedSrcSize);
|
||||||
ZSTDLIB_API size_t ZSTD_copyCCtx(ZSTD_CCtx* cctx, const ZSTD_CCtx* preparedCCtx);
|
ZSTDLIB_API size_t ZSTD_copyCCtx(ZSTD_CCtx* cctx, const ZSTD_CCtx* preparedCCtx);
|
||||||
|
|
||||||
ZSTDLIB_API size_t ZSTD_compressContinue(ZSTD_CCtx* cctx, void* dst, size_t dstCapacity, const void* src, size_t srcSize);
|
ZSTDLIB_API size_t ZSTD_compressContinue(ZSTD_CCtx* cctx, void* dst, size_t dstCapacity, const void* src, size_t srcSize);
|
||||||
@@ -341,10 +353,10 @@ ZSTDLIB_API size_t ZSTD_compressEnd(ZSTD_CCtx* cctx, void* dst, size_t dstCapaci
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
U64 frameContentSize;
|
unsigned long long frameContentSize;
|
||||||
U32 windowSize;
|
unsigned windowSize;
|
||||||
U32 dictID;
|
unsigned dictID;
|
||||||
U32 checksumFlag;
|
unsigned checksumFlag;
|
||||||
} ZSTD_frameParams;
|
} ZSTD_frameParams;
|
||||||
|
|
||||||
ZSTDLIB_API size_t ZSTD_getFrameParams(ZSTD_frameParams* fparamsPtr, const void* src, size_t srcSize); /**< doesn't consume input */
|
ZSTDLIB_API size_t ZSTD_getFrameParams(ZSTD_frameParams* fparamsPtr, const void* src, size_t srcSize); /**< doesn't consume input */
|
||||||
@@ -404,11 +416,14 @@ ZSTDLIB_API size_t ZSTD_decompressContinue(ZSTD_DCtx* dctx, void* dst, size_t ds
|
|||||||
* Block functions
|
* Block functions
|
||||||
****************************************/
|
****************************************/
|
||||||
/*! Block functions produce and decode raw zstd blocks, without frame metadata.
|
/*! Block functions produce and decode raw zstd blocks, without frame metadata.
|
||||||
|
Frame metadata cost is typically ~18 bytes, which is non-negligible on very small blocks.
|
||||||
User will have to take in charge required information to regenerate data, such as compressed and content sizes.
|
User will have to take in charge required information to regenerate data, such as compressed and content sizes.
|
||||||
|
|
||||||
A few rules to respect :
|
A few rules to respect :
|
||||||
- Uncompressed block size must be <= ZSTD_BLOCKSIZE_MAX (128 KB)
|
- Uncompressed block size must be <= ZSTD_BLOCKSIZE_MAX (128 KB)
|
||||||
- Compressing or decompressing requires a context structure
|
+ If you need to compress more, cut data into multiple blocks
|
||||||
|
+ Consider using the regular ZSTD_compress() instead, as frame metadata costs become negligible when source size is large.
|
||||||
|
- Compressing and decompressing require a context structure
|
||||||
+ Use ZSTD_createCCtx() and ZSTD_createDCtx()
|
+ Use ZSTD_createCCtx() and ZSTD_createDCtx()
|
||||||
- It is necessary to init context before starting
|
- It is necessary to init context before starting
|
||||||
+ compression : ZSTD_compressBegin()
|
+ compression : ZSTD_compressBegin()
|
||||||
@@ -418,23 +433,16 @@ ZSTDLIB_API size_t ZSTD_decompressContinue(ZSTD_DCtx* dctx, void* dst, size_t ds
|
|||||||
- When a block is considered not compressible enough, ZSTD_compressBlock() result will be zero.
|
- When a block is considered not compressible enough, ZSTD_compressBlock() result will be zero.
|
||||||
In which case, nothing is produced into `dst`.
|
In which case, nothing is produced into `dst`.
|
||||||
+ User must test for such outcome and deal directly with uncompressed data
|
+ User must test for such outcome and deal directly with uncompressed data
|
||||||
+ ZSTD_decompressBlock() doesn't accept uncompressed data as input !!
|
+ ZSTD_decompressBlock() doesn't accept uncompressed data as input !!!
|
||||||
|
+ In case of multiple successive blocks, decoder must be informed of uncompressed block existence to follow proper history.
|
||||||
|
Use ZSTD_insertBlock() in such a case.
|
||||||
|
Insert block once it's copied into its final position.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#define ZSTD_BLOCKSIZE_MAX (128 * 1024) /* define, for static allocation */
|
#define ZSTD_BLOCKSIZE_MAX (128 * 1024) /* define, for static allocation */
|
||||||
ZSTDLIB_API size_t ZSTD_compressBlock (ZSTD_CCtx* cctx, void* dst, size_t dstCapacity, const void* src, size_t srcSize);
|
ZSTDLIB_API size_t ZSTD_compressBlock (ZSTD_CCtx* cctx, void* dst, size_t dstCapacity, const void* src, size_t srcSize);
|
||||||
ZSTDLIB_API size_t ZSTD_decompressBlock(ZSTD_DCtx* dctx, void* dst, size_t dstCapacity, const void* src, size_t srcSize);
|
ZSTDLIB_API size_t ZSTD_decompressBlock(ZSTD_DCtx* dctx, void* dst, size_t dstCapacity, const void* src, size_t srcSize);
|
||||||
|
ZSTDLIB_API size_t ZSTD_insertBlock(ZSTD_DCtx* dctx, const void* blockStart, size_t blockSize); /**< insert block into `dctx` history. Useful to track uncompressed blocks */
|
||||||
|
|
||||||
/*-*************************************
|
|
||||||
* Error management
|
|
||||||
***************************************/
|
|
||||||
#include "error_public.h"
|
|
||||||
/*! ZSTD_getErrorCode() :
|
|
||||||
convert a `size_t` function result into a `ZSTD_ErrorCode` enum type,
|
|
||||||
which can be used to compare directly with enum list published into "error_public.h" */
|
|
||||||
ZSTDLIB_API ZSTD_ErrorCode ZSTD_getErrorCode(size_t functionResult);
|
|
||||||
ZSTDLIB_API const char* ZSTD_getErrorString(ZSTD_ErrorCode code);
|
|
||||||
|
|
||||||
|
|
||||||
#endif /* ZSTD_STATIC_LINKING_ONLY */
|
#endif /* ZSTD_STATIC_LINKING_ONLY */
|
||||||
|
|||||||
@@ -51,7 +51,7 @@
|
|||||||
/*-*************************************
|
/*-*************************************
|
||||||
* Common constants
|
* Common constants
|
||||||
***************************************/
|
***************************************/
|
||||||
#define ZSTD_OPT_DEBUG 0 // 3 = compression stats; 5 = check encoded sequences; 9 = full logs
|
#define ZSTD_OPT_DEBUG 0 /* 3 = compression stats; 5 = check encoded sequences; 9 = full logs */
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#if defined(ZSTD_OPT_DEBUG) && ZSTD_OPT_DEBUG>=9
|
#if defined(ZSTD_OPT_DEBUG) && ZSTD_OPT_DEBUG>=9
|
||||||
#define ZSTD_LOG_PARSER(...) printf(__VA_ARGS__)
|
#define ZSTD_LOG_PARSER(...) printf(__VA_ARGS__)
|
||||||
@@ -233,6 +233,6 @@ int ZSTD_isSkipFrame(ZSTD_DCtx* dctx);
|
|||||||
/* custom memory allocation functions */
|
/* custom memory allocation functions */
|
||||||
void* ZSTD_defaultAllocFunction(void* opaque, size_t size);
|
void* ZSTD_defaultAllocFunction(void* opaque, size_t size);
|
||||||
void ZSTD_defaultFreeFunction(void* opaque, void* address);
|
void ZSTD_defaultFreeFunction(void* opaque, void* address);
|
||||||
static ZSTD_customMem const defaultCustomMem = { ZSTD_defaultAllocFunction, ZSTD_defaultFreeFunction, NULL };
|
static const ZSTD_customMem defaultCustomMem = { ZSTD_defaultAllocFunction, ZSTD_defaultFreeFunction, NULL };
|
||||||
|
|
||||||
#endif /* ZSTD_CCOMMON_H_MODULE */
|
#endif /* ZSTD_CCOMMON_H_MODULE */
|
||||||
|
|||||||
@@ -137,7 +137,7 @@ size_t ZBUFF_freeCCtx(ZBUFF_CCtx* zbc)
|
|||||||
|
|
||||||
size_t ZBUFF_compressInit_advanced(ZBUFF_CCtx* zbc,
|
size_t ZBUFF_compressInit_advanced(ZBUFF_CCtx* zbc,
|
||||||
const void* dict, size_t dictSize,
|
const void* dict, size_t dictSize,
|
||||||
ZSTD_parameters params, U64 pledgedSrcSize)
|
ZSTD_parameters params, unsigned long long pledgedSrcSize)
|
||||||
{
|
{
|
||||||
/* allocate buffers */
|
/* allocate buffers */
|
||||||
{ size_t const neededInBuffSize = (size_t)1 << params.cParams.windowLog;
|
{ size_t const neededInBuffSize = (size_t)1 << params.cParams.windowLog;
|
||||||
@@ -170,9 +170,7 @@ size_t ZBUFF_compressInit_advanced(ZBUFF_CCtx* zbc,
|
|||||||
|
|
||||||
size_t ZBUFF_compressInitDictionary(ZBUFF_CCtx* zbc, const void* dict, size_t dictSize, int compressionLevel)
|
size_t ZBUFF_compressInitDictionary(ZBUFF_CCtx* zbc, const void* dict, size_t dictSize, int compressionLevel)
|
||||||
{
|
{
|
||||||
ZSTD_parameters params;
|
ZSTD_parameters const params = ZSTD_getParams(compressionLevel, 0, dictSize);
|
||||||
memset(¶ms, 0, sizeof(params));
|
|
||||||
params.cParams = ZSTD_getCParams(compressionLevel, 0, dictSize);
|
|
||||||
return ZBUFF_compressInit_advanced(zbc, dict, dictSize, params, 0);
|
return ZBUFF_compressInit_advanced(zbc, dict, dictSize, params, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -221,7 +221,7 @@ size_t ZSTD_checkCParams_advanced(ZSTD_compressionParameters cParams, U64 srcSiz
|
|||||||
Both `srcSize` and `dictSize` are optional (use 0 if unknown),
|
Both `srcSize` and `dictSize` are optional (use 0 if unknown),
|
||||||
but if both are 0, no optimization can be done.
|
but if both are 0, no optimization can be done.
|
||||||
Note : cPar is considered validated at this stage. Use ZSTD_checkParams() to ensure that. */
|
Note : cPar is considered validated at this stage. Use ZSTD_checkParams() to ensure that. */
|
||||||
ZSTD_compressionParameters ZSTD_adjustCParams(ZSTD_compressionParameters cPar, U64 srcSize, size_t dictSize)
|
ZSTD_compressionParameters ZSTD_adjustCParams(ZSTD_compressionParameters cPar, unsigned long long srcSize, size_t dictSize)
|
||||||
{
|
{
|
||||||
if (srcSize+dictSize == 0) return cPar; /* no size information available : no adjustment */
|
if (srcSize+dictSize == 0) return cPar; /* no size information available : no adjustment */
|
||||||
|
|
||||||
@@ -427,21 +427,8 @@ static void ZSTD_reduceIndex (ZSTD_CCtx* zc, const U32 reducerValue)
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
|
|
||||||
/* Frame descriptor
|
/* Frame header :
|
||||||
|
|
||||||
// old
|
|
||||||
1 byte - Alloc :
|
|
||||||
bit 0-3 : windowLog - ZSTD_WINDOWLOG_ABSOLUTEMIN (see zstd_internal.h)
|
|
||||||
bit 4 : reserved for windowLog (must be zero)
|
|
||||||
bit 5 : reserved (must be zero)
|
|
||||||
bit 6-7 : Frame content size : unknown, 1 byte, 2 bytes, 8 bytes
|
|
||||||
|
|
||||||
1 byte - checker :
|
|
||||||
bit 0-1 : dictID (0, 1, 2 or 4 bytes)
|
|
||||||
bit 2-7 : reserved (must be zero)
|
|
||||||
|
|
||||||
|
|
||||||
// new
|
|
||||||
1 byte - FrameHeaderDescription :
|
1 byte - FrameHeaderDescription :
|
||||||
bit 0-1 : dictID (0, 1, 2 or 4 bytes)
|
bit 0-1 : dictID (0, 1, 2 or 4 bytes)
|
||||||
bit 2-4 : reserved (must be zero)
|
bit 2-4 : reserved (must be zero)
|
||||||
@@ -453,24 +440,24 @@ static void ZSTD_reduceIndex (ZSTD_CCtx* zc, const U32 reducerValue)
|
|||||||
bit 0-2 : octal Fractional (1/8th)
|
bit 0-2 : octal Fractional (1/8th)
|
||||||
bit 3-7 : Power of 2, with 0 = 1 KB (up to 2 TB)
|
bit 3-7 : Power of 2, with 0 = 1 KB (up to 2 TB)
|
||||||
|
|
||||||
|
Optional : content size (0, 1, 2, 4 or 8 bytes)
|
||||||
|
0 : unknown
|
||||||
|
1 : 0-255 bytes
|
||||||
|
2 : 256 - 65535+256
|
||||||
|
8 : up to 16 exa
|
||||||
|
|
||||||
Optional : dictID (0, 1, 2 or 4 bytes)
|
Optional : dictID (0, 1, 2 or 4 bytes)
|
||||||
Automatic adaptation
|
Automatic adaptation
|
||||||
0 : no dictID
|
0 : no dictID
|
||||||
1 : 1 - 255
|
1 : 1 - 255
|
||||||
2 : 256 - 65535
|
2 : 256 - 65535
|
||||||
4 : all other values
|
4 : all other values
|
||||||
|
|
||||||
Optional : content size (0, 1, 2, 4 or 8 bytes)
|
|
||||||
0 : unknown
|
|
||||||
1 : 0-255 bytes
|
|
||||||
2 : 256 - 65535+256
|
|
||||||
8 : up to 16 exa
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
|
|
||||||
/* Block format description
|
/* Block format description
|
||||||
|
|
||||||
Block = Literal Section - Sequences Section
|
Block = Literals Section - Sequences Section
|
||||||
Prerequisite : size of (compressed) block, maximum size of regenerated data
|
Prerequisite : size of (compressed) block, maximum size of regenerated data
|
||||||
|
|
||||||
1) Literal Section
|
1) Literal Section
|
||||||
@@ -478,7 +465,7 @@ static void ZSTD_reduceIndex (ZSTD_CCtx* zc, const U32 reducerValue)
|
|||||||
1.1) Header : 1-5 bytes
|
1.1) Header : 1-5 bytes
|
||||||
flags: 2 bits
|
flags: 2 bits
|
||||||
00 compressed by Huff0
|
00 compressed by Huff0
|
||||||
01 unused
|
01 repeat
|
||||||
10 is Raw (uncompressed)
|
10 is Raw (uncompressed)
|
||||||
11 is Rle
|
11 is Rle
|
||||||
Note : using 01 => Huff0 with precomputed table ?
|
Note : using 01 => Huff0 with precomputed table ?
|
||||||
@@ -514,7 +501,7 @@ static void ZSTD_reduceIndex (ZSTD_CCtx* zc, const U32 reducerValue)
|
|||||||
else => 5 bytes (2-2-18-18)
|
else => 5 bytes (2-2-18-18)
|
||||||
big endian convention
|
big endian convention
|
||||||
|
|
||||||
1- CTable available (stored into workspace ?)
|
1- CTable available (stored into workspace)
|
||||||
2- Small input (fast heuristic ? Full comparison ? depend on clevel ?)
|
2- Small input (fast heuristic ? Full comparison ? depend on clevel ?)
|
||||||
|
|
||||||
|
|
||||||
@@ -936,7 +923,7 @@ _check_compressibility:
|
|||||||
`offsetCode` : distance to match, or 0 == repCode.
|
`offsetCode` : distance to match, or 0 == repCode.
|
||||||
`matchCode` : matchLength - MINMATCH
|
`matchCode` : matchLength - MINMATCH
|
||||||
*/
|
*/
|
||||||
MEM_STATIC void ZSTD_storeSeq(seqStore_t* seqStorePtr, size_t litLength, const void* literals, size_t offsetCode, size_t matchCode)
|
MEM_STATIC void ZSTD_storeSeq(seqStore_t* seqStorePtr, size_t litLength, const void* literals, U32 offsetCode, size_t matchCode)
|
||||||
{
|
{
|
||||||
#if 0 /* for debug */
|
#if 0 /* for debug */
|
||||||
static const BYTE* g_start = NULL;
|
static const BYTE* g_start = NULL;
|
||||||
@@ -957,7 +944,7 @@ MEM_STATIC void ZSTD_storeSeq(seqStore_t* seqStorePtr, size_t litLength, const v
|
|||||||
*seqStorePtr->litLength++ = (U16)litLength;
|
*seqStorePtr->litLength++ = (U16)litLength;
|
||||||
|
|
||||||
/* match offset */
|
/* match offset */
|
||||||
*(seqStorePtr->offset++) = (U32)offsetCode + 1;
|
*(seqStorePtr->offset++) = offsetCode + 1;
|
||||||
|
|
||||||
/* match Length */
|
/* match Length */
|
||||||
if (matchCode>0xFFFF) { seqStorePtr->longLengthID = 2; seqStorePtr->longLengthPos = (U32)(seqStorePtr->matchLength - seqStorePtr->matchLengthStart); }
|
if (matchCode>0xFFFF) { seqStorePtr->longLengthID = 2; seqStorePtr->longLengthPos = (U32)(seqStorePtr->matchLength - seqStorePtr->matchLengthStart); }
|
||||||
@@ -1063,7 +1050,7 @@ static size_t ZSTD_count_2segments(const BYTE* ip, const BYTE* match, const BYTE
|
|||||||
***************************************/
|
***************************************/
|
||||||
static const U32 prime3bytes = 506832829U;
|
static const U32 prime3bytes = 506832829U;
|
||||||
static U32 ZSTD_hash3(U32 u, U32 h) { return ((u << (32-24)) * prime3bytes) >> (32-h) ; }
|
static U32 ZSTD_hash3(U32 u, U32 h) { return ((u << (32-24)) * prime3bytes) >> (32-h) ; }
|
||||||
static size_t ZSTD_hash3Ptr(const void* ptr, U32 h) { return ZSTD_hash3(MEM_readLE32(ptr), h); }
|
MEM_STATIC size_t ZSTD_hash3Ptr(const void* ptr, U32 h) { return ZSTD_hash3(MEM_readLE32(ptr), h); } /* only in zstd_opt.h */
|
||||||
|
|
||||||
static const U32 prime4bytes = 2654435761U;
|
static const U32 prime4bytes = 2654435761U;
|
||||||
static U32 ZSTD_hash4(U32 u, U32 h) { return (u * prime4bytes) >> (32-h) ; }
|
static U32 ZSTD_hash4(U32 u, U32 h) { return (u * prime4bytes) >> (32-h) ; }
|
||||||
@@ -1129,13 +1116,14 @@ void ZSTD_compressBlock_fast_generic(ZSTD_CCtx* cctx,
|
|||||||
const BYTE* const lowest = base + lowestIndex;
|
const BYTE* const lowest = base + lowestIndex;
|
||||||
const BYTE* const iend = istart + srcSize;
|
const BYTE* const iend = istart + srcSize;
|
||||||
const BYTE* const ilimit = iend - 8;
|
const BYTE* const ilimit = iend - 8;
|
||||||
size_t offset_1=cctx->rep[0], offset_2=cctx->rep[1];
|
U32 offset_1=cctx->rep[0], offset_2=cctx->rep[1];
|
||||||
|
U32 offsetSaved = 0;
|
||||||
|
|
||||||
/* init */
|
/* init */
|
||||||
ip += (ip==lowest);
|
ip += (ip==lowest);
|
||||||
{ U32 const maxRep = (U32)(ip-lowest);
|
{ U32 const maxRep = (U32)(ip-lowest);
|
||||||
if (offset_1 > maxRep) offset_1 = 0;
|
if (offset_2 > maxRep) offsetSaved = offset_2, offset_2 = 0;
|
||||||
if (offset_2 > maxRep) offset_2 = 0;
|
if (offset_1 > maxRep) offsetSaved = offset_1, offset_1 = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Main Search Loop */
|
/* Main Search Loop */
|
||||||
@@ -1152,13 +1140,13 @@ void ZSTD_compressBlock_fast_generic(ZSTD_CCtx* cctx,
|
|||||||
ip++;
|
ip++;
|
||||||
ZSTD_storeSeq(seqStorePtr, ip-anchor, anchor, 0, mLength-MINMATCH);
|
ZSTD_storeSeq(seqStorePtr, ip-anchor, anchor, 0, mLength-MINMATCH);
|
||||||
} else {
|
} else {
|
||||||
size_t offset;
|
U32 offset;
|
||||||
if ( (matchIndex <= lowestIndex) || (MEM_read32(match) != MEM_read32(ip)) ) {
|
if ( (matchIndex <= lowestIndex) || (MEM_read32(match) != MEM_read32(ip)) ) {
|
||||||
ip += ((ip-anchor) >> g_searchStrength) + 1;
|
ip += ((ip-anchor) >> g_searchStrength) + 1;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
mLength = ZSTD_count(ip+EQUAL_READ32, match+EQUAL_READ32, iend) + EQUAL_READ32;
|
mLength = ZSTD_count(ip+EQUAL_READ32, match+EQUAL_READ32, iend) + EQUAL_READ32;
|
||||||
offset = ip-match;
|
offset = (U32)(ip-match);
|
||||||
while (((ip>anchor) & (match>lowest)) && (ip[-1] == match[-1])) { ip--; match--; mLength++; } /* catch up */
|
while (((ip>anchor) & (match>lowest)) && (ip[-1] == match[-1])) { ip--; match--; mLength++; } /* catch up */
|
||||||
offset_2 = offset_1;
|
offset_2 = offset_1;
|
||||||
offset_1 = offset;
|
offset_1 = offset;
|
||||||
@@ -1180,7 +1168,7 @@ void ZSTD_compressBlock_fast_generic(ZSTD_CCtx* cctx,
|
|||||||
& (MEM_read32(ip) == MEM_read32(ip - offset_2)) )) {
|
& (MEM_read32(ip) == MEM_read32(ip - offset_2)) )) {
|
||||||
/* store sequence */
|
/* store sequence */
|
||||||
size_t const rLength = ZSTD_count(ip+EQUAL_READ32, ip+EQUAL_READ32-offset_2, iend) + EQUAL_READ32;
|
size_t const rLength = ZSTD_count(ip+EQUAL_READ32, ip+EQUAL_READ32-offset_2, iend) + EQUAL_READ32;
|
||||||
{ size_t const tmpOff = offset_2; offset_2 = offset_1; offset_1 = tmpOff; } /* swap offset_2 <=> offset_1 */
|
{ U32 const tmpOff = offset_2; offset_2 = offset_1; offset_1 = tmpOff; } /* swap offset_2 <=> offset_1 */
|
||||||
hashTable[ZSTD_hashPtr(ip, hBits, mls)] = (U32)(ip-base);
|
hashTable[ZSTD_hashPtr(ip, hBits, mls)] = (U32)(ip-base);
|
||||||
ZSTD_storeSeq(seqStorePtr, 0, anchor, 0, rLength-MINMATCH);
|
ZSTD_storeSeq(seqStorePtr, 0, anchor, 0, rLength-MINMATCH);
|
||||||
ip += rLength;
|
ip += rLength;
|
||||||
@@ -1189,8 +1177,8 @@ void ZSTD_compressBlock_fast_generic(ZSTD_CCtx* cctx,
|
|||||||
} } }
|
} } }
|
||||||
|
|
||||||
/* save reps for next block */
|
/* save reps for next block */
|
||||||
cctx->savedRep[0] = offset_1 ? (U32)offset_1 : (U32)(iend - base) + 1;
|
cctx->savedRep[0] = offset_1 ? offset_1 : offsetSaved;
|
||||||
cctx->savedRep[1] = offset_2 ? (U32)offset_2 : (U32)(iend - base) + 1;
|
cctx->savedRep[1] = offset_2 ? offset_2 : offsetSaved;
|
||||||
|
|
||||||
/* Last Literals */
|
/* Last Literals */
|
||||||
{ size_t const lastLLSize = iend - anchor;
|
{ size_t const lastLLSize = iend - anchor;
|
||||||
@@ -1364,17 +1352,19 @@ static U32 ZSTD_insertBt1(ZSTD_CCtx* zc, const BYTE* const ip, const U32 mls, co
|
|||||||
const U32 windowLow = zc->lowLimit;
|
const U32 windowLow = zc->lowLimit;
|
||||||
U32 matchEndIdx = current+8;
|
U32 matchEndIdx = current+8;
|
||||||
size_t bestLength = 8;
|
size_t bestLength = 8;
|
||||||
|
#ifdef ZSTD_C_PREDICT
|
||||||
U32 predictedSmall = *(bt + 2*((current-1)&btMask) + 0);
|
U32 predictedSmall = *(bt + 2*((current-1)&btMask) + 0);
|
||||||
U32 predictedLarge = *(bt + 2*((current-1)&btMask) + 1);
|
U32 predictedLarge = *(bt + 2*((current-1)&btMask) + 1);
|
||||||
predictedSmall += (predictedSmall>0);
|
predictedSmall += (predictedSmall>0);
|
||||||
predictedLarge += (predictedLarge>0);
|
predictedLarge += (predictedLarge>0);
|
||||||
|
#endif /* ZSTD_C_PREDICT */
|
||||||
|
|
||||||
hashTable[h] = current; /* Update Hash Table */
|
hashTable[h] = current; /* Update Hash Table */
|
||||||
|
|
||||||
while (nbCompares-- && (matchIndex > windowLow)) {
|
while (nbCompares-- && (matchIndex > windowLow)) {
|
||||||
U32* nextPtr = bt + 2*(matchIndex & btMask);
|
U32* nextPtr = bt + 2*(matchIndex & btMask);
|
||||||
size_t matchLength = MIN(commonLengthSmaller, commonLengthLarger); /* guaranteed minimum nb of common bytes */
|
size_t matchLength = MIN(commonLengthSmaller, commonLengthLarger); /* guaranteed minimum nb of common bytes */
|
||||||
#if 0 /* note : can create issues when hlog small <= 11 */
|
#ifdef ZSTD_C_PREDICT /* note : can create issues when hlog small <= 11 */
|
||||||
const U32* predictPtr = bt + 2*((matchIndex-1) & btMask); /* written this way, as bt is a roll buffer */
|
const U32* predictPtr = bt + 2*((matchIndex-1) & btMask); /* written this way, as bt is a roll buffer */
|
||||||
if (matchIndex == predictedSmall) {
|
if (matchIndex == predictedSmall) {
|
||||||
/* no need to check length, result known */
|
/* no need to check length, result known */
|
||||||
@@ -1731,17 +1721,15 @@ void ZSTD_compressBlock_lazy_generic(ZSTD_CCtx* ctx,
|
|||||||
size_t* offsetPtr,
|
size_t* offsetPtr,
|
||||||
U32 maxNbAttempts, U32 matchLengthSearch);
|
U32 maxNbAttempts, U32 matchLengthSearch);
|
||||||
searchMax_f const searchMax = searchMethod ? ZSTD_BtFindBestMatch_selectMLS : ZSTD_HcFindBestMatch_selectMLS;
|
searchMax_f const searchMax = searchMethod ? ZSTD_BtFindBestMatch_selectMLS : ZSTD_HcFindBestMatch_selectMLS;
|
||||||
U32 rep[ZSTD_REP_INIT];
|
U32 offset_1 = ctx->rep[0], offset_2 = ctx->rep[1], savedOffset=0;
|
||||||
|
|
||||||
/* init */
|
/* init */
|
||||||
ip += (ip==base);
|
ip += (ip==base);
|
||||||
ctx->nextToUpdate3 = ctx->nextToUpdate;
|
ctx->nextToUpdate3 = ctx->nextToUpdate;
|
||||||
{ U32 i;
|
{ U32 const maxRep = (U32)(ip-base);
|
||||||
U32 const maxRep = (U32)(ip-base);
|
if (offset_2 > maxRep) savedOffset = offset_2, offset_2 = 0;
|
||||||
for (i=0; i<ZSTD_REP_INIT; i++) {
|
if (offset_1 > maxRep) savedOffset = offset_1, offset_1 = 0;
|
||||||
rep[i]=ctx->rep[i];
|
}
|
||||||
if (rep[i]>maxRep) rep[i]=0;
|
|
||||||
} }
|
|
||||||
|
|
||||||
/* Match Loop */
|
/* Match Loop */
|
||||||
while (ip < ilimit) {
|
while (ip < ilimit) {
|
||||||
@@ -1750,9 +1738,9 @@ void ZSTD_compressBlock_lazy_generic(ZSTD_CCtx* ctx,
|
|||||||
const BYTE* start=ip+1;
|
const BYTE* start=ip+1;
|
||||||
|
|
||||||
/* check repCode */
|
/* check repCode */
|
||||||
if ((rep[0]>0) & (MEM_read32(ip+1) == MEM_read32(ip+1 - rep[0]))) {
|
if ((offset_1>0) & (MEM_read32(ip+1) == MEM_read32(ip+1 - offset_1))) {
|
||||||
/* repcode : we take it */
|
/* repcode : we take it */
|
||||||
matchLength = ZSTD_count(ip+1+EQUAL_READ32, ip+1+EQUAL_READ32-rep[0], iend) + EQUAL_READ32;
|
matchLength = ZSTD_count(ip+1+EQUAL_READ32, ip+1+EQUAL_READ32-offset_1, iend) + EQUAL_READ32;
|
||||||
if (depth==0) goto _storeSequence;
|
if (depth==0) goto _storeSequence;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1772,8 +1760,8 @@ void ZSTD_compressBlock_lazy_generic(ZSTD_CCtx* ctx,
|
|||||||
if (depth>=1)
|
if (depth>=1)
|
||||||
while (ip<ilimit) {
|
while (ip<ilimit) {
|
||||||
ip ++;
|
ip ++;
|
||||||
if ((offset) && ((rep[0]>0) & (MEM_read32(ip) == MEM_read32(ip - rep[0])))) {
|
if ((offset) && ((offset_1>0) & (MEM_read32(ip) == MEM_read32(ip - offset_1)))) {
|
||||||
size_t const mlRep = ZSTD_count(ip+EQUAL_READ32, ip+EQUAL_READ32-rep[0], iend) + EQUAL_READ32;
|
size_t const mlRep = ZSTD_count(ip+EQUAL_READ32, ip+EQUAL_READ32-offset_1, iend) + EQUAL_READ32;
|
||||||
int const gain2 = (int)(mlRep * 3);
|
int const gain2 = (int)(mlRep * 3);
|
||||||
int const gain1 = (int)(matchLength*3 - ZSTD_highbit32((U32)offset+1) + 1);
|
int const gain1 = (int)(matchLength*3 - ZSTD_highbit32((U32)offset+1) + 1);
|
||||||
if ((mlRep >= EQUAL_READ32) && (gain2 > gain1))
|
if ((mlRep >= EQUAL_READ32) && (gain2 > gain1))
|
||||||
@@ -1791,8 +1779,8 @@ void ZSTD_compressBlock_lazy_generic(ZSTD_CCtx* ctx,
|
|||||||
/* let's find an even better one */
|
/* let's find an even better one */
|
||||||
if ((depth==2) && (ip<ilimit)) {
|
if ((depth==2) && (ip<ilimit)) {
|
||||||
ip ++;
|
ip ++;
|
||||||
if ((offset) && ((rep[0]>0) & (MEM_read32(ip) == MEM_read32(ip - rep[0])))) {
|
if ((offset) && ((offset_1>0) & (MEM_read32(ip) == MEM_read32(ip - offset_1)))) {
|
||||||
size_t const ml2 = ZSTD_count(ip+EQUAL_READ32, ip+EQUAL_READ32-rep[0], iend) + EQUAL_READ32;
|
size_t const ml2 = ZSTD_count(ip+EQUAL_READ32, ip+EQUAL_READ32-offset_1, iend) + EQUAL_READ32;
|
||||||
int const gain2 = (int)(ml2 * 4);
|
int const gain2 = (int)(ml2 * 4);
|
||||||
int const gain1 = (int)(matchLength*4 - ZSTD_highbit32((U32)offset+1) + 1);
|
int const gain1 = (int)(matchLength*4 - ZSTD_highbit32((U32)offset+1) + 1);
|
||||||
if ((ml2 >= EQUAL_READ32) && (gain2 > gain1))
|
if ((ml2 >= EQUAL_READ32) && (gain2 > gain1))
|
||||||
@@ -1813,23 +1801,23 @@ void ZSTD_compressBlock_lazy_generic(ZSTD_CCtx* ctx,
|
|||||||
if (offset) {
|
if (offset) {
|
||||||
while ((start>anchor) && (start>base+offset-ZSTD_REP_MOVE) && (start[-1] == start[-1-offset+ZSTD_REP_MOVE])) /* only search for offset within prefix */
|
while ((start>anchor) && (start>base+offset-ZSTD_REP_MOVE) && (start[-1] == start[-1-offset+ZSTD_REP_MOVE])) /* only search for offset within prefix */
|
||||||
{ start--; matchLength++; }
|
{ start--; matchLength++; }
|
||||||
rep[1] = rep[0]; rep[0] = (U32)(offset - ZSTD_REP_MOVE);
|
offset_2 = offset_1; offset_1 = (U32)(offset - ZSTD_REP_MOVE);
|
||||||
}
|
}
|
||||||
|
|
||||||
/* store sequence */
|
/* store sequence */
|
||||||
_storeSequence:
|
_storeSequence:
|
||||||
{ size_t const litLength = start - anchor;
|
{ size_t const litLength = start - anchor;
|
||||||
ZSTD_storeSeq(seqStorePtr, litLength, anchor, offset, matchLength-MINMATCH);
|
ZSTD_storeSeq(seqStorePtr, litLength, anchor, (U32)offset, matchLength-MINMATCH);
|
||||||
anchor = ip = start + matchLength;
|
anchor = ip = start + matchLength;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* check immediate repcode */
|
/* check immediate repcode */
|
||||||
while ( (ip <= ilimit)
|
while ( (ip <= ilimit)
|
||||||
&& ((rep[1]>0)
|
&& ((offset_2>0)
|
||||||
& (MEM_read32(ip) == MEM_read32(ip - rep[1])) )) {
|
& (MEM_read32(ip) == MEM_read32(ip - offset_2)) )) {
|
||||||
/* store sequence */
|
/* store sequence */
|
||||||
matchLength = ZSTD_count(ip+EQUAL_READ32, ip+EQUAL_READ32-rep[1], iend) + EQUAL_READ32;
|
matchLength = ZSTD_count(ip+EQUAL_READ32, ip+EQUAL_READ32-offset_2, iend) + EQUAL_READ32;
|
||||||
offset = rep[1]; rep[1] = rep[0]; rep[0] = (U32)offset; /* swap repcodes */
|
offset = offset_2; offset_2 = offset_1; offset_1 = (U32)offset; /* swap repcodes */
|
||||||
ZSTD_storeSeq(seqStorePtr, 0, anchor, 0, matchLength-MINMATCH);
|
ZSTD_storeSeq(seqStorePtr, 0, anchor, 0, matchLength-MINMATCH);
|
||||||
ip += matchLength;
|
ip += matchLength;
|
||||||
anchor = ip;
|
anchor = ip;
|
||||||
@@ -1837,11 +1825,8 @@ _storeSequence:
|
|||||||
} }
|
} }
|
||||||
|
|
||||||
/* Save reps for next block */
|
/* Save reps for next block */
|
||||||
{ int i;
|
ctx->savedRep[0] = offset_1 ? offset_1 : savedOffset;
|
||||||
for (i=0; i<ZSTD_REP_NUM; i++) {
|
ctx->savedRep[1] = offset_2 ? offset_2 : savedOffset;
|
||||||
if (!rep[i]) rep[i] = (U32)(iend - ctx->base) + 1; /* in case some zero are left */
|
|
||||||
ctx->savedRep[i] = rep[i];
|
|
||||||
} }
|
|
||||||
|
|
||||||
/* Last Literals */
|
/* Last Literals */
|
||||||
{ size_t const lastLLSize = iend - anchor;
|
{ size_t const lastLLSize = iend - anchor;
|
||||||
@@ -1900,10 +1885,9 @@ void ZSTD_compressBlock_lazy_extDict_generic(ZSTD_CCtx* ctx,
|
|||||||
U32 maxNbAttempts, U32 matchLengthSearch);
|
U32 maxNbAttempts, U32 matchLengthSearch);
|
||||||
searchMax_f searchMax = searchMethod ? ZSTD_BtFindBestMatch_selectMLS_extDict : ZSTD_HcFindBestMatch_extDict_selectMLS;
|
searchMax_f searchMax = searchMethod ? ZSTD_BtFindBestMatch_selectMLS_extDict : ZSTD_HcFindBestMatch_extDict_selectMLS;
|
||||||
|
|
||||||
/* init */
|
U32 offset_1 = ctx->rep[0], offset_2 = ctx->rep[1];
|
||||||
U32 rep[ZSTD_REP_INIT];
|
|
||||||
{ U32 i; for (i=0; i<ZSTD_REP_INIT; i++) rep[i]=ctx->rep[i]; }
|
|
||||||
|
|
||||||
|
/* init */
|
||||||
ctx->nextToUpdate3 = ctx->nextToUpdate;
|
ctx->nextToUpdate3 = ctx->nextToUpdate;
|
||||||
ip += (ip == prefixStart);
|
ip += (ip == prefixStart);
|
||||||
|
|
||||||
@@ -1915,7 +1899,7 @@ void ZSTD_compressBlock_lazy_extDict_generic(ZSTD_CCtx* ctx,
|
|||||||
U32 current = (U32)(ip-base);
|
U32 current = (U32)(ip-base);
|
||||||
|
|
||||||
/* check repCode */
|
/* check repCode */
|
||||||
{ const U32 repIndex = (U32)(current+1 - rep[0]);
|
{ const U32 repIndex = (U32)(current+1 - offset_1);
|
||||||
const BYTE* const repBase = repIndex < dictLimit ? dictBase : base;
|
const BYTE* const repBase = repIndex < dictLimit ? dictBase : base;
|
||||||
const BYTE* const repMatch = repBase + repIndex;
|
const BYTE* const repMatch = repBase + repIndex;
|
||||||
if (((U32)((dictLimit-1) - repIndex) >= 3) & (repIndex > lowestIndex)) /* intentional overflow */
|
if (((U32)((dictLimit-1) - repIndex) >= 3) & (repIndex > lowestIndex)) /* intentional overflow */
|
||||||
@@ -1945,7 +1929,7 @@ void ZSTD_compressBlock_lazy_extDict_generic(ZSTD_CCtx* ctx,
|
|||||||
current++;
|
current++;
|
||||||
/* check repCode */
|
/* check repCode */
|
||||||
if (offset) {
|
if (offset) {
|
||||||
const U32 repIndex = (U32)(current - rep[0]);
|
const U32 repIndex = (U32)(current - offset_1);
|
||||||
const BYTE* const repBase = repIndex < dictLimit ? dictBase : base;
|
const BYTE* const repBase = repIndex < dictLimit ? dictBase : base;
|
||||||
const BYTE* const repMatch = repBase + repIndex;
|
const BYTE* const repMatch = repBase + repIndex;
|
||||||
if (((U32)((dictLimit-1) - repIndex) >= 3) & (repIndex > lowestIndex)) /* intentional overflow */
|
if (((U32)((dictLimit-1) - repIndex) >= 3) & (repIndex > lowestIndex)) /* intentional overflow */
|
||||||
@@ -1975,7 +1959,7 @@ void ZSTD_compressBlock_lazy_extDict_generic(ZSTD_CCtx* ctx,
|
|||||||
current++;
|
current++;
|
||||||
/* check repCode */
|
/* check repCode */
|
||||||
if (offset) {
|
if (offset) {
|
||||||
const U32 repIndex = (U32)(current - rep[0]);
|
const U32 repIndex = (U32)(current - offset_1);
|
||||||
const BYTE* const repBase = repIndex < dictLimit ? dictBase : base;
|
const BYTE* const repBase = repIndex < dictLimit ? dictBase : base;
|
||||||
const BYTE* const repMatch = repBase + repIndex;
|
const BYTE* const repMatch = repBase + repIndex;
|
||||||
if (((U32)((dictLimit-1) - repIndex) >= 3) & (repIndex > lowestIndex)) /* intentional overflow */
|
if (((U32)((dictLimit-1) - repIndex) >= 3) & (repIndex > lowestIndex)) /* intentional overflow */
|
||||||
@@ -2007,19 +1991,19 @@ void ZSTD_compressBlock_lazy_extDict_generic(ZSTD_CCtx* ctx,
|
|||||||
const BYTE* match = (matchIndex < dictLimit) ? dictBase + matchIndex : base + matchIndex;
|
const BYTE* match = (matchIndex < dictLimit) ? dictBase + matchIndex : base + matchIndex;
|
||||||
const BYTE* const mStart = (matchIndex < dictLimit) ? dictStart : prefixStart;
|
const BYTE* const mStart = (matchIndex < dictLimit) ? dictStart : prefixStart;
|
||||||
while ((start>anchor) && (match>mStart) && (start[-1] == match[-1])) { start--; match--; matchLength++; } /* catch up */
|
while ((start>anchor) && (match>mStart) && (start[-1] == match[-1])) { start--; match--; matchLength++; } /* catch up */
|
||||||
rep[1] = rep[0]; rep[0] = (U32)(offset - ZSTD_REP_MOVE);
|
offset_2 = offset_1; offset_1 = (U32)(offset - ZSTD_REP_MOVE);
|
||||||
}
|
}
|
||||||
|
|
||||||
/* store sequence */
|
/* store sequence */
|
||||||
_storeSequence:
|
_storeSequence:
|
||||||
{ size_t const litLength = start - anchor;
|
{ size_t const litLength = start - anchor;
|
||||||
ZSTD_storeSeq(seqStorePtr, litLength, anchor, offset, matchLength-MINMATCH);
|
ZSTD_storeSeq(seqStorePtr, litLength, anchor, (U32)offset, matchLength-MINMATCH);
|
||||||
anchor = ip = start + matchLength;
|
anchor = ip = start + matchLength;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* check immediate repcode */
|
/* check immediate repcode */
|
||||||
while (ip <= ilimit) {
|
while (ip <= ilimit) {
|
||||||
const U32 repIndex = (U32)((ip-base) - rep[1]);
|
const U32 repIndex = (U32)((ip-base) - offset_2);
|
||||||
const BYTE* const repBase = repIndex < dictLimit ? dictBase : base;
|
const BYTE* const repBase = repIndex < dictLimit ? dictBase : base;
|
||||||
const BYTE* const repMatch = repBase + repIndex;
|
const BYTE* const repMatch = repBase + repIndex;
|
||||||
if (((U32)((dictLimit-1) - repIndex) >= 3) & (repIndex > lowestIndex)) /* intentional overflow */
|
if (((U32)((dictLimit-1) - repIndex) >= 3) & (repIndex > lowestIndex)) /* intentional overflow */
|
||||||
@@ -2027,7 +2011,7 @@ _storeSequence:
|
|||||||
/* repcode detected we should take it */
|
/* repcode detected we should take it */
|
||||||
const BYTE* const repEnd = repIndex < dictLimit ? dictEnd : iend;
|
const BYTE* const repEnd = repIndex < dictLimit ? dictEnd : iend;
|
||||||
matchLength = ZSTD_count_2segments(ip+EQUAL_READ32, repMatch+EQUAL_READ32, iend, repEnd, prefixStart) + EQUAL_READ32;
|
matchLength = ZSTD_count_2segments(ip+EQUAL_READ32, repMatch+EQUAL_READ32, iend, repEnd, prefixStart) + EQUAL_READ32;
|
||||||
offset = rep[1]; rep[1] = rep[0]; rep[0] = (U32)offset; /* swap offset history */
|
offset = offset_2; offset_2 = offset_1; offset_1 = (U32)offset; /* swap offset history */
|
||||||
ZSTD_storeSeq(seqStorePtr, 0, anchor, 0, matchLength-MINMATCH);
|
ZSTD_storeSeq(seqStorePtr, 0, anchor, 0, matchLength-MINMATCH);
|
||||||
ip += matchLength;
|
ip += matchLength;
|
||||||
anchor = ip;
|
anchor = ip;
|
||||||
@@ -2037,7 +2021,7 @@ _storeSequence:
|
|||||||
} }
|
} }
|
||||||
|
|
||||||
/* Save reps for next block */
|
/* Save reps for next block */
|
||||||
ctx->savedRep[0] = rep[0]; ctx->savedRep[1] = rep[1]; ctx->savedRep[2] = rep[2];
|
ctx->savedRep[0] = offset_1; ctx->savedRep[1] = offset_2;
|
||||||
|
|
||||||
/* Last Literals */
|
/* Last Literals */
|
||||||
{ size_t const lastLLSize = iend - anchor;
|
{ size_t const lastLLSize = iend - anchor;
|
||||||
@@ -2068,18 +2052,27 @@ static void ZSTD_compressBlock_btlazy2_extDict(ZSTD_CCtx* ctx, const void* src,
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
/* The optimal parser */
|
/* The optimal parser */
|
||||||
#include "zstd_opt.h"
|
#include "zstd_opt.h"
|
||||||
|
|
||||||
static void ZSTD_compressBlock_btopt(ZSTD_CCtx* ctx, const void* src, size_t srcSize)
|
static void ZSTD_compressBlock_btopt(ZSTD_CCtx* ctx, const void* src, size_t srcSize)
|
||||||
{
|
{
|
||||||
|
#ifdef ZSTD_OPT_H_91842398743
|
||||||
ZSTD_compressBlock_opt_generic(ctx, src, srcSize);
|
ZSTD_compressBlock_opt_generic(ctx, src, srcSize);
|
||||||
|
#else
|
||||||
|
(void)ctx; (void)src; (void)srcSize;
|
||||||
|
return;
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
static void ZSTD_compressBlock_btopt_extDict(ZSTD_CCtx* ctx, const void* src, size_t srcSize)
|
static void ZSTD_compressBlock_btopt_extDict(ZSTD_CCtx* ctx, const void* src, size_t srcSize)
|
||||||
{
|
{
|
||||||
|
#ifdef ZSTD_OPT_H_91842398743
|
||||||
ZSTD_compressBlock_opt_extDict_generic(ctx, src, srcSize);
|
ZSTD_compressBlock_opt_extDict_generic(ctx, src, srcSize);
|
||||||
|
#else
|
||||||
|
(void)ctx; (void)src; (void)srcSize;
|
||||||
|
return;
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -2414,7 +2407,7 @@ static size_t ZSTD_compressBegin_internal(ZSTD_CCtx* zc,
|
|||||||
* @return : 0, or an error code */
|
* @return : 0, or an error code */
|
||||||
size_t ZSTD_compressBegin_advanced(ZSTD_CCtx* cctx,
|
size_t ZSTD_compressBegin_advanced(ZSTD_CCtx* cctx,
|
||||||
const void* dict, size_t dictSize,
|
const void* dict, size_t dictSize,
|
||||||
ZSTD_parameters params, U64 pledgedSrcSize)
|
ZSTD_parameters params, unsigned long long pledgedSrcSize)
|
||||||
{
|
{
|
||||||
/* compression parameters verification and optimization */
|
/* compression parameters verification and optimization */
|
||||||
{ size_t const errorCode = ZSTD_checkCParams_advanced(params.cParams, pledgedSrcSize);
|
{ size_t const errorCode = ZSTD_checkCParams_advanced(params.cParams, pledgedSrcSize);
|
||||||
@@ -2426,9 +2419,7 @@ size_t ZSTD_compressBegin_advanced(ZSTD_CCtx* cctx,
|
|||||||
|
|
||||||
size_t ZSTD_compressBegin_usingDict(ZSTD_CCtx* cctx, const void* dict, size_t dictSize, int compressionLevel)
|
size_t ZSTD_compressBegin_usingDict(ZSTD_CCtx* cctx, const void* dict, size_t dictSize, int compressionLevel)
|
||||||
{
|
{
|
||||||
ZSTD_parameters params;
|
ZSTD_parameters const params = ZSTD_getParams(compressionLevel, 0, dictSize);
|
||||||
memset(¶ms, 0, sizeof(params));
|
|
||||||
params.cParams = ZSTD_getCParams(compressionLevel, 0, dictSize);
|
|
||||||
ZSTD_LOG_BLOCK("%p: ZSTD_compressBegin_usingDict compressionLevel=%d\n", cctx->base, compressionLevel);
|
ZSTD_LOG_BLOCK("%p: ZSTD_compressBegin_usingDict compressionLevel=%d\n", cctx->base, compressionLevel);
|
||||||
return ZSTD_compressBegin_internal(cctx, dict, dictSize, params, 0);
|
return ZSTD_compressBegin_internal(cctx, dict, dictSize, params, 0);
|
||||||
}
|
}
|
||||||
@@ -2538,11 +2529,9 @@ size_t ZSTD_compress_advanced (ZSTD_CCtx* ctx,
|
|||||||
|
|
||||||
size_t ZSTD_compress_usingDict(ZSTD_CCtx* ctx, void* dst, size_t dstCapacity, const void* src, size_t srcSize, const void* dict, size_t dictSize, int compressionLevel)
|
size_t ZSTD_compress_usingDict(ZSTD_CCtx* ctx, void* dst, size_t dstCapacity, const void* src, size_t srcSize, const void* dict, size_t dictSize, int compressionLevel)
|
||||||
{
|
{
|
||||||
ZSTD_parameters params;
|
ZSTD_parameters params = ZSTD_getParams(compressionLevel, srcSize, dictSize);
|
||||||
memset(¶ms, 0, sizeof(params));
|
|
||||||
ZSTD_LOG_BLOCK("%p: ZSTD_compress_usingDict srcSize=%d dictSize=%d compressionLevel=%d\n", ctx->base, (int)srcSize, (int)dictSize, compressionLevel);
|
|
||||||
params.cParams = ZSTD_getCParams(compressionLevel, srcSize, dictSize);
|
|
||||||
params.fParams.contentSizeFlag = 1;
|
params.fParams.contentSizeFlag = 1;
|
||||||
|
ZSTD_LOG_BLOCK("%p: ZSTD_compress_usingDict srcSize=%d dictSize=%d compressionLevel=%d\n", ctx->base, (int)srcSize, (int)dictSize, compressionLevel);
|
||||||
return ZSTD_compress_internal(ctx, dst, dstCapacity, src, srcSize, dict, dictSize, params);
|
return ZSTD_compress_internal(ctx, dst, dstCapacity, src, srcSize, dict, dictSize, params);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2577,7 +2566,7 @@ ZSTD_CDict* ZSTD_createCDict_advanced(const void* dict, size_t dictSize, ZSTD_pa
|
|||||||
if (!customMem.customAlloc && !customMem.customFree)
|
if (!customMem.customAlloc && !customMem.customFree)
|
||||||
customMem = defaultCustomMem;
|
customMem = defaultCustomMem;
|
||||||
|
|
||||||
if (!customMem.customAlloc || !customMem.customFree)
|
if (!customMem.customAlloc || !customMem.customFree) /* can't have 1/2 custom alloc/free as NULL */
|
||||||
return NULL;
|
return NULL;
|
||||||
|
|
||||||
{ ZSTD_CDict* const cdict = (ZSTD_CDict*) customMem.customAlloc(customMem.opaque, sizeof(*cdict));
|
{ ZSTD_CDict* const cdict = (ZSTD_CDict*) customMem.customAlloc(customMem.opaque, sizeof(*cdict));
|
||||||
@@ -2755,7 +2744,7 @@ static const ZSTD_compressionParameters ZSTD_defaultCParameters[4][ZSTD_MAX_CLEV
|
|||||||
/*! ZSTD_getCParams() :
|
/*! ZSTD_getCParams() :
|
||||||
* @return ZSTD_compressionParameters structure for a selected compression level, `srcSize` and `dictSize`.
|
* @return ZSTD_compressionParameters structure for a selected compression level, `srcSize` and `dictSize`.
|
||||||
* Size values are optional, provide 0 if not known or unused */
|
* Size values are optional, provide 0 if not known or unused */
|
||||||
ZSTD_compressionParameters ZSTD_getCParams(int compressionLevel, U64 srcSize, size_t dictSize)
|
ZSTD_compressionParameters ZSTD_getCParams(int compressionLevel, unsigned long long srcSize, size_t dictSize)
|
||||||
{
|
{
|
||||||
ZSTD_compressionParameters cp;
|
ZSTD_compressionParameters cp;
|
||||||
size_t const addedSize = srcSize ? 0 : 500;
|
size_t const addedSize = srcSize ? 0 : 500;
|
||||||
@@ -2772,3 +2761,14 @@ ZSTD_compressionParameters ZSTD_getCParams(int compressionLevel, U64 srcSize, si
|
|||||||
cp = ZSTD_adjustCParams(cp, srcSize, dictSize);
|
cp = ZSTD_adjustCParams(cp, srcSize, dictSize);
|
||||||
return cp;
|
return cp;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*! ZSTD_getParams() :
|
||||||
|
* same as ZSTD_getCParams(), but @return a `ZSTD_parameters` object instead of a `ZSTD_compressionParameters`.
|
||||||
|
* All fields of `ZSTD_frameParameters` are set to default (0) */
|
||||||
|
ZSTD_parameters ZSTD_getParams(int compressionLevel, unsigned long long srcSize, size_t dictSize) {
|
||||||
|
ZSTD_parameters params;
|
||||||
|
ZSTD_compressionParameters const cParams = ZSTD_getCParams(compressionLevel, srcSize, dictSize);
|
||||||
|
memset(¶ms, 0, sizeof(params));
|
||||||
|
params.cParams = cParams;
|
||||||
|
return params;
|
||||||
|
}
|
||||||
|
|||||||
@@ -34,6 +34,10 @@
|
|||||||
/* Note : this file is intended to be included within zstd_compress.c */
|
/* Note : this file is intended to be included within zstd_compress.c */
|
||||||
|
|
||||||
|
|
||||||
|
#ifndef ZSTD_OPT_H_91842398743
|
||||||
|
#define ZSTD_OPT_H_91842398743
|
||||||
|
|
||||||
|
|
||||||
#define ZSTD_FREQ_DIV 5
|
#define ZSTD_FREQ_DIV 5
|
||||||
|
|
||||||
/*-*************************************
|
/*-*************************************
|
||||||
@@ -110,7 +114,7 @@ FORCE_INLINE U32 ZSTD_getLiteralPrice(seqStore_t* ssPtr, U32 litLength, const BY
|
|||||||
|
|
||||||
/* literals */
|
/* literals */
|
||||||
if (ssPtr->cachedLiterals == literals) {
|
if (ssPtr->cachedLiterals == literals) {
|
||||||
U32 additional = litLength - ssPtr->cachedLitLength;
|
U32 const additional = litLength - ssPtr->cachedLitLength;
|
||||||
const BYTE* literals2 = ssPtr->cachedLiterals + ssPtr->cachedLitLength;
|
const BYTE* literals2 = ssPtr->cachedLiterals + ssPtr->cachedLitLength;
|
||||||
price = ssPtr->cachedPrice + additional * ssPtr->log2litSum;
|
price = ssPtr->cachedPrice + additional * ssPtr->log2litSum;
|
||||||
for (u=0; u < additional; u++)
|
for (u=0; u < additional; u++)
|
||||||
@@ -150,7 +154,7 @@ FORCE_INLINE U32 ZSTD_getLiteralPrice(seqStore_t* ssPtr, U32 litLength, const BY
|
|||||||
FORCE_INLINE U32 ZSTD_getPrice(seqStore_t* seqStorePtr, U32 litLength, const BYTE* literals, U32 offset, U32 matchLength)
|
FORCE_INLINE U32 ZSTD_getPrice(seqStore_t* seqStorePtr, U32 litLength, const BYTE* literals, U32 offset, U32 matchLength)
|
||||||
{
|
{
|
||||||
/* offset */
|
/* offset */
|
||||||
BYTE offCode = (BYTE)ZSTD_highbit32(offset+1);
|
BYTE const offCode = (BYTE)ZSTD_highbit32(offset+1);
|
||||||
U32 price = offCode + seqStorePtr->log2offCodeSum - ZSTD_highbit32(seqStorePtr->offCodeFreq[offCode]+1);
|
U32 price = offCode + seqStorePtr->log2offCodeSum - ZSTD_highbit32(seqStorePtr->offCodeFreq[offCode]+1);
|
||||||
|
|
||||||
/* match Length */
|
/* match Length */
|
||||||
@@ -196,7 +200,7 @@ MEM_STATIC void ZSTD_updatePrice(seqStore_t* seqStorePtr, U32 litLength, const B
|
|||||||
}
|
}
|
||||||
|
|
||||||
/* match offset */
|
/* match offset */
|
||||||
{ BYTE offCode = (BYTE)ZSTD_highbit32(offset+1);
|
{ BYTE const offCode = (BYTE)ZSTD_highbit32(offset+1);
|
||||||
seqStorePtr->offCodeSum++;
|
seqStorePtr->offCodeSum++;
|
||||||
seqStorePtr->offCodeFreq[offCode]++;
|
seqStorePtr->offCodeFreq[offCode]++;
|
||||||
}
|
}
|
||||||
@@ -232,7 +236,6 @@ MEM_STATIC void ZSTD_updatePrice(seqStore_t* seqStorePtr, U32 litLength, const B
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
/* Update hashTable3 up to ip (excluded)
|
/* Update hashTable3 up to ip (excluded)
|
||||||
Assumption : always within prefix (ie. not within extDict) */
|
Assumption : always within prefix (ie. not within extDict) */
|
||||||
FORCE_INLINE
|
FORCE_INLINE
|
||||||
@@ -1039,3 +1042,5 @@ _storeSequence: /* cur, last_pos, best_mlen, best_off have to be set */
|
|||||||
seqStorePtr->lit += lastLLSize;
|
seqStorePtr->lit += lastLLSize;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#endif /* ZSTD_OPT_H_91842398743 */
|
||||||
|
|||||||
@@ -173,7 +173,7 @@ size_t ZBUFF_decompressContinue(ZBUFF_DCtx* zbd,
|
|||||||
if (ZSTD_isError(hSize)) return hSize;
|
if (ZSTD_isError(hSize)) return hSize;
|
||||||
if (toLoad > (size_t)(iend-ip)) { /* not enough input to load full header */
|
if (toLoad > (size_t)(iend-ip)) { /* not enough input to load full header */
|
||||||
memcpy(zbd->headerBuffer + zbd->lhSize, ip, iend-ip);
|
memcpy(zbd->headerBuffer + zbd->lhSize, ip, iend-ip);
|
||||||
zbd->lhSize += iend-ip; ip = iend; notDone = 0;
|
zbd->lhSize += iend-ip;
|
||||||
*dstCapacityPtr = 0;
|
*dstCapacityPtr = 0;
|
||||||
return (hSize - zbd->lhSize) + ZSTD_blockHeaderSize; /* remaining header bytes + next block header */
|
return (hSize - zbd->lhSize) + ZSTD_blockHeaderSize; /* remaining header bytes + next block header */
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -195,7 +195,7 @@ void ZSTD_copyDCtx(ZSTD_DCtx* dstDCtx, const ZSTD_DCtx* srcDCtx)
|
|||||||
/* Frame format description
|
/* Frame format description
|
||||||
Frame Header - [ Block Header - Block ] - Frame End
|
Frame Header - [ Block Header - Block ] - Frame End
|
||||||
1) Frame Header
|
1) Frame Header
|
||||||
- 4 bytes - Magic Number : ZSTD_MAGICNUMBER (defined within zstd_static.h)
|
- 4 bytes - Magic Number : ZSTD_MAGICNUMBER (defined within zstd.h)
|
||||||
- 1 byte - Frame Descriptor
|
- 1 byte - Frame Descriptor
|
||||||
2) Block Header
|
2) Block Header
|
||||||
- 3 bytes, starting with a 2-bits descriptor
|
- 3 bytes, starting with a 2-bits descriptor
|
||||||
@@ -207,20 +207,8 @@ void ZSTD_copyDCtx(ZSTD_DCtx* dstDCtx, const ZSTD_DCtx* srcDCtx)
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
|
|
||||||
/* Frame descriptor
|
/* Frame Header :
|
||||||
|
|
||||||
// old
|
|
||||||
1 byte - Alloc :
|
|
||||||
bit 0-3 : windowLog - ZSTD_WINDOWLOG_ABSOLUTEMIN (see zstd_internal.h)
|
|
||||||
bit 4 : reserved for windowLog (must be zero)
|
|
||||||
bit 5 : reserved (must be zero)
|
|
||||||
bit 6-7 : Frame content size : unknown, 1 byte, 2 bytes, 8 bytes
|
|
||||||
|
|
||||||
1 byte - checker :
|
|
||||||
bit 0-1 : dictID (0, 1, 2 or 4 bytes)
|
|
||||||
bit 2-7 : reserved (must be zero)
|
|
||||||
|
|
||||||
// new
|
|
||||||
1 byte - FrameHeaderDescription :
|
1 byte - FrameHeaderDescription :
|
||||||
bit 0-1 : dictID (0, 1, 2 or 4 bytes)
|
bit 0-1 : dictID (0, 1, 2 or 4 bytes)
|
||||||
bit 2 : checksumFlag
|
bit 2 : checksumFlag
|
||||||
@@ -403,6 +391,26 @@ size_t ZSTD_getFrameParams(ZSTD_frameParams* fparamsPtr, const void* src, size_t
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
/** ZSTD_getDecompressedSize() :
|
||||||
|
* compatible with legacy mode
|
||||||
|
* @return : decompressed size if known, 0 otherwise
|
||||||
|
note : 0 can mean any of the following :
|
||||||
|
- decompressed size is not provided within frame header
|
||||||
|
- frame header unknown / not supported
|
||||||
|
- frame header not completely provided (`srcSize` too small) */
|
||||||
|
unsigned long long ZSTD_getDecompressedSize(const void* src, size_t srcSize)
|
||||||
|
{
|
||||||
|
#if defined(ZSTD_LEGACY_SUPPORT) && (ZSTD_LEGACY_SUPPORT==1)
|
||||||
|
if (ZSTD_isLegacy(src, srcSize)) return ZSTD_getDecompressedSize_legacy(src, srcSize);
|
||||||
|
#endif
|
||||||
|
{ ZSTD_frameParams fparams;
|
||||||
|
size_t const frResult = ZSTD_getFrameParams(&fparams, src, srcSize);
|
||||||
|
if (frResult!=0) return 0;
|
||||||
|
return fparams.frameContentSize;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
/** ZSTD_decodeFrameHeader() :
|
/** ZSTD_decodeFrameHeader() :
|
||||||
* `srcSize` must be the size provided by ZSTD_frameHeaderSize().
|
* `srcSize` must be the size provided by ZSTD_frameHeaderSize().
|
||||||
* @return : 0 if success, or an error code, which can be tested using ZSTD_isError() */
|
* @return : 0 if success, or an error code, which can be tested using ZSTD_isError() */
|
||||||
@@ -454,16 +462,14 @@ size_t ZSTD_decodeLiteralsBlock(ZSTD_DCtx* dctx,
|
|||||||
const void* src, size_t srcSize) /* note : srcSize < BLOCKSIZE */
|
const void* src, size_t srcSize) /* note : srcSize < BLOCKSIZE */
|
||||||
{
|
{
|
||||||
const BYTE* const istart = (const BYTE*) src;
|
const BYTE* const istart = (const BYTE*) src;
|
||||||
litBlockType_t lbt;
|
|
||||||
|
|
||||||
if (srcSize < MIN_CBLOCK_SIZE) return ERROR(corruption_detected);
|
if (srcSize < MIN_CBLOCK_SIZE) return ERROR(corruption_detected);
|
||||||
lbt = (litBlockType_t)(istart[0]>> 6);
|
|
||||||
|
|
||||||
switch(lbt)
|
switch((litBlockType_t)(istart[0]>> 6))
|
||||||
{
|
{
|
||||||
case lbt_huffman:
|
case lbt_huffman:
|
||||||
{ size_t litSize, litCSize, singleStream=0;
|
{ size_t litSize, litCSize, singleStream=0;
|
||||||
U32 lhSize = ((istart[0]) >> 4) & 3;
|
U32 lhSize = (istart[0] >> 4) & 3;
|
||||||
if (srcSize < 5) return ERROR(corruption_detected); /* srcSize >= MIN_CBLOCK_SIZE == 3; here we need up to 5 for lhSize, + cSize (+nbSeq) */
|
if (srcSize < 5) return ERROR(corruption_detected); /* srcSize >= MIN_CBLOCK_SIZE == 3; here we need up to 5 for lhSize, + cSize (+nbSeq) */
|
||||||
switch(lhSize)
|
switch(lhSize)
|
||||||
{
|
{
|
||||||
@@ -643,7 +649,7 @@ size_t ZSTD_decodeSeqHeaders(int* nbSeqPtr,
|
|||||||
|
|
||||||
/* FSE table descriptors */
|
/* FSE table descriptors */
|
||||||
{ U32 const LLtype = *ip >> 6;
|
{ U32 const LLtype = *ip >> 6;
|
||||||
U32 const Offtype = (*ip >> 4) & 3;
|
U32 const OFtype = (*ip >> 4) & 3;
|
||||||
U32 const MLtype = (*ip >> 2) & 3;
|
U32 const MLtype = (*ip >> 2) & 3;
|
||||||
ip++;
|
ip++;
|
||||||
|
|
||||||
@@ -651,17 +657,17 @@ size_t ZSTD_decodeSeqHeaders(int* nbSeqPtr,
|
|||||||
if (ip > iend-3) return ERROR(srcSize_wrong); /* min : all 3 are "raw", hence no header, but at least xxLog bits per type */
|
if (ip > iend-3) return ERROR(srcSize_wrong); /* min : all 3 are "raw", hence no header, but at least xxLog bits per type */
|
||||||
|
|
||||||
/* Build DTables */
|
/* Build DTables */
|
||||||
{ size_t const bhSize = ZSTD_buildSeqTable(DTableLL, LLtype, MaxLL, LLFSELog, ip, iend-ip, LL_defaultNorm, LL_defaultNormLog, flagRepeatTable);
|
{ size_t const llhSize = ZSTD_buildSeqTable(DTableLL, LLtype, MaxLL, LLFSELog, ip, iend-ip, LL_defaultNorm, LL_defaultNormLog, flagRepeatTable);
|
||||||
if (ZSTD_isError(bhSize)) return ERROR(corruption_detected);
|
if (ZSTD_isError(llhSize)) return ERROR(corruption_detected);
|
||||||
ip += bhSize;
|
ip += llhSize;
|
||||||
}
|
}
|
||||||
{ size_t const bhSize = ZSTD_buildSeqTable(DTableOffb, Offtype, MaxOff, OffFSELog, ip, iend-ip, OF_defaultNorm, OF_defaultNormLog, flagRepeatTable);
|
{ size_t const ofhSize = ZSTD_buildSeqTable(DTableOffb, OFtype, MaxOff, OffFSELog, ip, iend-ip, OF_defaultNorm, OF_defaultNormLog, flagRepeatTable);
|
||||||
if (ZSTD_isError(bhSize)) return ERROR(corruption_detected);
|
if (ZSTD_isError(ofhSize)) return ERROR(corruption_detected);
|
||||||
ip += bhSize;
|
ip += ofhSize;
|
||||||
}
|
}
|
||||||
{ size_t const bhSize = ZSTD_buildSeqTable(DTableML, MLtype, MaxML, MLFSELog, ip, iend-ip, ML_defaultNorm, ML_defaultNormLog, flagRepeatTable);
|
{ size_t const mlhSize = ZSTD_buildSeqTable(DTableML, MLtype, MaxML, MLFSELog, ip, iend-ip, ML_defaultNorm, ML_defaultNormLog, flagRepeatTable);
|
||||||
if (ZSTD_isError(bhSize)) return ERROR(corruption_detected);
|
if (ZSTD_isError(mlhSize)) return ERROR(corruption_detected);
|
||||||
ip += bhSize;
|
ip += mlhSize;
|
||||||
} }
|
} }
|
||||||
|
|
||||||
return ip-istart;
|
return ip-istart;
|
||||||
@@ -702,42 +708,37 @@ static seq_t ZSTD_decodeSequence(seqState_t* seqState)
|
|||||||
0x2000, 0x4000, 0x8000, 0x10000 };
|
0x2000, 0x4000, 0x8000, 0x10000 };
|
||||||
|
|
||||||
static const U32 ML_base[MaxML+1] = {
|
static const U32 ML_base[MaxML+1] = {
|
||||||
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
|
3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18,
|
||||||
16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31,
|
19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32, 33, 34,
|
||||||
32, 34, 36, 38, 40, 44, 48, 56, 64, 80, 96, 0x80, 0x100, 0x200, 0x400, 0x800,
|
35, 37, 39, 41, 43, 47, 51, 59, 67, 83, 99, 0x83, 0x103, 0x203, 0x403, 0x803,
|
||||||
0x1000, 0x2000, 0x4000, 0x8000, 0x10000 };
|
0x1003, 0x2003, 0x4003, 0x8003, 0x10003 };
|
||||||
|
|
||||||
static const U32 OF_base[MaxOff+1] = {
|
static const U32 OF_base[MaxOff+1] = {
|
||||||
0, 1, 3, 7, 0xF, 0x1F, 0x3F, 0x7F,
|
0, 1, 1, 5, 0xD, 0x1D, 0x3D, 0x7D,
|
||||||
0xFF, 0x1FF, 0x3FF, 0x7FF, 0xFFF, 0x1FFF, 0x3FFF, 0x7FFF,
|
0xFD, 0x1FD, 0x3FD, 0x7FD, 0xFFD, 0x1FFD, 0x3FFD, 0x7FFD,
|
||||||
0xFFFF, 0x1FFFF, 0x3FFFF, 0x7FFFF, 0xFFFFF, 0x1FFFFF, 0x3FFFFF, 0x7FFFFF,
|
0xFFFD, 0x1FFFD, 0x3FFFD, 0x7FFFD, 0xFFFFD, 0x1FFFFD, 0x3FFFFD, 0x7FFFFD,
|
||||||
0xFFFFFF, 0x1FFFFFF, 0x3FFFFFF, /*fake*/ 1, 1 };
|
0xFFFFFD, 0x1FFFFFD, 0x3FFFFFD, 0x7FFFFFD, 0xFFFFFFD };
|
||||||
|
|
||||||
/* sequence */
|
/* sequence */
|
||||||
{ size_t offset;
|
{ size_t offset;
|
||||||
if (!ofCode)
|
if (!ofCode)
|
||||||
offset = 0;
|
offset = 0;
|
||||||
else {
|
else {
|
||||||
offset = OF_base[ofCode] + BIT_readBits(&(seqState->DStream), ofBits); /* <= 26 bits */
|
offset = OF_base[ofCode] + BIT_readBits(&(seqState->DStream), ofBits); /* <= (ZSTD_WINDOWLOG_MAX-1) bits */
|
||||||
if (MEM_32bits()) BIT_reloadDStream(&(seqState->DStream));
|
if (MEM_32bits()) BIT_reloadDStream(&(seqState->DStream));
|
||||||
}
|
}
|
||||||
|
|
||||||
if (offset < ZSTD_REP_NUM) {
|
if (ofCode <= 1) {
|
||||||
if (llCode == 0 && offset <= 1) offset = 1-offset;
|
if ((llCode == 0) & (offset <= 1)) offset = 1-offset;
|
||||||
|
if (offset) {
|
||||||
if (offset != 0) {
|
size_t const temp = seqState->prevOffset[offset];
|
||||||
size_t temp = seqState->prevOffset[offset];
|
if (offset != 1) seqState->prevOffset[2] = seqState->prevOffset[1];
|
||||||
if (offset != 1) {
|
|
||||||
seqState->prevOffset[2] = seqState->prevOffset[1];
|
|
||||||
}
|
|
||||||
seqState->prevOffset[1] = seqState->prevOffset[0];
|
seqState->prevOffset[1] = seqState->prevOffset[0];
|
||||||
seqState->prevOffset[0] = offset = temp;
|
seqState->prevOffset[0] = offset = temp;
|
||||||
|
|
||||||
} else {
|
} else {
|
||||||
offset = seqState->prevOffset[0];
|
offset = seqState->prevOffset[0];
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
offset -= ZSTD_REP_MOVE;
|
|
||||||
seqState->prevOffset[2] = seqState->prevOffset[1];
|
seqState->prevOffset[2] = seqState->prevOffset[1];
|
||||||
seqState->prevOffset[1] = seqState->prevOffset[0];
|
seqState->prevOffset[1] = seqState->prevOffset[0];
|
||||||
seqState->prevOffset[0] = offset;
|
seqState->prevOffset[0] = offset;
|
||||||
@@ -745,11 +746,11 @@ static seq_t ZSTD_decodeSequence(seqState_t* seqState)
|
|||||||
seq.offset = offset;
|
seq.offset = offset;
|
||||||
}
|
}
|
||||||
|
|
||||||
seq.matchLength = ML_base[mlCode] + MINMATCH + ((mlCode>31) ? BIT_readBits(&(seqState->DStream), mlBits) : 0); /* <= 16 bits */
|
seq.matchLength = ML_base[mlCode] + ((mlCode>31) ? BIT_readBits(&(seqState->DStream), mlBits) : 0); /* <= 16 bits */
|
||||||
if (MEM_32bits() && (mlBits+llBits>24)) BIT_reloadDStream(&(seqState->DStream));
|
if (MEM_32bits() && (mlBits+llBits>24)) BIT_reloadDStream(&(seqState->DStream));
|
||||||
|
|
||||||
seq.litLength = LL_base[llCode] + ((llCode>15) ? BIT_readBits(&(seqState->DStream), llBits) : 0); /* <= 16 bits */
|
seq.litLength = LL_base[llCode] + ((llCode>15) ? BIT_readBits(&(seqState->DStream), llBits) : 0); /* <= 16 bits */
|
||||||
if (MEM_32bits() |
|
if (MEM_32bits() ||
|
||||||
(totalBits > 64 - 7 - (LLFSELog+MLFSELog+OffFSELog)) ) BIT_reloadDStream(&(seqState->DStream));
|
(totalBits > 64 - 7 - (LLFSELog+MLFSELog+OffFSELog)) ) BIT_reloadDStream(&(seqState->DStream));
|
||||||
|
|
||||||
/* ANS state update */
|
/* ANS state update */
|
||||||
@@ -930,8 +931,21 @@ size_t ZSTD_decompressBlock(ZSTD_DCtx* dctx,
|
|||||||
void* dst, size_t dstCapacity,
|
void* dst, size_t dstCapacity,
|
||||||
const void* src, size_t srcSize)
|
const void* src, size_t srcSize)
|
||||||
{
|
{
|
||||||
|
size_t dSize;
|
||||||
ZSTD_checkContinuity(dctx, dst);
|
ZSTD_checkContinuity(dctx, dst);
|
||||||
return ZSTD_decompressBlock_internal(dctx, dst, dstCapacity, src, srcSize);
|
dSize = ZSTD_decompressBlock_internal(dctx, dst, dstCapacity, src, srcSize);
|
||||||
|
dctx->previousDstEnd = (char*)dst + dSize;
|
||||||
|
return dSize;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
/** ZSTD_insertBlock() :
|
||||||
|
insert `src` block into `dctx` history. Useful to track uncompressed blocks. */
|
||||||
|
ZSTDLIB_API size_t ZSTD_insertBlock(ZSTD_DCtx* dctx, const void* blockStart, size_t blockSize)
|
||||||
|
{
|
||||||
|
ZSTD_checkContinuity(dctx, blockStart);
|
||||||
|
dctx->previousDstEnd = (const char*)blockStart + blockSize;
|
||||||
|
return blockSize;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -1031,10 +1045,7 @@ size_t ZSTD_decompress_usingDict(ZSTD_DCtx* dctx,
|
|||||||
const void* dict, size_t dictSize)
|
const void* dict, size_t dictSize)
|
||||||
{
|
{
|
||||||
#if defined(ZSTD_LEGACY_SUPPORT) && (ZSTD_LEGACY_SUPPORT==1)
|
#if defined(ZSTD_LEGACY_SUPPORT) && (ZSTD_LEGACY_SUPPORT==1)
|
||||||
{ U32 const magicNumber = MEM_readLE32(src);
|
if (ZSTD_isLegacy(src, srcSize)) return ZSTD_decompressLegacy(dst, dstCapacity, src, srcSize, dict, dictSize);
|
||||||
if (ZSTD_isLegacy(magicNumber))
|
|
||||||
return ZSTD_decompressLegacy(dst, dstCapacity, src, srcSize, dict, dictSize, magicNumber);
|
|
||||||
}
|
|
||||||
#endif
|
#endif
|
||||||
ZSTD_decompressBegin_usingDict(dctx, dict, dictSize);
|
ZSTD_decompressBegin_usingDict(dctx, dict, dictSize);
|
||||||
ZSTD_checkContinuity(dctx, dst);
|
ZSTD_checkContinuity(dctx, dst);
|
||||||
@@ -1273,8 +1284,8 @@ size_t ZSTD_decompressBegin_usingDict(ZSTD_DCtx* dctx, const void* dict, size_t
|
|||||||
|
|
||||||
|
|
||||||
struct ZSTD_DDict_s {
|
struct ZSTD_DDict_s {
|
||||||
void* dictContent;
|
void* dict;
|
||||||
size_t dictContentSize;
|
size_t dictSize;
|
||||||
ZSTD_DCtx* refContext;
|
ZSTD_DCtx* refContext;
|
||||||
}; /* typedef'd tp ZSTD_CDict within zstd.h */
|
}; /* typedef'd tp ZSTD_CDict within zstd.h */
|
||||||
|
|
||||||
@@ -1306,8 +1317,8 @@ ZSTD_DDict* ZSTD_createDDict_advanced(const void* dict, size_t dictSize, ZSTD_cu
|
|||||||
return NULL;
|
return NULL;
|
||||||
} }
|
} }
|
||||||
|
|
||||||
ddict->dictContent = dictContent;
|
ddict->dict = dictContent;
|
||||||
ddict->dictContentSize = dictSize;
|
ddict->dictSize = dictSize;
|
||||||
ddict->refContext = dctx;
|
ddict->refContext = dctx;
|
||||||
return ddict;
|
return ddict;
|
||||||
}
|
}
|
||||||
@@ -1327,7 +1338,7 @@ size_t ZSTD_freeDDict(ZSTD_DDict* ddict)
|
|||||||
ZSTD_freeFunction const cFree = ddict->refContext->customMem.customFree;
|
ZSTD_freeFunction const cFree = ddict->refContext->customMem.customFree;
|
||||||
void* const opaque = ddict->refContext->customMem.opaque;
|
void* const opaque = ddict->refContext->customMem.opaque;
|
||||||
ZSTD_freeDCtx(ddict->refContext);
|
ZSTD_freeDCtx(ddict->refContext);
|
||||||
cFree(opaque, ddict->dictContent);
|
cFree(opaque, ddict->dict);
|
||||||
cFree(opaque, ddict);
|
cFree(opaque, ddict);
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
@@ -1340,6 +1351,9 @@ ZSTDLIB_API size_t ZSTD_decompress_usingDDict(ZSTD_DCtx* dctx,
|
|||||||
const void* src, size_t srcSize,
|
const void* src, size_t srcSize,
|
||||||
const ZSTD_DDict* ddict)
|
const ZSTD_DDict* ddict)
|
||||||
{
|
{
|
||||||
|
#if defined(ZSTD_LEGACY_SUPPORT) && (ZSTD_LEGACY_SUPPORT==1)
|
||||||
|
if (ZSTD_isLegacy(src, srcSize)) return ZSTD_decompressLegacy(dst, dstCapacity, src, srcSize, ddict->dict, ddict->dictSize);
|
||||||
|
#endif
|
||||||
return ZSTD_decompress_usingPreparedDCtx(dctx, ddict->refContext,
|
return ZSTD_decompress_usingPreparedDCtx(dctx, ddict->refContext,
|
||||||
dst, dstCapacity,
|
dst, dstCapacity,
|
||||||
src, srcSize);
|
src, srcSize);
|
||||||
|
|||||||
+33
-27
@@ -31,14 +31,15 @@
|
|||||||
- Zstd homepage : https://www.zstd.net
|
- Zstd homepage : https://www.zstd.net
|
||||||
*/
|
*/
|
||||||
|
|
||||||
|
/*-**************************************
|
||||||
|
* Tuning parameters
|
||||||
|
****************************************/
|
||||||
|
#define ZDICT_MAX_SAMPLES_SIZE (2000U << 20)
|
||||||
|
|
||||||
|
|
||||||
/*-**************************************
|
/*-**************************************
|
||||||
* Compiler Options
|
* Compiler Options
|
||||||
****************************************/
|
****************************************/
|
||||||
/* Disable some Visual warning messages */
|
|
||||||
#ifdef _MSC_VER
|
|
||||||
# pragma warning(disable : 4127) /* disable: C4127: conditional expression is constant */
|
|
||||||
#endif
|
|
||||||
|
|
||||||
/* Unix Large Files support (>4GB) */
|
/* Unix Large Files support (>4GB) */
|
||||||
#define _FILE_OFFSET_BITS 64
|
#define _FILE_OFFSET_BITS 64
|
||||||
#if (defined(__sun__) && (!defined(__LP64__))) /* Sun Solaris 32-bits requires specific definitions */
|
#if (defined(__sun__) && (!defined(__LP64__))) /* Sun Solaris 32-bits requires specific definitions */
|
||||||
@@ -58,13 +59,15 @@
|
|||||||
|
|
||||||
#include "mem.h" /* read */
|
#include "mem.h" /* read */
|
||||||
#include "error_private.h"
|
#include "error_private.h"
|
||||||
#include "fse.h"
|
#include "fse.h" /* FSE_normalizeCount, FSE_writeNCount */
|
||||||
#define HUF_STATIC_LINKING_ONLY
|
#define HUF_STATIC_LINKING_ONLY
|
||||||
#include "huf.h"
|
#include "huf.h"
|
||||||
#include "zstd_internal.h" /* includes zstd.h */
|
#include "zstd_internal.h" /* includes zstd.h */
|
||||||
#include "xxhash.h"
|
#include "xxhash.h"
|
||||||
#include "divsufsort.h"
|
#include "divsufsort.h"
|
||||||
#define ZDICT_STATIC_LINKING_ONLY
|
#ifndef ZDICT_STATIC_LINKING_ONLY
|
||||||
|
# define ZDICT_STATIC_LINKING_ONLY
|
||||||
|
#endif
|
||||||
#include "zdict.h"
|
#include "zdict.h"
|
||||||
|
|
||||||
|
|
||||||
@@ -91,17 +94,19 @@ static const size_t g_min_fast_dictContent = 192;
|
|||||||
/*-*************************************
|
/*-*************************************
|
||||||
* Console display
|
* Console display
|
||||||
***************************************/
|
***************************************/
|
||||||
#define DISPLAY(...) fprintf(stderr, __VA_ARGS__)
|
#define DISPLAY(...) { fprintf(stderr, __VA_ARGS__); fflush( stderr ); }
|
||||||
#define DISPLAYLEVEL(l, ...) if (g_displayLevel>=l) { DISPLAY(__VA_ARGS__); }
|
#define DISPLAYLEVEL(l, ...) if (g_displayLevel>=l) { DISPLAY(__VA_ARGS__); }
|
||||||
static unsigned g_displayLevel = 0; /* 0 : no display; 1: errors; 2: default; 4: full information */
|
static unsigned g_displayLevel = 0; /* 0 : no display; 1: errors; 2: default; 4: full information */
|
||||||
|
|
||||||
#define DISPLAYUPDATE(l, ...) if (g_displayLevel>=l) { \
|
#define DISPLAYUPDATE(l, ...) if (g_displayLevel>=l) { \
|
||||||
if (ZDICT_GetMilliSpan(g_time) > refreshRate) \
|
if (ZDICT_clockSpan(g_time) > refreshRate) \
|
||||||
{ g_time = clock(); DISPLAY(__VA_ARGS__); \
|
{ g_time = clock(); DISPLAY(__VA_ARGS__); \
|
||||||
if (g_displayLevel>=4) fflush(stdout); } }
|
if (g_displayLevel>=4) fflush(stdout); } }
|
||||||
static const unsigned refreshRate = 300;
|
static const clock_t refreshRate = CLOCKS_PER_SEC * 3 / 10;
|
||||||
static clock_t g_time = 0;
|
static clock_t g_time = 0;
|
||||||
|
|
||||||
|
static clock_t ZDICT_clockSpan(clock_t nPrevious) { return clock() - nPrevious; }
|
||||||
|
|
||||||
static void ZDICT_printHex(U32 dlevel, const void* ptr, size_t length)
|
static void ZDICT_printHex(U32 dlevel, const void* ptr, size_t length)
|
||||||
{
|
{
|
||||||
const BYTE* const b = (const BYTE*)ptr;
|
const BYTE* const b = (const BYTE*)ptr;
|
||||||
@@ -117,13 +122,6 @@ static void ZDICT_printHex(U32 dlevel, const void* ptr, size_t length)
|
|||||||
/*-********************************************************
|
/*-********************************************************
|
||||||
* Helper functions
|
* Helper functions
|
||||||
**********************************************************/
|
**********************************************************/
|
||||||
static unsigned ZDICT_GetMilliSpan(clock_t nPrevious)
|
|
||||||
{
|
|
||||||
clock_t nCurrent = clock();
|
|
||||||
unsigned nSpan = (unsigned)(((nCurrent - nPrevious) * 1000) / CLOCKS_PER_SEC);
|
|
||||||
return nSpan;
|
|
||||||
}
|
|
||||||
|
|
||||||
unsigned ZDICT_isError(size_t errorCode) { return ERR_isError(errorCode); }
|
unsigned ZDICT_isError(size_t errorCode) { return ERR_isError(errorCode); }
|
||||||
|
|
||||||
const char* ZDICT_getErrorName(size_t errorCode) { return ERR_getErrorName(errorCode); }
|
const char* ZDICT_getErrorName(size_t errorCode) { return ERR_getErrorName(errorCode); }
|
||||||
@@ -489,7 +487,7 @@ static U32 ZDICT_dictSize(const dictItem* dictList)
|
|||||||
|
|
||||||
|
|
||||||
static size_t ZDICT_trainBuffer(dictItem* dictList, U32 dictListSize,
|
static size_t ZDICT_trainBuffer(dictItem* dictList, U32 dictListSize,
|
||||||
const void* const buffer, const size_t bufferSize, /* buffer must end with noisy guard band */
|
const void* const buffer, size_t bufferSize, /* buffer must end with noisy guard band */
|
||||||
const size_t* fileSizes, unsigned nbFiles,
|
const size_t* fileSizes, unsigned nbFiles,
|
||||||
U32 shiftRatio, unsigned maxDictSize)
|
U32 shiftRatio, unsigned maxDictSize)
|
||||||
{
|
{
|
||||||
@@ -499,7 +497,6 @@ static size_t ZDICT_trainBuffer(dictItem* dictList, U32 dictListSize,
|
|||||||
BYTE* doneMarks = (BYTE*)malloc((bufferSize+16)*sizeof(*doneMarks)); /* +16 for overflow security */
|
BYTE* doneMarks = (BYTE*)malloc((bufferSize+16)*sizeof(*doneMarks)); /* +16 for overflow security */
|
||||||
U32* filePos = (U32*)malloc(nbFiles * sizeof(*filePos));
|
U32* filePos = (U32*)malloc(nbFiles * sizeof(*filePos));
|
||||||
U32 minRatio = nbFiles >> shiftRatio;
|
U32 minRatio = nbFiles >> shiftRatio;
|
||||||
int divSuftSortResult;
|
|
||||||
size_t result = 0;
|
size_t result = 0;
|
||||||
|
|
||||||
/* init */
|
/* init */
|
||||||
@@ -511,15 +508,18 @@ static size_t ZDICT_trainBuffer(dictItem* dictList, U32 dictListSize,
|
|||||||
if (minRatio < MINRATIO) minRatio = MINRATIO;
|
if (minRatio < MINRATIO) minRatio = MINRATIO;
|
||||||
memset(doneMarks, 0, bufferSize+16);
|
memset(doneMarks, 0, bufferSize+16);
|
||||||
|
|
||||||
|
/* limit sample set size (divsufsort limitation)*/
|
||||||
|
if (bufferSize > ZDICT_MAX_SAMPLES_SIZE) DISPLAYLEVEL(3, "sample set too large : reduced to %u MB ...\n", (U32)(ZDICT_MAX_SAMPLES_SIZE>>20));
|
||||||
|
while (bufferSize > ZDICT_MAX_SAMPLES_SIZE) bufferSize -= fileSizes[--nbFiles];
|
||||||
|
|
||||||
/* sort */
|
/* sort */
|
||||||
DISPLAYLEVEL(2, "sorting %u files of total size %u MB ...\n", nbFiles, (U32)(bufferSize>>20));
|
DISPLAYLEVEL(2, "sorting %u files of total size %u MB ...\n", nbFiles, (U32)(bufferSize>>20));
|
||||||
divSuftSortResult = divsufsort((const unsigned char*)buffer, suffix, (int)bufferSize, 0);
|
{ int const divSuftSortResult = divsufsort((const unsigned char*)buffer, suffix, (int)bufferSize, 0);
|
||||||
if (divSuftSortResult != 0) { result = ERROR(GENERIC); goto _cleanup; }
|
if (divSuftSortResult != 0) { result = ERROR(GENERIC); goto _cleanup; } }
|
||||||
suffix[bufferSize] = (int)bufferSize; /* leads into noise */
|
suffix[bufferSize] = (int)bufferSize; /* leads into noise */
|
||||||
suffix0[0] = (int)bufferSize; /* leads into noise */
|
suffix0[0] = (int)bufferSize; /* leads into noise */
|
||||||
{
|
/* build reverse suffix sort */
|
||||||
/* build reverse suffix sort */
|
{ size_t pos;
|
||||||
size_t pos;
|
|
||||||
for (pos=0; pos < bufferSize; pos++)
|
for (pos=0; pos < bufferSize; pos++)
|
||||||
reverseSuffix[suffix[pos]] = (U32)pos;
|
reverseSuffix[suffix[pos]] = (U32)pos;
|
||||||
/* build file pos */
|
/* build file pos */
|
||||||
@@ -587,7 +587,9 @@ static void ZDICT_countEStats(EStats_ress_t esr,
|
|||||||
size_t cSize;
|
size_t cSize;
|
||||||
|
|
||||||
if (srcSize > ZSTD_BLOCKSIZE_MAX) srcSize = ZSTD_BLOCKSIZE_MAX; /* protection vs large samples */
|
if (srcSize > ZSTD_BLOCKSIZE_MAX) srcSize = ZSTD_BLOCKSIZE_MAX; /* protection vs large samples */
|
||||||
ZSTD_copyCCtx(esr.zc, esr.ref);
|
{ size_t const errorCode = ZSTD_copyCCtx(esr.zc, esr.ref);
|
||||||
|
if (ZSTD_isError(errorCode)) { DISPLAYLEVEL(1, "warning : ZSTD_copyCCtx failed \n"); return; }
|
||||||
|
}
|
||||||
cSize = ZSTD_compressBlock(esr.zc, esr.workPlace, ZSTD_BLOCKSIZE_MAX, src, srcSize);
|
cSize = ZSTD_compressBlock(esr.zc, esr.workPlace, ZSTD_BLOCKSIZE_MAX, src, srcSize);
|
||||||
if (ZSTD_isError(cSize)) { DISPLAYLEVEL(1, "warning : could not compress sample size %u \n", (U32)srcSize); return; }
|
if (ZSTD_isError(cSize)) { DISPLAYLEVEL(1, "warning : could not compress sample size %u \n", (U32)srcSize); return; }
|
||||||
|
|
||||||
@@ -709,9 +711,13 @@ static size_t ZDICT_analyzeEntropy(void* dstBuffer, size_t maxDstSize,
|
|||||||
}
|
}
|
||||||
if (compressionLevel==0) compressionLevel=g_compressionLevel_default;
|
if (compressionLevel==0) compressionLevel=g_compressionLevel_default;
|
||||||
params.cParams = ZSTD_getCParams(compressionLevel, averageSampleSize, dictBufferSize);
|
params.cParams = ZSTD_getCParams(compressionLevel, averageSampleSize, dictBufferSize);
|
||||||
params.cParams.strategy = ZSTD_greedy;
|
|
||||||
params.fParams.contentSizeFlag = 0;
|
params.fParams.contentSizeFlag = 0;
|
||||||
ZSTD_compressBegin_advanced(esr.ref, dictBuffer, dictBufferSize, params, 0);
|
{ size_t const beginResult = ZSTD_compressBegin_advanced(esr.ref, dictBuffer, dictBufferSize, params, 0);
|
||||||
|
if (ZSTD_isError(beginResult)) {
|
||||||
|
eSize = ERROR(GENERIC);
|
||||||
|
DISPLAYLEVEL(1, "error : ZSTD_compressBegin_advanced failed ");
|
||||||
|
goto _cleanup;
|
||||||
|
} }
|
||||||
|
|
||||||
/* collect stats on all files */
|
/* collect stats on all files */
|
||||||
for (u=0; u<nbFiles; u++) {
|
for (u=0; u<nbFiles; u++) {
|
||||||
|
|||||||
+35
-10
@@ -54,8 +54,11 @@ extern "C" {
|
|||||||
@return : > 0 if supported by legacy decoder. 0 otherwise.
|
@return : > 0 if supported by legacy decoder. 0 otherwise.
|
||||||
return value is the version.
|
return value is the version.
|
||||||
*/
|
*/
|
||||||
MEM_STATIC unsigned ZSTD_isLegacy (U32 magicNumberLE)
|
MEM_STATIC unsigned ZSTD_isLegacy(const void* src, size_t srcSize)
|
||||||
{
|
{
|
||||||
|
U32 magicNumberLE;
|
||||||
|
if (srcSize<4) return 0;
|
||||||
|
magicNumberLE = MEM_readLE32(src);
|
||||||
switch(magicNumberLE)
|
switch(magicNumberLE)
|
||||||
{
|
{
|
||||||
case ZSTDv01_magicNumberLE:return 1;
|
case ZSTDv01_magicNumberLE:return 1;
|
||||||
@@ -69,23 +72,45 @@ MEM_STATIC unsigned ZSTD_isLegacy (U32 magicNumberLE)
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
MEM_STATIC unsigned long long ZSTD_getDecompressedSize_legacy(const void* src, size_t srcSize)
|
||||||
|
{
|
||||||
|
if (srcSize < 4) return 0;
|
||||||
|
|
||||||
|
{ U32 const version = ZSTD_isLegacy(src, srcSize);
|
||||||
|
if (version < 5) return 0; /* no decompressed size in frame header, or not a legacy format */
|
||||||
|
if (version==5) {
|
||||||
|
ZSTDv05_parameters fParams;
|
||||||
|
size_t const frResult = ZSTDv05_getFrameParams(&fParams, src, srcSize);
|
||||||
|
if (frResult != 0) return 0;
|
||||||
|
return fParams.srcSize;
|
||||||
|
}
|
||||||
|
if (version==6) {
|
||||||
|
ZSTDv06_frameParams fParams;
|
||||||
|
size_t const frResult = ZSTDv06_getFrameParams(&fParams, src, srcSize);
|
||||||
|
if (frResult != 0) return 0;
|
||||||
|
return fParams.frameContentSize;
|
||||||
|
}
|
||||||
|
return 0; /* should not be possible */
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
MEM_STATIC size_t ZSTD_decompressLegacy(
|
MEM_STATIC size_t ZSTD_decompressLegacy(
|
||||||
void* dst, size_t dstCapacity,
|
void* dst, size_t dstCapacity,
|
||||||
const void* src, size_t compressedSize,
|
const void* src, size_t compressedSize,
|
||||||
const void* dict,size_t dictSize,
|
const void* dict,size_t dictSize)
|
||||||
U32 magicNumberLE)
|
|
||||||
{
|
{
|
||||||
switch(magicNumberLE)
|
U32 const version = ZSTD_isLegacy(src, compressedSize);
|
||||||
|
switch(version)
|
||||||
{
|
{
|
||||||
case ZSTDv01_magicNumberLE :
|
case 1 :
|
||||||
return ZSTDv01_decompress(dst, dstCapacity, src, compressedSize);
|
return ZSTDv01_decompress(dst, dstCapacity, src, compressedSize);
|
||||||
case ZSTDv02_magicNumber :
|
case 2 :
|
||||||
return ZSTDv02_decompress(dst, dstCapacity, src, compressedSize);
|
return ZSTDv02_decompress(dst, dstCapacity, src, compressedSize);
|
||||||
case ZSTDv03_magicNumber :
|
case 3 :
|
||||||
return ZSTDv03_decompress(dst, dstCapacity, src, compressedSize);
|
return ZSTDv03_decompress(dst, dstCapacity, src, compressedSize);
|
||||||
case ZSTDv04_magicNumber :
|
case 4 :
|
||||||
return ZSTDv04_decompress(dst, dstCapacity, src, compressedSize);
|
return ZSTDv04_decompress(dst, dstCapacity, src, compressedSize);
|
||||||
case ZSTDv05_MAGICNUMBER :
|
case 5 :
|
||||||
{ size_t result;
|
{ size_t result;
|
||||||
ZSTDv05_DCtx* const zd = ZSTDv05_createDCtx();
|
ZSTDv05_DCtx* const zd = ZSTDv05_createDCtx();
|
||||||
if (zd==NULL) return ERROR(memory_allocation);
|
if (zd==NULL) return ERROR(memory_allocation);
|
||||||
@@ -93,7 +118,7 @@ MEM_STATIC size_t ZSTD_decompressLegacy(
|
|||||||
ZSTDv05_freeDCtx(zd);
|
ZSTDv05_freeDCtx(zd);
|
||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
case ZSTDv06_MAGICNUMBER :
|
case 6 :
|
||||||
{ size_t result;
|
{ size_t result;
|
||||||
ZSTDv06_DCtx* const zd = ZSTDv06_createDCtx();
|
ZSTDv06_DCtx* const zd = ZSTDv06_createDCtx();
|
||||||
if (zd==NULL) return ERROR(memory_allocation);
|
if (zd==NULL) return ERROR(memory_allocation);
|
||||||
|
|||||||
@@ -36,7 +36,7 @@
|
|||||||
#include "zstd_v06.h"
|
#include "zstd_v06.h"
|
||||||
#include <stddef.h> /* size_t, ptrdiff_t */
|
#include <stddef.h> /* size_t, ptrdiff_t */
|
||||||
#include <string.h> /* memcpy */
|
#include <string.h> /* memcpy */
|
||||||
#include <stdlib.h> /* malloc, free, qsort */
|
#include <stdlib.h> /* malloc, free, qsort */
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@@ -535,8 +535,6 @@ ZSTDLIB_API size_t ZSTDv06_decompress_usingPreparedDCtx(
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
struct ZSTDv06_frameParams_s { U64 frameContentSize; U32 windowLog; };
|
|
||||||
|
|
||||||
#define ZSTDv06_FRAMEHEADERSIZE_MAX 13 /* for static allocation */
|
#define ZSTDv06_FRAMEHEADERSIZE_MAX 13 /* for static allocation */
|
||||||
static const size_t ZSTDv06_frameHeaderSize_min = 5;
|
static const size_t ZSTDv06_frameHeaderSize_min = 5;
|
||||||
static const size_t ZSTDv06_frameHeaderSize_max = ZSTDv06_FRAMEHEADERSIZE_MAX;
|
static const size_t ZSTDv06_frameHeaderSize_max = ZSTDv06_FRAMEHEADERSIZE_MAX;
|
||||||
|
|||||||
@@ -107,7 +107,7 @@ ZSTDLIB_API size_t ZSTDv06_decompress_usingDict(ZSTDv06_DCtx* dctx,
|
|||||||
/*-************************
|
/*-************************
|
||||||
* Advanced Streaming API
|
* Advanced Streaming API
|
||||||
***************************/
|
***************************/
|
||||||
|
struct ZSTDv06_frameParams_s { unsigned long long frameContentSize; unsigned windowLog; };
|
||||||
typedef struct ZSTDv06_frameParams_s ZSTDv06_frameParams;
|
typedef struct ZSTDv06_frameParams_s ZSTDv06_frameParams;
|
||||||
|
|
||||||
ZSTDLIB_API size_t ZSTDv06_getFrameParams(ZSTDv06_frameParams* fparamsPtr, const void* src, size_t srcSize); /**< doesn't consume input */
|
ZSTDLIB_API size_t ZSTDv06_getFrameParams(ZSTDv06_frameParams* fparamsPtr, const void* src, size_t srcSize); /**< doesn't consume input */
|
||||||
|
|||||||
@@ -43,6 +43,7 @@ _*
|
|||||||
tmp*
|
tmp*
|
||||||
*.zst
|
*.zst
|
||||||
result
|
result
|
||||||
|
out
|
||||||
|
|
||||||
# fuzzer
|
# fuzzer
|
||||||
afl
|
afl
|
||||||
|
|||||||
+4
-4
@@ -154,10 +154,10 @@ clean:
|
|||||||
@echo Cleaning completed
|
@echo Cleaning completed
|
||||||
|
|
||||||
|
|
||||||
#------------------------------------------------------------------------
|
#---------------------------------------------------------------------------------
|
||||||
#make install is validated only for Linux, OSX, kFreeBSD and Hurd targets
|
#make install is validated only for Linux, OSX, kFreeBSD, Hurd and OpenBSD targets
|
||||||
#------------------------------------------------------------------------
|
#---------------------------------------------------------------------------------
|
||||||
ifneq (,$(filter $(shell uname),Linux Darwin GNU/kFreeBSD GNU))
|
ifneq (,$(filter $(shell uname),Linux Darwin GNU/kFreeBSD GNU OpenBSD))
|
||||||
HOST_OS = POSIX
|
HOST_OS = POSIX
|
||||||
install: zstd
|
install: zstd
|
||||||
@echo Installing binaries
|
@echo Installing binaries
|
||||||
|
|||||||
+10
-11
@@ -148,16 +148,14 @@ static int BMK_benchMem(const void* srcBuffer, size_t srcSize,
|
|||||||
size_t const maxCompressedSize = ZSTD_compressBound(srcSize) + (maxNbBlocks * 1024); /* add some room for safety */
|
size_t const maxCompressedSize = ZSTD_compressBound(srcSize) + (maxNbBlocks * 1024); /* add some room for safety */
|
||||||
void* const compressedBuffer = malloc(maxCompressedSize);
|
void* const compressedBuffer = malloc(maxCompressedSize);
|
||||||
void* const resultBuffer = malloc(srcSize);
|
void* const resultBuffer = malloc(srcSize);
|
||||||
ZSTD_CCtx* refCtx = ZSTD_createCCtx();
|
|
||||||
ZSTD_CCtx* ctx = ZSTD_createCCtx();
|
ZSTD_CCtx* ctx = ZSTD_createCCtx();
|
||||||
ZSTD_DCtx* refDCtx = ZSTD_createDCtx();
|
|
||||||
ZSTD_DCtx* dctx = ZSTD_createDCtx();
|
ZSTD_DCtx* dctx = ZSTD_createDCtx();
|
||||||
U32 nbBlocks;
|
U32 nbBlocks;
|
||||||
UTIL_time_t ticksPerSecond;
|
UTIL_time_t ticksPerSecond;
|
||||||
|
|
||||||
/* checks */
|
/* checks */
|
||||||
if (!compressedBuffer || !resultBuffer || !blockTable || !refCtx || !ctx || !refDCtx || !dctx)
|
if (!compressedBuffer || !resultBuffer || !blockTable || !ctx || !dctx)
|
||||||
EXM_THROW(31, "not enough memory");
|
EXM_THROW(31, "allocation error : not enough memory");
|
||||||
|
|
||||||
/* init */
|
/* init */
|
||||||
if (strlen(displayName)>17) displayName += strlen(displayName)-17; /* can only display 17 characters */
|
if (strlen(displayName)>17) displayName += strlen(displayName)-17; /* can only display 17 characters */
|
||||||
@@ -213,12 +211,15 @@ static int BMK_benchMem(const void* srcBuffer, size_t srcSize,
|
|||||||
DISPLAYLEVEL(2, "%2i-%-17.17s :%10u ->\r", testNb, displayName, (U32)srcSize);
|
DISPLAYLEVEL(2, "%2i-%-17.17s :%10u ->\r", testNb, displayName, (U32)srcSize);
|
||||||
memset(compressedBuffer, 0xE5, maxCompressedSize); /* warm up and erase result buffer */
|
memset(compressedBuffer, 0xE5, maxCompressedSize); /* warm up and erase result buffer */
|
||||||
|
|
||||||
UTIL_sleepMilli(1); /* give processor time to other processes */
|
UTIL_sleepMilli(1); /* give processor time to other processes */
|
||||||
UTIL_waitForNextTick(ticksPerSecond);
|
UTIL_waitForNextTick(ticksPerSecond);
|
||||||
UTIL_getTime(&clockStart);
|
UTIL_getTime(&clockStart);
|
||||||
|
|
||||||
{ U32 nbLoops = 0;
|
{ size_t const refSrcSize = (nbBlocks == 1) ? srcSize : 0;
|
||||||
ZSTD_CDict* cdict = ZSTD_createCDict(dictBuffer, dictBufferSize, cLevel);
|
ZSTD_parameters const zparams = ZSTD_getParams(cLevel, refSrcSize, dictBufferSize);
|
||||||
|
ZSTD_customMem const cmem = { NULL, NULL, NULL };
|
||||||
|
U32 nbLoops = 0;
|
||||||
|
ZSTD_CDict* cdict = ZSTD_createCDict_advanced(dictBuffer, dictBufferSize, zparams, cmem);
|
||||||
if (cdict==NULL) EXM_THROW(1, "ZSTD_createCDict() allocation failure");
|
if (cdict==NULL) EXM_THROW(1, "ZSTD_createCDict() allocation failure");
|
||||||
do {
|
do {
|
||||||
U32 blockNb;
|
U32 blockNb;
|
||||||
@@ -227,7 +228,7 @@ static int BMK_benchMem(const void* srcBuffer, size_t srcSize,
|
|||||||
blockTable[blockNb].cPtr, blockTable[blockNb].cRoom,
|
blockTable[blockNb].cPtr, blockTable[blockNb].cRoom,
|
||||||
blockTable[blockNb].srcPtr,blockTable[blockNb].srcSize,
|
blockTable[blockNb].srcPtr,blockTable[blockNb].srcSize,
|
||||||
cdict);
|
cdict);
|
||||||
if (ZSTD_isError(rSize)) EXM_THROW(1, "ZSTD_compress_usingPreparedCCtx() failed : %s", ZSTD_getErrorName(rSize));
|
if (ZSTD_isError(rSize)) EXM_THROW(1, "ZSTD_compress_usingCDict() failed : %s", ZSTD_getErrorName(rSize));
|
||||||
blockTable[blockNb].cSize = rSize;
|
blockTable[blockNb].cSize = rSize;
|
||||||
}
|
}
|
||||||
nbLoops++;
|
nbLoops++;
|
||||||
@@ -264,7 +265,7 @@ static int BMK_benchMem(const void* srcBuffer, size_t srcSize,
|
|||||||
blockTable[blockNb].cPtr, blockTable[blockNb].cSize,
|
blockTable[blockNb].cPtr, blockTable[blockNb].cSize,
|
||||||
ddict);
|
ddict);
|
||||||
if (ZSTD_isError(regenSize)) {
|
if (ZSTD_isError(regenSize)) {
|
||||||
DISPLAY("ZSTD_decompress_usingPreparedDCtx() failed on block %u : %s \n",
|
DISPLAY("ZSTD_decompress_usingDDict() failed on block %u : %s \n",
|
||||||
blockNb, ZSTD_getErrorName(regenSize));
|
blockNb, ZSTD_getErrorName(regenSize));
|
||||||
clockLoop = 0; /* force immediate test end */
|
clockLoop = 0; /* force immediate test end */
|
||||||
break;
|
break;
|
||||||
@@ -321,9 +322,7 @@ static int BMK_benchMem(const void* srcBuffer, size_t srcSize,
|
|||||||
free(blockTable);
|
free(blockTable);
|
||||||
free(compressedBuffer);
|
free(compressedBuffer);
|
||||||
free(resultBuffer);
|
free(resultBuffer);
|
||||||
ZSTD_freeCCtx(refCtx);
|
|
||||||
ZSTD_freeCCtx(ctx);
|
ZSTD_freeCCtx(ctx);
|
||||||
ZSTD_freeDCtx(refDCtx);
|
|
||||||
ZSTD_freeDCtx(dctx);
|
ZSTD_freeDCtx(dctx);
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
+11
-5
@@ -23,12 +23,19 @@
|
|||||||
- source repository : https://github.com/Cyan4973/zstd
|
- source repository : https://github.com/Cyan4973/zstd
|
||||||
*/
|
*/
|
||||||
|
|
||||||
|
/* *************************************
|
||||||
|
* Compiler Options
|
||||||
|
***************************************/
|
||||||
|
#define _CRT_SECURE_NO_WARNINGS /* removes Visual warning on strerror() */
|
||||||
|
|
||||||
|
|
||||||
/*-************************************
|
/*-************************************
|
||||||
* Includes
|
* Includes
|
||||||
**************************************/
|
**************************************/
|
||||||
#include <stdlib.h> /* malloc */
|
#include <stdlib.h> /* malloc */
|
||||||
#include <stdio.h> /* FILE, fwrite, fprintf */
|
#include <stdio.h> /* FILE, fwrite, fprintf */
|
||||||
#include <string.h> /* memcpy */
|
#include <string.h> /* memcpy */
|
||||||
|
#include <errno.h> /* errno */
|
||||||
#include "mem.h" /* U32 */
|
#include "mem.h" /* U32 */
|
||||||
|
|
||||||
|
|
||||||
@@ -104,7 +111,7 @@ static BYTE RDG_genChar(U32* seed, const BYTE* ldt)
|
|||||||
U32 const id = RDG_rand(seed) & LTMASK;
|
U32 const id = RDG_rand(seed) & LTMASK;
|
||||||
//TRACE(" %u : \n", id);
|
//TRACE(" %u : \n", id);
|
||||||
//TRACE(" %4u [%4u] ; val : %4u \n", id, id&255, ldt[id]);
|
//TRACE(" %4u [%4u] ; val : %4u \n", id, id&255, ldt[id]);
|
||||||
return (ldt[id]); /* memory-sanitizer fails here, stating "uninitialized value" when table initialized with 0.0. Checked : table is fully initialized */
|
return ldt[id]; /* memory-sanitizer fails here, stating "uninitialized value" when table initialized with P==0.0. Checked : table is fully initialized */
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -115,8 +122,7 @@ static U32 RDG_rand15Bits (unsigned* seedPtr)
|
|||||||
|
|
||||||
static U32 RDG_randLength(unsigned* seedPtr)
|
static U32 RDG_randLength(unsigned* seedPtr)
|
||||||
{
|
{
|
||||||
if (RDG_rand(seedPtr) & 7)
|
if (RDG_rand(seedPtr) & 7) return (RDG_rand(seedPtr) & 0xF); /* small length */
|
||||||
return (RDG_rand(seedPtr) & 0xF);
|
|
||||||
return (RDG_rand(seedPtr) & 0x1FF) + 0xF;
|
return (RDG_rand(seedPtr) & 0x1FF) + 0xF;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -185,10 +191,10 @@ void RDG_genStdout(unsigned long long size, double matchProba, double litProba,
|
|||||||
size_t const stdDictSize = 32 KB;
|
size_t const stdDictSize = 32 KB;
|
||||||
BYTE* const buff = (BYTE*)malloc(stdDictSize + stdBlockSize);
|
BYTE* const buff = (BYTE*)malloc(stdDictSize + stdBlockSize);
|
||||||
U64 total = 0;
|
U64 total = 0;
|
||||||
BYTE ldt[LTSIZE];
|
BYTE ldt[LTSIZE]; /* literals distribution table */
|
||||||
|
|
||||||
/* init */
|
/* init */
|
||||||
if (buff==NULL) { fprintf(stdout, "not enough memory\n"); exit(1); }
|
if (buff==NULL) { fprintf(stderr, "datagen: error: %s \n", strerror(errno)); exit(1); }
|
||||||
if (litProba<=0.0) litProba = matchProba / 4.5;
|
if (litProba<=0.0) litProba = matchProba / 4.5;
|
||||||
memset(ldt, '0', sizeof(ldt));
|
memset(ldt, '0', sizeof(ldt));
|
||||||
RDG_fillLiteralDistrib(ldt, litProba);
|
RDG_fillLiteralDistrib(ldt, litProba);
|
||||||
|
|||||||
+30
-31
@@ -30,6 +30,7 @@
|
|||||||
#include <string.h> /* memset */
|
#include <string.h> /* memset */
|
||||||
#include <stdio.h> /* fprintf, fopen, ftello64 */
|
#include <stdio.h> /* fprintf, fopen, ftello64 */
|
||||||
#include <time.h> /* clock_t, clock, CLOCKS_PER_SEC */
|
#include <time.h> /* clock_t, clock, CLOCKS_PER_SEC */
|
||||||
|
#include <errno.h> /* errno */
|
||||||
|
|
||||||
#include "mem.h" /* read */
|
#include "mem.h" /* read */
|
||||||
#include "error_private.h"
|
#include "error_private.h"
|
||||||
@@ -43,13 +44,10 @@
|
|||||||
#define MB *(1 <<20)
|
#define MB *(1 <<20)
|
||||||
#define GB *(1U<<30)
|
#define GB *(1U<<30)
|
||||||
|
|
||||||
#define DICTLISTSIZE 10000
|
|
||||||
#define MEMMULT 11
|
#define MEMMULT 11
|
||||||
static const size_t maxMemory = (sizeof(size_t) == 4) ? (2 GB - 64 MB) : ((size_t)(512 MB) << sizeof(size_t));
|
static const size_t maxMemory = (sizeof(size_t) == 4) ? (2 GB - 64 MB) : ((size_t)(512 MB) << sizeof(size_t));
|
||||||
|
|
||||||
#define NOISELENGTH 32
|
#define NOISELENGTH 32
|
||||||
#define PRIME1 2654435761U
|
|
||||||
#define PRIME2 2246822519U
|
|
||||||
|
|
||||||
|
|
||||||
/*-*************************************
|
/*-*************************************
|
||||||
@@ -60,17 +58,13 @@ static const size_t maxMemory = (sizeof(size_t) == 4) ? (2 GB - 64 MB) : ((size_
|
|||||||
static unsigned g_displayLevel = 0; /* 0 : no display; 1: errors; 2: default; 4: full information */
|
static unsigned g_displayLevel = 0; /* 0 : no display; 1: errors; 2: default; 4: full information */
|
||||||
|
|
||||||
#define DISPLAYUPDATE(l, ...) if (g_displayLevel>=l) { \
|
#define DISPLAYUPDATE(l, ...) if (g_displayLevel>=l) { \
|
||||||
if ((DIB_GetMilliSpan(g_time) > refreshRate) || (g_displayLevel>=4)) \
|
if ((DIB_clockSpan(g_time) > refreshRate) || (g_displayLevel>=4)) \
|
||||||
{ g_time = clock(); DISPLAY(__VA_ARGS__); \
|
{ g_time = clock(); DISPLAY(__VA_ARGS__); \
|
||||||
if (g_displayLevel>=4) fflush(stdout); } }
|
if (g_displayLevel>=4) fflush(stdout); } }
|
||||||
static const unsigned refreshRate = 150;
|
static const clock_t refreshRate = CLOCKS_PER_SEC * 2 / 10;
|
||||||
static clock_t g_time = 0;
|
static clock_t g_time = 0;
|
||||||
|
|
||||||
static unsigned DIB_GetMilliSpan(clock_t nPrevious)
|
static clock_t DIB_clockSpan(clock_t nPrevious) { return clock() - nPrevious; }
|
||||||
{
|
|
||||||
clock_t const nCurrent = clock();
|
|
||||||
return (unsigned)(((nCurrent - nPrevious) * 1000) / CLOCKS_PER_SEC);
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
/*-*************************************
|
/*-*************************************
|
||||||
@@ -97,13 +91,15 @@ unsigned DiB_isError(size_t errorCode) { return ERR_isError(errorCode); }
|
|||||||
|
|
||||||
const char* DiB_getErrorName(size_t errorCode) { return ERR_getErrorName(errorCode); }
|
const char* DiB_getErrorName(size_t errorCode) { return ERR_getErrorName(errorCode); }
|
||||||
|
|
||||||
|
#define MIN(a,b) ( (a) < (b) ? (a) : (b) )
|
||||||
|
|
||||||
|
|
||||||
/* ********************************************************
|
/* ********************************************************
|
||||||
* File related operations
|
* File related operations
|
||||||
**********************************************************/
|
**********************************************************/
|
||||||
/** DiB_loadFiles() :
|
/** DiB_loadFiles() :
|
||||||
* @return : nb of files effectively loaded into `buffer` */
|
* @return : nb of files effectively loaded into `buffer` */
|
||||||
static unsigned DiB_loadFiles(void* buffer, size_t bufferSize,
|
static unsigned DiB_loadFiles(void* buffer, size_t* bufferSizePtr,
|
||||||
size_t* fileSizes,
|
size_t* fileSizes,
|
||||||
const char** fileNamesTable, unsigned nbFiles)
|
const char** fileNamesTable, unsigned nbFiles)
|
||||||
{
|
{
|
||||||
@@ -112,18 +108,20 @@ static unsigned DiB_loadFiles(void* buffer, size_t bufferSize,
|
|||||||
unsigned n;
|
unsigned n;
|
||||||
|
|
||||||
for (n=0; n<nbFiles; n++) {
|
for (n=0; n<nbFiles; n++) {
|
||||||
unsigned long long const fs64 = UTIL_getFileSize(fileNamesTable[n]);
|
const char* const fileName = fileNamesTable[n];
|
||||||
size_t const fileSize = (size_t)(fs64 > bufferSize-pos ? 0 : fs64);
|
unsigned long long const fs64 = UTIL_getFileSize(fileName);
|
||||||
FILE* const f = fopen(fileNamesTable[n], "rb");
|
size_t const fileSize = (size_t) MIN(fs64, 128 KB);
|
||||||
if (f==NULL) EXM_THROW(10, "impossible to open file %s", fileNamesTable[n]);
|
if (fileSize > *bufferSizePtr-pos) break;
|
||||||
DISPLAYUPDATE(2, "Loading %s... \r", fileNamesTable[n]);
|
{ FILE* const f = fopen(fileName, "rb");
|
||||||
{ size_t const readSize = fread(buff+pos, 1, fileSize, f);
|
if (f==NULL) EXM_THROW(10, "zstd: dictBuilder: %s %s ", fileName, strerror(errno));
|
||||||
if (readSize != fileSize) EXM_THROW(11, "could not read %s", fileNamesTable[n]);
|
DISPLAYUPDATE(2, "Loading %s... \r", fileName);
|
||||||
pos += readSize; }
|
{ size_t const readSize = fread(buff+pos, 1, fileSize, f);
|
||||||
fileSizes[n] = fileSize;
|
if (readSize != fileSize) EXM_THROW(11, "Pb reading %s", fileName);
|
||||||
fclose(f);
|
pos += readSize; }
|
||||||
if (fileSize == 0) break; /* stop there, not enough memory to load all files */
|
fileSizes[n] = fileSize;
|
||||||
}
|
fclose(f);
|
||||||
|
} }
|
||||||
|
*bufferSizePtr = pos;
|
||||||
return n;
|
return n;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -137,26 +135,28 @@ static size_t DiB_findMaxMem(unsigned long long requiredMem)
|
|||||||
void* testmem = NULL;
|
void* testmem = NULL;
|
||||||
|
|
||||||
requiredMem = (((requiredMem >> 23) + 1) << 23);
|
requiredMem = (((requiredMem >> 23) + 1) << 23);
|
||||||
requiredMem += 2 * step;
|
requiredMem += step;
|
||||||
if (requiredMem > maxMemory) requiredMem = maxMemory;
|
if (requiredMem > maxMemory) requiredMem = maxMemory;
|
||||||
|
|
||||||
while (!testmem) {
|
while (!testmem) {
|
||||||
requiredMem -= step;
|
|
||||||
testmem = malloc((size_t)requiredMem);
|
testmem = malloc((size_t)requiredMem);
|
||||||
|
requiredMem -= step;
|
||||||
}
|
}
|
||||||
|
|
||||||
free(testmem);
|
free(testmem);
|
||||||
return (size_t)(requiredMem - step);
|
return (size_t)requiredMem;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
static void DiB_fillNoise(void* buffer, size_t length)
|
static void DiB_fillNoise(void* buffer, size_t length)
|
||||||
{
|
{
|
||||||
unsigned acc = PRIME1;
|
unsigned const prime1 = 2654435761U;
|
||||||
|
unsigned const prime2 = 2246822519U;
|
||||||
|
unsigned acc = prime1;
|
||||||
size_t p=0;;
|
size_t p=0;;
|
||||||
|
|
||||||
for (p=0; p<length; p++) {
|
for (p=0; p<length; p++) {
|
||||||
acc *= PRIME2;
|
acc *= prime2;
|
||||||
((unsigned char*)buffer)[p] = (unsigned char)(acc >> 21);
|
((unsigned char*)buffer)[p] = (unsigned char)(acc >> 21);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -188,7 +188,6 @@ size_t ZDICT_trainFromBuffer_unsafe(void* dictBuffer, size_t dictBufferCapacity,
|
|||||||
ZDICT_params_t parameters);
|
ZDICT_params_t parameters);
|
||||||
|
|
||||||
|
|
||||||
#define MIN(a,b) ((a)<(b)?(a):(b))
|
|
||||||
int DiB_trainFromFiles(const char* dictFileName, unsigned maxDictSize,
|
int DiB_trainFromFiles(const char* dictFileName, unsigned maxDictSize,
|
||||||
const char** fileNamesTable, unsigned nbFiles,
|
const char** fileNamesTable, unsigned nbFiles,
|
||||||
ZDICT_params_t params)
|
ZDICT_params_t params)
|
||||||
@@ -197,7 +196,7 @@ int DiB_trainFromFiles(const char* dictFileName, unsigned maxDictSize,
|
|||||||
size_t* const fileSizes = (size_t*)malloc(nbFiles * sizeof(size_t));
|
size_t* const fileSizes = (size_t*)malloc(nbFiles * sizeof(size_t));
|
||||||
unsigned long long const totalSizeToLoad = UTIL_getTotalFileSize(fileNamesTable, nbFiles);
|
unsigned long long const totalSizeToLoad = UTIL_getTotalFileSize(fileNamesTable, nbFiles);
|
||||||
size_t const maxMem = DiB_findMaxMem(totalSizeToLoad * MEMMULT) / MEMMULT;
|
size_t const maxMem = DiB_findMaxMem(totalSizeToLoad * MEMMULT) / MEMMULT;
|
||||||
size_t const benchedSize = MIN (maxMem, (size_t)totalSizeToLoad);
|
size_t benchedSize = MIN (maxMem, (size_t)totalSizeToLoad);
|
||||||
void* const srcBuffer = malloc(benchedSize+NOISELENGTH);
|
void* const srcBuffer = malloc(benchedSize+NOISELENGTH);
|
||||||
int result = 0;
|
int result = 0;
|
||||||
|
|
||||||
@@ -210,7 +209,7 @@ int DiB_trainFromFiles(const char* dictFileName, unsigned maxDictSize,
|
|||||||
DISPLAYLEVEL(1, "Not enough memory; training on %u MB only...\n", (unsigned)(benchedSize >> 20));
|
DISPLAYLEVEL(1, "Not enough memory; training on %u MB only...\n", (unsigned)(benchedSize >> 20));
|
||||||
|
|
||||||
/* Load input buffer */
|
/* Load input buffer */
|
||||||
nbFiles = DiB_loadFiles(srcBuffer, benchedSize, fileSizes, fileNamesTable, nbFiles);
|
nbFiles = DiB_loadFiles(srcBuffer, &benchedSize, fileSizes, fileNamesTable, nbFiles);
|
||||||
DiB_fillNoise((char*)srcBuffer + benchedSize, NOISELENGTH); /* guard band, for end of buffer condition */
|
DiB_fillNoise((char*)srcBuffer + benchedSize, NOISELENGTH); /* guard band, for end of buffer condition */
|
||||||
|
|
||||||
{ size_t const dictSize = ZDICT_trainFromBuffer_unsafe(dictBuffer, maxDictSize,
|
{ size_t const dictSize = ZDICT_trainFromBuffer_unsafe(dictBuffer, maxDictSize,
|
||||||
|
|||||||
+34
-34
@@ -41,13 +41,14 @@
|
|||||||
/* *************************************
|
/* *************************************
|
||||||
* Compiler Options
|
* Compiler Options
|
||||||
***************************************/
|
***************************************/
|
||||||
#define _POSIX_SOURCE 1 /* enable %llu on Windows */
|
#define _POSIX_SOURCE 1 /* enable %llu on Windows */
|
||||||
|
#define _CRT_SECURE_NO_WARNINGS /* removes Visual warning on strerror() */
|
||||||
|
|
||||||
|
|
||||||
/*-*************************************
|
/*-*************************************
|
||||||
* Includes
|
* Includes
|
||||||
***************************************/
|
***************************************/
|
||||||
#include "util.h" /* Compiler options, UTIL_GetFileSize */
|
#include "util.h" /* Compiler options, UTIL_GetFileSize, _LARGEFILE64_SOURCE */
|
||||||
#include <stdio.h> /* fprintf, fopen, fread, _fileno, stdin, stdout */
|
#include <stdio.h> /* fprintf, fopen, fread, _fileno, stdin, stdout */
|
||||||
#include <stdlib.h> /* malloc, free */
|
#include <stdlib.h> /* malloc, free */
|
||||||
#include <string.h> /* strcmp, strlen */
|
#include <string.h> /* strcmp, strlen */
|
||||||
@@ -58,7 +59,6 @@
|
|||||||
#include "fileio.h"
|
#include "fileio.h"
|
||||||
#define ZSTD_STATIC_LINKING_ONLY /* ZSTD_magicNumber, ZSTD_frameHeaderSize_max */
|
#define ZSTD_STATIC_LINKING_ONLY /* ZSTD_magicNumber, ZSTD_frameHeaderSize_max */
|
||||||
#include "zstd.h"
|
#include "zstd.h"
|
||||||
#include "zstd_internal.h" /* MIN, KB, MB */
|
|
||||||
#define ZBUFF_STATIC_LINKING_ONLY
|
#define ZBUFF_STATIC_LINKING_ONLY
|
||||||
#include "zbuff.h"
|
#include "zbuff.h"
|
||||||
|
|
||||||
@@ -84,6 +84,10 @@
|
|||||||
/*-*************************************
|
/*-*************************************
|
||||||
* Constants
|
* Constants
|
||||||
***************************************/
|
***************************************/
|
||||||
|
#define KB *(1<<10)
|
||||||
|
#define MB *(1<<20)
|
||||||
|
#define GB *(1U<<30)
|
||||||
|
|
||||||
#define _1BIT 0x01
|
#define _1BIT 0x01
|
||||||
#define _2BITS 0x03
|
#define _2BITS 0x03
|
||||||
#define _3BITS 0x07
|
#define _3BITS 0x07
|
||||||
@@ -113,21 +117,17 @@ static U32 g_displayLevel = 2; /* 0 : no display; 1: errors; 2 : + result
|
|||||||
void FIO_setNotificationLevel(unsigned level) { g_displayLevel=level; }
|
void FIO_setNotificationLevel(unsigned level) { g_displayLevel=level; }
|
||||||
|
|
||||||
#define DISPLAYUPDATE(l, ...) if (g_displayLevel>=l) { \
|
#define DISPLAYUPDATE(l, ...) if (g_displayLevel>=l) { \
|
||||||
if ((FIO_GetMilliSpan(g_time) > refreshRate) || (g_displayLevel>=4)) \
|
if ((clock() - g_time > refreshRate) || (g_displayLevel>=4)) \
|
||||||
{ g_time = clock(); DISPLAY(__VA_ARGS__); \
|
{ g_time = clock(); DISPLAY(__VA_ARGS__); \
|
||||||
if (g_displayLevel>=4) fflush(stdout); } }
|
if (g_displayLevel>=4) fflush(stdout); } }
|
||||||
static const unsigned refreshRate = 150;
|
static const clock_t refreshRate = CLOCKS_PER_SEC * 15 / 100;
|
||||||
static clock_t g_time = 0;
|
static clock_t g_time = 0;
|
||||||
|
|
||||||
static unsigned FIO_GetMilliSpan(clock_t nPrevious)
|
#define MIN(a,b) ((a) < (b) ? (a) : (b))
|
||||||
{
|
|
||||||
clock_t const nCurrent = clock();
|
|
||||||
return (unsigned)(((nCurrent - nPrevious) * 1000) / CLOCKS_PER_SEC);
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
/*-*************************************
|
/*-*************************************
|
||||||
* Local Parameters
|
* Local Parameters - Not thread safe
|
||||||
***************************************/
|
***************************************/
|
||||||
static U32 g_overwrite = 0;
|
static U32 g_overwrite = 0;
|
||||||
void FIO_overwriteMode(void) { g_overwrite=1; }
|
void FIO_overwriteMode(void) { g_overwrite=1; }
|
||||||
@@ -175,7 +175,7 @@ static FILE* FIO_openSrcFile(const char* srcFileName)
|
|||||||
f = fopen(srcFileName, "rb");
|
f = fopen(srcFileName, "rb");
|
||||||
}
|
}
|
||||||
|
|
||||||
if ( f==NULL ) DISPLAYLEVEL(1, "zstd: %s: No such file\n", srcFileName);
|
if ( f==NULL ) DISPLAYLEVEL(1, "zstd: %s: %s \n", srcFileName, strerror(errno));
|
||||||
|
|
||||||
return f;
|
return f;
|
||||||
}
|
}
|
||||||
@@ -201,18 +201,20 @@ static FILE* FIO_openDstFile(const char* dstFileName)
|
|||||||
if (g_displayLevel <= 1) {
|
if (g_displayLevel <= 1) {
|
||||||
/* No interaction possible */
|
/* No interaction possible */
|
||||||
DISPLAY("zstd: %s already exists; not overwritten \n", dstFileName);
|
DISPLAY("zstd: %s already exists; not overwritten \n", dstFileName);
|
||||||
return 0;
|
return NULL;
|
||||||
}
|
}
|
||||||
DISPLAY("zstd: %s already exists; do you wish to overwrite (y/N) ? ", dstFileName);
|
DISPLAY("zstd: %s already exists; do you wish to overwrite (y/N) ? ", dstFileName);
|
||||||
{ int ch = getchar();
|
{ int ch = getchar();
|
||||||
if ((ch!='Y') && (ch!='y')) {
|
if ((ch!='Y') && (ch!='y')) {
|
||||||
DISPLAY(" not overwritten \n");
|
DISPLAY(" not overwritten \n");
|
||||||
return 0;
|
return NULL;
|
||||||
}
|
}
|
||||||
while ((ch!=EOF) && (ch!='\n')) ch = getchar(); /* flush rest of input line */
|
while ((ch!=EOF) && (ch!='\n')) ch = getchar(); /* flush rest of input line */
|
||||||
} } }
|
} } }
|
||||||
f = fopen( dstFileName, "wb" );
|
f = fopen( dstFileName, "wb" );
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (f==NULL) DISPLAYLEVEL(1, "zstd: %s: %s\n", dstFileName, strerror(errno));
|
||||||
return f;
|
return f;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -233,18 +235,18 @@ static size_t FIO_loadFile(void** bufferPtr, const char* fileName)
|
|||||||
|
|
||||||
DISPLAYLEVEL(4,"Loading %s as dictionary \n", fileName);
|
DISPLAYLEVEL(4,"Loading %s as dictionary \n", fileName);
|
||||||
fileHandle = fopen(fileName, "rb");
|
fileHandle = fopen(fileName, "rb");
|
||||||
if (fileHandle==0) EXM_THROW(31, "Error opening file %s", fileName);
|
if (fileHandle==0) EXM_THROW(31, "zstd: %s: %s", fileName, strerror(errno));
|
||||||
fileSize = UTIL_getFileSize(fileName);
|
fileSize = UTIL_getFileSize(fileName);
|
||||||
if (fileSize > MAX_DICT_SIZE) {
|
if (fileSize > MAX_DICT_SIZE) {
|
||||||
int seekResult;
|
int seekResult;
|
||||||
if (fileSize > 1 GB) EXM_THROW(32, "Dictionary file %s is too large", fileName); /* avoid extreme cases */
|
if (fileSize > 1 GB) EXM_THROW(32, "Dictionary file %s is too large", fileName); /* avoid extreme cases */
|
||||||
DISPLAYLEVEL(2,"Dictionary %s is too large : using last %u bytes only \n", fileName, MAX_DICT_SIZE);
|
DISPLAYLEVEL(2,"Dictionary %s is too large : using last %u bytes only \n", fileName, MAX_DICT_SIZE);
|
||||||
seekResult = fseek(fileHandle, (long int)(fileSize-MAX_DICT_SIZE), SEEK_SET); /* use end of file */
|
seekResult = fseek(fileHandle, (long int)(fileSize-MAX_DICT_SIZE), SEEK_SET); /* use end of file */
|
||||||
if (seekResult != 0) EXM_THROW(33, "Error seeking into file %s", fileName);
|
if (seekResult != 0) EXM_THROW(33, "zstd: %s: %s", fileName, strerror(errno));
|
||||||
fileSize = MAX_DICT_SIZE;
|
fileSize = MAX_DICT_SIZE;
|
||||||
}
|
}
|
||||||
*bufferPtr = (BYTE*)malloc((size_t)fileSize);
|
*bufferPtr = malloc((size_t)fileSize);
|
||||||
if (*bufferPtr==NULL) EXM_THROW(34, "Allocation error : not enough memory for dictBuffer");
|
if (*bufferPtr==NULL) EXM_THROW(34, "zstd: %s", strerror(errno));
|
||||||
{ size_t const readSize = fread(*bufferPtr, 1, (size_t)fileSize, fileHandle);
|
{ size_t const readSize = fread(*bufferPtr, 1, (size_t)fileSize, fileHandle);
|
||||||
if (readSize!=fileSize) EXM_THROW(35, "Error reading dictionary file %s", fileName); }
|
if (readSize!=fileSize) EXM_THROW(35, "Error reading dictionary file %s", fileName); }
|
||||||
fclose(fileHandle);
|
fclose(fileHandle);
|
||||||
@@ -273,14 +275,14 @@ static cRess_t FIO_createCResources(const char* dictFileName)
|
|||||||
cRess_t ress;
|
cRess_t ress;
|
||||||
|
|
||||||
ress.ctx = ZBUFF_createCCtx();
|
ress.ctx = ZBUFF_createCCtx();
|
||||||
if (ress.ctx == NULL) EXM_THROW(30, "Allocation error : can't create ZBUFF context");
|
if (ress.ctx == NULL) EXM_THROW(30, "zstd: allocation error : can't create ZBUFF context");
|
||||||
|
|
||||||
/* Allocate Memory */
|
/* Allocate Memory */
|
||||||
ress.srcBufferSize = ZBUFF_recommendedCInSize();
|
ress.srcBufferSize = ZBUFF_recommendedCInSize();
|
||||||
ress.srcBuffer = malloc(ress.srcBufferSize);
|
ress.srcBuffer = malloc(ress.srcBufferSize);
|
||||||
ress.dstBufferSize = ZBUFF_recommendedCOutSize();
|
ress.dstBufferSize = ZBUFF_recommendedCOutSize();
|
||||||
ress.dstBuffer = malloc(ress.dstBufferSize);
|
ress.dstBuffer = malloc(ress.dstBufferSize);
|
||||||
if (!ress.srcBuffer || !ress.dstBuffer) EXM_THROW(31, "Allocation error : not enough memory");
|
if (!ress.srcBuffer || !ress.dstBuffer) EXM_THROW(31, "zstd: allocation error : not enough memory");
|
||||||
|
|
||||||
/* dictionary */
|
/* dictionary */
|
||||||
ress.dictBufferSize = FIO_loadFile(&(ress.dictBuffer), dictFileName);
|
ress.dictBufferSize = FIO_loadFile(&(ress.dictBuffer), dictFileName);
|
||||||
@@ -295,7 +297,7 @@ static void FIO_freeCResources(cRess_t ress)
|
|||||||
free(ress.dstBuffer);
|
free(ress.dstBuffer);
|
||||||
free(ress.dictBuffer);
|
free(ress.dictBuffer);
|
||||||
errorCode = ZBUFF_freeCCtx(ress.ctx);
|
errorCode = ZBUFF_freeCCtx(ress.ctx);
|
||||||
if (ZBUFF_isError(errorCode)) EXM_THROW(38, "Error : can't release ZBUFF context resource : %s", ZBUFF_getErrorName(errorCode));
|
if (ZBUFF_isError(errorCode)) EXM_THROW(38, "zstd: error : can't release ZBUFF context resource : %s", ZBUFF_getErrorName(errorCode));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -315,9 +317,7 @@ static int FIO_compressFilename_internal(cRess_t ress,
|
|||||||
U64 const fileSize = UTIL_getFileSize(srcFileName);
|
U64 const fileSize = UTIL_getFileSize(srcFileName);
|
||||||
|
|
||||||
/* init */
|
/* init */
|
||||||
{ ZSTD_parameters params;
|
{ ZSTD_parameters params = ZSTD_getParams(cLevel, fileSize, ress.dictBufferSize);
|
||||||
memset(¶ms, 0, sizeof(params));
|
|
||||||
params.cParams = ZSTD_getCParams(cLevel, fileSize, ress.dictBufferSize);
|
|
||||||
params.fParams.contentSizeFlag = 1;
|
params.fParams.contentSizeFlag = 1;
|
||||||
params.fParams.checksumFlag = g_checksumFlag;
|
params.fParams.checksumFlag = g_checksumFlag;
|
||||||
params.fParams.noDictIDFlag = !g_dictIDFlag;
|
params.fParams.noDictIDFlag = !g_dictIDFlag;
|
||||||
@@ -375,8 +375,8 @@ static int FIO_compressFilename_internal(cRess_t ress,
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
/*! FIO_compressFilename_internal() :
|
/*! FIO_compressFilename_srcFile() :
|
||||||
* same as FIO_compressFilename_extRess(), with ress.destFile already opened (typically stdout)
|
* note : ress.destFile already opened
|
||||||
* @return : 0 : compression completed correctly,
|
* @return : 0 : compression completed correctly,
|
||||||
* 1 : missing or pb opening srcFileName
|
* 1 : missing or pb opening srcFileName
|
||||||
*/
|
*/
|
||||||
@@ -417,7 +417,7 @@ static int FIO_compressFilename_dstFile(cRess_t ress,
|
|||||||
|
|
||||||
result = FIO_compressFilename_srcFile(ress, dstFileName, srcFileName, cLevel);
|
result = FIO_compressFilename_srcFile(ress, dstFileName, srcFileName, cLevel);
|
||||||
|
|
||||||
if (fclose(ress.dstFile)) EXM_THROW(28, "Write error : cannot properly close %s", dstFileName);
|
if (fclose(ress.dstFile)) { DISPLAYLEVEL(1, "zstd: %s: %s \n", dstFileName, strerror(errno)); result=1; }
|
||||||
if (result!=0) remove(dstFileName); /* remove operation artefact */
|
if (result!=0) remove(dstFileName); /* remove operation artefact */
|
||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
@@ -429,13 +429,13 @@ int FIO_compressFilename(const char* dstFileName, const char* srcFileName,
|
|||||||
clock_t const start = clock();
|
clock_t const start = clock();
|
||||||
|
|
||||||
cRess_t const ress = FIO_createCResources(dictFileName);
|
cRess_t const ress = FIO_createCResources(dictFileName);
|
||||||
int const issueWithSrcFile = FIO_compressFilename_dstFile(ress, dstFileName, srcFileName, compressionLevel);
|
int const result = FIO_compressFilename_dstFile(ress, dstFileName, srcFileName, compressionLevel);
|
||||||
FIO_freeCResources(ress);
|
|
||||||
|
|
||||||
{ double const seconds = (double)(clock() - start) / CLOCKS_PER_SEC;
|
double const seconds = (double)(clock() - start) / CLOCKS_PER_SEC;
|
||||||
DISPLAYLEVEL(4, "Completed in %.2f sec \n", seconds);
|
DISPLAYLEVEL(4, "Completed in %.2f sec \n", seconds);
|
||||||
}
|
|
||||||
return issueWithSrcFile;
|
FIO_freeCResources(ress);
|
||||||
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -700,7 +700,7 @@ static int FIO_decompressSrcFile(dRess_t ress, const char* srcFileName)
|
|||||||
if (sizeCheck != toRead) EXM_THROW(31, "zstd: %s read error : cannot read header", srcFileName);
|
if (sizeCheck != toRead) EXM_THROW(31, "zstd: %s read error : cannot read header", srcFileName);
|
||||||
{ U32 const magic = MEM_readLE32(ress.srcBuffer);
|
{ U32 const magic = MEM_readLE32(ress.srcBuffer);
|
||||||
#if defined(ZSTD_LEGACY_SUPPORT) && (ZSTD_LEGACY_SUPPORT>=1)
|
#if defined(ZSTD_LEGACY_SUPPORT) && (ZSTD_LEGACY_SUPPORT>=1)
|
||||||
if (ZSTD_isLegacy(magic)) {
|
if (ZSTD_isLegacy(ress.srcBuffer, 4)) {
|
||||||
filesize += FIO_decompressLegacyFrame(dstFile, srcFile, ress.dictBuffer, ress.dictBufferSize, magic);
|
filesize += FIO_decompressLegacyFrame(dstFile, srcFile, ress.dictBuffer, ress.dictBufferSize, magic);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|||||||
+51
-31
@@ -40,8 +40,9 @@
|
|||||||
#include <sys/timeb.h> /* timeb */
|
#include <sys/timeb.h> /* timeb */
|
||||||
#include <string.h> /* strcmp */
|
#include <string.h> /* strcmp */
|
||||||
#include <time.h> /* clock_t */
|
#include <time.h> /* clock_t */
|
||||||
#define ZSTD_STATIC_LINKING_ONLY /* ZSTD_compressContinue */
|
#define ZSTD_STATIC_LINKING_ONLY /* ZSTD_compressContinue, ZSTD_compressBlock */
|
||||||
#include "zstd.h" /* ZSTD_VERSION_STRING, ZSTD_getErrorCode */
|
#include "zstd.h" /* ZSTD_VERSION_STRING */
|
||||||
|
#include "error_public.h" /* ZSTD_getErrorCode */
|
||||||
#include "zdict.h" /* ZDICT_trainFromBuffer */
|
#include "zdict.h" /* ZDICT_trainFromBuffer */
|
||||||
#include "datagen.h" /* RDG_genBuffer */
|
#include "datagen.h" /* RDG_genBuffer */
|
||||||
#include "mem.h"
|
#include "mem.h"
|
||||||
@@ -109,9 +110,9 @@ static unsigned FUZ_highbit32(U32 v32)
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
#define CHECKTEST(var, fn) size_t const var = fn; if (ZSTD_isError(var)) goto _output_error
|
#define CHECK_V(var, fn) size_t const var = fn; if (ZSTD_isError(var)) goto _output_error
|
||||||
#define CHECK(fn) { CHECKTEST(err, fn); }
|
#define CHECK(fn) { CHECK_V(err, fn); }
|
||||||
#define CHECKPLUS(var, fn, more) { CHECKTEST(var, fn); more; }
|
#define CHECKPLUS(var, fn, more) { CHECK_V(var, fn); more; }
|
||||||
static int basicUnitTests(U32 seed, double compressibility)
|
static int basicUnitTests(U32 seed, double compressibility)
|
||||||
{
|
{
|
||||||
size_t const CNBuffSize = 5 MB;
|
size_t const CNBuffSize = 5 MB;
|
||||||
@@ -137,6 +138,12 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
cSize=r );
|
cSize=r );
|
||||||
DISPLAYLEVEL(4, "OK (%u bytes : %.2f%%)\n", (U32)cSize, (double)cSize/CNBuffSize*100);
|
DISPLAYLEVEL(4, "OK (%u bytes : %.2f%%)\n", (U32)cSize, (double)cSize/CNBuffSize*100);
|
||||||
|
|
||||||
|
DISPLAYLEVEL(4, "test%3i : decompressed size test : ", testNb++);
|
||||||
|
{ unsigned long long const rSize = ZSTD_getDecompressedSize(compressedBuffer, cSize);
|
||||||
|
if (rSize != CNBuffSize) goto _output_error;
|
||||||
|
}
|
||||||
|
DISPLAYLEVEL(4, "OK \n");
|
||||||
|
|
||||||
DISPLAYLEVEL(4, "test%3i : decompress %u bytes : ", testNb++, (U32)CNBuffSize);
|
DISPLAYLEVEL(4, "test%3i : decompress %u bytes : ", testNb++, (U32)CNBuffSize);
|
||||||
CHECKPLUS( r , ZSTD_decompress(decodedBuffer, CNBuffSize, compressedBuffer, cSize),
|
CHECKPLUS( r , ZSTD_decompress(decodedBuffer, CNBuffSize, compressedBuffer, cSize),
|
||||||
if (r != CNBuffSize) goto _output_error);
|
if (r != CNBuffSize) goto _output_error);
|
||||||
@@ -216,10 +223,8 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
|
|
||||||
DISPLAYLEVEL(4, "test%3i : check content size on duplicated context : ", testNb++);
|
DISPLAYLEVEL(4, "test%3i : check content size on duplicated context : ", testNb++);
|
||||||
{ size_t const testSize = CNBuffSize / 3;
|
{ size_t const testSize = CNBuffSize / 3;
|
||||||
{ ZSTD_compressionParameters const cPar = ZSTD_getCParams(2, testSize, dictSize);
|
{ ZSTD_parameters p = ZSTD_getParams(2, testSize, dictSize);
|
||||||
ZSTD_frameParameters const fPar = { 1 , 0 , 0 };
|
p.fParams.contentSizeFlag = 1;
|
||||||
ZSTD_parameters p;
|
|
||||||
p.cParams = cPar; p.fParams = fPar;
|
|
||||||
CHECK( ZSTD_compressBegin_advanced(ctxOrig, CNBuffer, dictSize, p, testSize-1) );
|
CHECK( ZSTD_compressBegin_advanced(ctxOrig, CNBuffer, dictSize, p, testSize-1) );
|
||||||
}
|
}
|
||||||
CHECK( ZSTD_copyCCtx(ctxDuplicated, ctxOrig) );
|
CHECK( ZSTD_copyCCtx(ctxDuplicated, ctxOrig) );
|
||||||
@@ -277,10 +282,8 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
DISPLAYLEVEL(4, "OK \n");
|
DISPLAYLEVEL(4, "OK \n");
|
||||||
|
|
||||||
DISPLAYLEVEL(4, "test%3i : compress without dictID : ", testNb++);
|
DISPLAYLEVEL(4, "test%3i : compress without dictID : ", testNb++);
|
||||||
{ ZSTD_frameParameters const fParams = { 0 /*contentSize*/, 0 /*checksum*/, 1 /*NoDictID*/ };
|
{ ZSTD_parameters p = ZSTD_getParams(3, CNBuffSize, dictSize);
|
||||||
ZSTD_compressionParameters const cParams = ZSTD_getCParams(3, CNBuffSize, dictSize);
|
p.fParams.noDictIDFlag = 1;
|
||||||
ZSTD_parameters p;
|
|
||||||
p.cParams = cParams; p.fParams = fParams;
|
|
||||||
cSize = ZSTD_compress_advanced(cctx, compressedBuffer, ZSTD_compressBound(CNBuffSize),
|
cSize = ZSTD_compress_advanced(cctx, compressedBuffer, ZSTD_compressBound(CNBuffSize),
|
||||||
CNBuffer, CNBuffSize,
|
CNBuffer, CNBuffSize,
|
||||||
dictBuffer, dictSize, p);
|
dictBuffer, dictSize, p);
|
||||||
@@ -320,6 +323,7 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
ZSTD_DCtx* const dctx = ZSTD_createDCtx();
|
ZSTD_DCtx* const dctx = ZSTD_createDCtx();
|
||||||
static const size_t blockSize = 100 KB;
|
static const size_t blockSize = 100 KB;
|
||||||
static const size_t dictSize = 16 KB;
|
static const size_t dictSize = 16 KB;
|
||||||
|
size_t cSize2;
|
||||||
|
|
||||||
/* basic block compression */
|
/* basic block compression */
|
||||||
DISPLAYLEVEL(4, "test%3i : Block compression test : ", testNb++);
|
DISPLAYLEVEL(4, "test%3i : Block compression test : ", testNb++);
|
||||||
@@ -330,7 +334,7 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
|
|
||||||
DISPLAYLEVEL(4, "test%3i : Block decompression test : ", testNb++);
|
DISPLAYLEVEL(4, "test%3i : Block decompression test : ", testNb++);
|
||||||
CHECK( ZSTD_decompressBegin(dctx) );
|
CHECK( ZSTD_decompressBegin(dctx) );
|
||||||
{ CHECKTEST(r, ZSTD_decompressBlock(dctx, decodedBuffer, CNBuffSize, compressedBuffer, cSize) );
|
{ CHECK_V(r, ZSTD_decompressBlock(dctx, decodedBuffer, CNBuffSize, compressedBuffer, cSize) );
|
||||||
if (r != blockSize) goto _output_error; }
|
if (r != blockSize) goto _output_error; }
|
||||||
DISPLAYLEVEL(4, "OK \n");
|
DISPLAYLEVEL(4, "OK \n");
|
||||||
|
|
||||||
@@ -339,11 +343,20 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
CHECK( ZSTD_compressBegin_usingDict(cctx, CNBuffer, dictSize, 5) );
|
CHECK( ZSTD_compressBegin_usingDict(cctx, CNBuffer, dictSize, 5) );
|
||||||
cSize = ZSTD_compressBlock(cctx, compressedBuffer, ZSTD_compressBound(blockSize), (char*)CNBuffer+dictSize, blockSize);
|
cSize = ZSTD_compressBlock(cctx, compressedBuffer, ZSTD_compressBound(blockSize), (char*)CNBuffer+dictSize, blockSize);
|
||||||
if (ZSTD_isError(cSize)) goto _output_error;
|
if (ZSTD_isError(cSize)) goto _output_error;
|
||||||
|
cSize2 = ZSTD_compressBlock(cctx, (char*)compressedBuffer+cSize, ZSTD_compressBound(blockSize), (char*)CNBuffer+dictSize+blockSize, blockSize);
|
||||||
|
if (ZSTD_isError(cSize2)) goto _output_error;
|
||||||
|
memcpy((char*)compressedBuffer+cSize, (char*)CNBuffer+dictSize+blockSize, blockSize); /* fake non-compressed block */
|
||||||
|
cSize2 = ZSTD_compressBlock(cctx, (char*)compressedBuffer+cSize+blockSize, ZSTD_compressBound(blockSize),
|
||||||
|
(char*)CNBuffer+dictSize+2*blockSize, blockSize);
|
||||||
|
if (ZSTD_isError(cSize2)) goto _output_error;
|
||||||
DISPLAYLEVEL(4, "OK \n");
|
DISPLAYLEVEL(4, "OK \n");
|
||||||
|
|
||||||
DISPLAYLEVEL(4, "test%3i : Dictionary Block decompression test : ", testNb++);
|
DISPLAYLEVEL(4, "test%3i : Dictionary Block decompression test : ", testNb++);
|
||||||
CHECK( ZSTD_decompressBegin_usingDict(dctx, CNBuffer, dictSize) );
|
CHECK( ZSTD_decompressBegin_usingDict(dctx, CNBuffer, dictSize) );
|
||||||
{ CHECKTEST( r, ZSTD_decompressBlock(dctx, decodedBuffer, CNBuffSize, compressedBuffer, cSize) );
|
{ CHECK_V( r, ZSTD_decompressBlock(dctx, decodedBuffer, CNBuffSize, compressedBuffer, cSize) );
|
||||||
|
if (r != blockSize) goto _output_error; }
|
||||||
|
ZSTD_insertBlock(dctx, (char*)decodedBuffer+blockSize, blockSize); /* insert non-compressed block into dctx history */
|
||||||
|
{ CHECK_V( r, ZSTD_decompressBlock(dctx, (char*)decodedBuffer+2*blockSize, CNBuffSize, (char*)compressedBuffer+cSize+blockSize, cSize2) );
|
||||||
if (r != blockSize) goto _output_error; }
|
if (r != blockSize) goto _output_error; }
|
||||||
DISPLAYLEVEL(4, "OK \n");
|
DISPLAYLEVEL(4, "OK \n");
|
||||||
|
|
||||||
@@ -361,7 +374,7 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
sampleSize += 96 KB;
|
sampleSize += 96 KB;
|
||||||
cSize = ZSTD_compress(compressedBuffer, ZSTD_compressBound(sampleSize), CNBuffer, sampleSize, 1);
|
cSize = ZSTD_compress(compressedBuffer, ZSTD_compressBound(sampleSize), CNBuffer, sampleSize, 1);
|
||||||
if (ZSTD_isError(cSize)) goto _output_error;
|
if (ZSTD_isError(cSize)) goto _output_error;
|
||||||
{ CHECKTEST(regenSize, ZSTD_decompress(decodedBuffer, sampleSize, compressedBuffer, cSize));
|
{ CHECK_V(regenSize, ZSTD_decompress(decodedBuffer, sampleSize, compressedBuffer, cSize));
|
||||||
if (regenSize!=sampleSize) goto _output_error; }
|
if (regenSize!=sampleSize) goto _output_error; }
|
||||||
DISPLAYLEVEL(4, "OK \n");
|
DISPLAYLEVEL(4, "OK \n");
|
||||||
}
|
}
|
||||||
@@ -370,12 +383,12 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
#define ZEROESLENGTH 100
|
#define ZEROESLENGTH 100
|
||||||
DISPLAYLEVEL(4, "test%3i : compress %u zeroes : ", testNb++, ZEROESLENGTH);
|
DISPLAYLEVEL(4, "test%3i : compress %u zeroes : ", testNb++, ZEROESLENGTH);
|
||||||
memset(CNBuffer, 0, ZEROESLENGTH);
|
memset(CNBuffer, 0, ZEROESLENGTH);
|
||||||
{ CHECKTEST(r, ZSTD_compress(compressedBuffer, ZSTD_compressBound(ZEROESLENGTH), CNBuffer, ZEROESLENGTH, 1) );
|
{ CHECK_V(r, ZSTD_compress(compressedBuffer, ZSTD_compressBound(ZEROESLENGTH), CNBuffer, ZEROESLENGTH, 1) );
|
||||||
cSize = r; }
|
cSize = r; }
|
||||||
DISPLAYLEVEL(4, "OK (%u bytes : %.2f%%)\n", (U32)cSize, (double)cSize/ZEROESLENGTH*100);
|
DISPLAYLEVEL(4, "OK (%u bytes : %.2f%%)\n", (U32)cSize, (double)cSize/ZEROESLENGTH*100);
|
||||||
|
|
||||||
DISPLAYLEVEL(4, "test%3i : decompress %u zeroes : ", testNb++, ZEROESLENGTH);
|
DISPLAYLEVEL(4, "test%3i : decompress %u zeroes : ", testNb++, ZEROESLENGTH);
|
||||||
{ CHECKTEST(r, ZSTD_decompress(decodedBuffer, ZEROESLENGTH, compressedBuffer, cSize) );
|
{ CHECK_V(r, ZSTD_decompress(decodedBuffer, ZEROESLENGTH, compressedBuffer, cSize) );
|
||||||
if (r != ZEROESLENGTH) goto _output_error; }
|
if (r != ZEROESLENGTH) goto _output_error; }
|
||||||
DISPLAYLEVEL(4, "OK \n");
|
DISPLAYLEVEL(4, "OK \n");
|
||||||
|
|
||||||
@@ -389,27 +402,29 @@ static int basicUnitTests(U32 seed, double compressibility)
|
|||||||
U32 rSeed = 1;
|
U32 rSeed = 1;
|
||||||
|
|
||||||
/* create batch of 3-bytes sequences */
|
/* create batch of 3-bytes sequences */
|
||||||
{ int i; for (i=0; i < NB3BYTESSEQ; i++) {
|
{ int i;
|
||||||
_3BytesSeqs[i][0] = (BYTE)(FUZ_rand(&rSeed) & 255);
|
for (i=0; i < NB3BYTESSEQ; i++) {
|
||||||
_3BytesSeqs[i][1] = (BYTE)(FUZ_rand(&rSeed) & 255);
|
_3BytesSeqs[i][0] = (BYTE)(FUZ_rand(&rSeed) & 255);
|
||||||
_3BytesSeqs[i][2] = (BYTE)(FUZ_rand(&rSeed) & 255);
|
_3BytesSeqs[i][1] = (BYTE)(FUZ_rand(&rSeed) & 255);
|
||||||
}}
|
_3BytesSeqs[i][2] = (BYTE)(FUZ_rand(&rSeed) & 255);
|
||||||
|
} }
|
||||||
|
|
||||||
/* randomly fills CNBuffer with prepared 3-bytes sequences */
|
/* randomly fills CNBuffer with prepared 3-bytes sequences */
|
||||||
{ int i; for (i=0; i < _3BYTESTESTLENGTH; i += 3) { /* note : CNBuffer size > _3BYTESTESTLENGTH+3 */
|
{ int i;
|
||||||
U32 const id = FUZ_rand(&rSeed) & NB3BYTESSEQMASK;
|
for (i=0; i < _3BYTESTESTLENGTH; i += 3) { /* note : CNBuffer size > _3BYTESTESTLENGTH+3 */
|
||||||
((BYTE*)CNBuffer)[i+0] = _3BytesSeqs[id][0];
|
U32 const id = FUZ_rand(&rSeed) & NB3BYTESSEQMASK;
|
||||||
((BYTE*)CNBuffer)[i+1] = _3BytesSeqs[id][1];
|
((BYTE*)CNBuffer)[i+0] = _3BytesSeqs[id][0];
|
||||||
((BYTE*)CNBuffer)[i+2] = _3BytesSeqs[id][2];
|
((BYTE*)CNBuffer)[i+1] = _3BytesSeqs[id][1];
|
||||||
} }}
|
((BYTE*)CNBuffer)[i+2] = _3BytesSeqs[id][2];
|
||||||
|
} } }
|
||||||
DISPLAYLEVEL(4, "test%3i : compress lots 3-bytes sequences : ", testNb++);
|
DISPLAYLEVEL(4, "test%3i : compress lots 3-bytes sequences : ", testNb++);
|
||||||
{ CHECKTEST(r, ZSTD_compress(compressedBuffer, ZSTD_compressBound(_3BYTESTESTLENGTH),
|
{ CHECK_V(r, ZSTD_compress(compressedBuffer, ZSTD_compressBound(_3BYTESTESTLENGTH),
|
||||||
CNBuffer, _3BYTESTESTLENGTH, 19) );
|
CNBuffer, _3BYTESTESTLENGTH, 19) );
|
||||||
cSize = r; }
|
cSize = r; }
|
||||||
DISPLAYLEVEL(4, "OK (%u bytes : %.2f%%)\n", (U32)cSize, (double)cSize/_3BYTESTESTLENGTH*100);
|
DISPLAYLEVEL(4, "OK (%u bytes : %.2f%%)\n", (U32)cSize, (double)cSize/_3BYTESTESTLENGTH*100);
|
||||||
|
|
||||||
DISPLAYLEVEL(4, "test%3i : decompress lots 3-bytes sequence : ", testNb++);
|
DISPLAYLEVEL(4, "test%3i : decompress lots 3-bytes sequence : ", testNb++);
|
||||||
{ CHECKTEST(r, ZSTD_decompress(decodedBuffer, _3BYTESTESTLENGTH, compressedBuffer, cSize) );
|
{ CHECK_V(r, ZSTD_decompress(decodedBuffer, _3BYTESTESTLENGTH, compressedBuffer, cSize) );
|
||||||
if (r != _3BYTESTESTLENGTH) goto _output_error; }
|
if (r != _3BYTESTESTLENGTH) goto _output_error; }
|
||||||
DISPLAYLEVEL(4, "OK \n");
|
DISPLAYLEVEL(4, "OK \n");
|
||||||
|
|
||||||
@@ -555,6 +570,11 @@ static int fuzzerTests(U32 seed, U32 nbTests, unsigned startTest, U32 const maxD
|
|||||||
CHECK(endCheck != endMark, "ZSTD_compressCCtx : dst buffer overflow"); }
|
CHECK(endCheck != endMark, "ZSTD_compressCCtx : dst buffer overflow"); }
|
||||||
} }
|
} }
|
||||||
|
|
||||||
|
/* Decompressed size test */
|
||||||
|
{ unsigned long long const rSize = ZSTD_getDecompressedSize(cBuffer, cSize);
|
||||||
|
CHECK(rSize != sampleSize, "decompressed size incorrect");
|
||||||
|
}
|
||||||
|
|
||||||
/* frame header decompression test */
|
/* frame header decompression test */
|
||||||
{ ZSTD_frameParams dParams;
|
{ ZSTD_frameParams dParams;
|
||||||
size_t const check = ZSTD_getFrameParams(&dParams, cBuffer, cSize);
|
size_t const check = ZSTD_getFrameParams(&dParams, cBuffer, cSize);
|
||||||
|
|||||||
@@ -381,13 +381,9 @@ static int fuzzerTests(U32 seed, U32 nbTests, unsigned startTest, double compres
|
|||||||
{ size_t const dictStart = FUZ_rand(&lseed) % (srcBufferSize - dictSize);
|
{ size_t const dictStart = FUZ_rand(&lseed) % (srcBufferSize - dictSize);
|
||||||
dict = srcBuffer + dictStart;
|
dict = srcBuffer + dictStart;
|
||||||
}
|
}
|
||||||
{ ZSTD_compressionParameters const cPar = ZSTD_getCParams(cLevel, 0, dictSize);
|
{ ZSTD_parameters params = ZSTD_getParams(cLevel, 0, dictSize);
|
||||||
U32 const checksum = FUZ_rand(&lseed) & 1;
|
params.fParams.checksumFlag = FUZ_rand(&lseed) & 1;
|
||||||
U32 const noDictIDFlag = FUZ_rand(&lseed) & 1;
|
params.fParams.noDictIDFlag = FUZ_rand(&lseed) & 1;
|
||||||
ZSTD_frameParameters const fPar = { 0, checksum, noDictIDFlag };
|
|
||||||
ZSTD_parameters params;
|
|
||||||
params.cParams = cPar;
|
|
||||||
params.fParams = fPar;
|
|
||||||
{ size_t const initError = ZBUFF_compressInit_advanced(zc, dict, dictSize, params, 0);
|
{ size_t const initError = ZBUFF_compressInit_advanced(zc, dict, dictSize, params, 0);
|
||||||
CHECK (ZBUFF_isError(initError),"init error : %s", ZBUFF_getErrorName(initError));
|
CHECK (ZBUFF_isError(initError),"init error : %s", ZBUFF_getErrorName(initError));
|
||||||
} } }
|
} } }
|
||||||
|
|||||||
+16
-9
@@ -33,12 +33,12 @@ It is based on the \fBLZ77\fR family, with further FSE & huff0 entropy stages.
|
|||||||
It also features a very fast decoder, with speed > 500 MB/s per core.
|
It also features a very fast decoder, with speed > 500 MB/s per core.
|
||||||
|
|
||||||
\fBzstd\fR command line is generally similar to gzip, but features the following differences :
|
\fBzstd\fR command line is generally similar to gzip, but features the following differences :
|
||||||
- Original files are preserved
|
- Source files are preserved by default
|
||||||
|
It's possible to remove them automatically by using \fB--rm\fR command
|
||||||
- By default, when compressing a single file, \fBzstd\fR displays progress notifications and result summary.
|
- By default, when compressing a single file, \fBzstd\fR displays progress notifications and result summary.
|
||||||
Use \fB-q\fR to turn them off
|
Use \fB-q\fR to turn them off
|
||||||
|
|
||||||
|
|
||||||
\fBzstd\fR supports the following options :
|
|
||||||
|
|
||||||
.SH OPTIONS
|
.SH OPTIONS
|
||||||
.TP
|
.TP
|
||||||
@@ -57,6 +57,19 @@ It also features a very fast decoder, with speed > 500 MB/s per core.
|
|||||||
.BR \-f ", " --force
|
.BR \-f ", " --force
|
||||||
overwrite output without prompting
|
overwrite output without prompting
|
||||||
.TP
|
.TP
|
||||||
|
.BR \-c ", " --stdout
|
||||||
|
force write to standard output, even if it is the console
|
||||||
|
.TP
|
||||||
|
.BR \--rm
|
||||||
|
remove source file(s) after successful compression or decompression
|
||||||
|
.TP
|
||||||
|
.BR \-k ", " --keep
|
||||||
|
keep source file(s) after successful compression or decompression.
|
||||||
|
This is the default behavior.
|
||||||
|
.TP
|
||||||
|
.BR \-r
|
||||||
|
operate recursively on directories
|
||||||
|
.TP
|
||||||
.BR \-h/\-H ", " --help
|
.BR \-h/\-H ", " --help
|
||||||
display help/long help and exit
|
display help/long help and exit
|
||||||
.TP
|
.TP
|
||||||
@@ -69,14 +82,11 @@ It also features a very fast decoder, with speed > 500 MB/s per core.
|
|||||||
.BR \-q ", " --quiet
|
.BR \-q ", " --quiet
|
||||||
suppress warnings and notifications; specify twice to suppress errors too
|
suppress warnings and notifications; specify twice to suppress errors too
|
||||||
.TP
|
.TP
|
||||||
.BR \-c ", " --stdout
|
|
||||||
force write to standard output, even if it is the console
|
|
||||||
.TP
|
|
||||||
.BR \-C ", " --check
|
.BR \-C ", " --check
|
||||||
add integrity check computed from uncompressed data
|
add integrity check computed from uncompressed data
|
||||||
.TP
|
.TP
|
||||||
.BR \-t ", " --test
|
.BR \-t ", " --test
|
||||||
Test the integrity of compressed files. This option is equivalent to \fB--decompress --stdout > /dev/null\fR.
|
Test the integrity of compressed files. This option is equivalent to \fB--decompress --stdout > /dev/null\fR.
|
||||||
No files are created or removed.
|
No files are created or removed.
|
||||||
|
|
||||||
.SH DICTIONARY
|
.SH DICTIONARY
|
||||||
@@ -121,9 +131,6 @@ Typical gains range from ~10% (at 64KB) to x5 better (at <1KB).
|
|||||||
.TP
|
.TP
|
||||||
.B \-B#
|
.B \-B#
|
||||||
cut file into independent blocks of size # (default: no block)
|
cut file into independent blocks of size # (default: no block)
|
||||||
.TP
|
|
||||||
.B \-r#
|
|
||||||
test all compression levels from 1 to # (default: disabled)
|
|
||||||
|
|
||||||
|
|
||||||
.SH BUGS
|
.SH BUGS
|
||||||
|
|||||||
+149
-139
@@ -34,6 +34,7 @@
|
|||||||
#include "util.h" /* Compiler options, UTIL_HAS_CREATEFILELIST */
|
#include "util.h" /* Compiler options, UTIL_HAS_CREATEFILELIST */
|
||||||
#include <string.h> /* strcmp, strlen */
|
#include <string.h> /* strcmp, strlen */
|
||||||
#include <ctype.h> /* toupper */
|
#include <ctype.h> /* toupper */
|
||||||
|
#include <errno.h> /* errno */
|
||||||
#include "fileio.h"
|
#include "fileio.h"
|
||||||
#ifndef ZSTD_NOBENCH
|
#ifndef ZSTD_NOBENCH
|
||||||
# include "bench.h" /* BMK_benchFiles, BMK_SetNbIterations */
|
# include "bench.h" /* BMK_benchFiles, BMK_SetNbIterations */
|
||||||
@@ -45,7 +46,6 @@
|
|||||||
#include "zstd.h" /* ZSTD_VERSION_STRING */
|
#include "zstd.h" /* ZSTD_VERSION_STRING */
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
/*-************************************
|
/*-************************************
|
||||||
* OS-specific Includes
|
* OS-specific Includes
|
||||||
**************************************/
|
**************************************/
|
||||||
@@ -53,12 +53,12 @@
|
|||||||
# include <io.h> /* _isatty */
|
# include <io.h> /* _isatty */
|
||||||
# define IS_CONSOLE(stdStream) _isatty(_fileno(stdStream))
|
# define IS_CONSOLE(stdStream) _isatty(_fileno(stdStream))
|
||||||
#else
|
#else
|
||||||
#if defined(_POSIX_C_SOURCE) || defined(_XOPEN_SOURCE) || defined(_POSIX_SOURCE)
|
# if defined(_POSIX_C_SOURCE) || defined(_XOPEN_SOURCE) || defined(_POSIX_SOURCE)
|
||||||
# include <unistd.h> /* isatty */
|
# include <unistd.h> /* isatty */
|
||||||
# define IS_CONSOLE(stdStream) isatty(fileno(stdStream))
|
# define IS_CONSOLE(stdStream) isatty(fileno(stdStream))
|
||||||
#else
|
# else
|
||||||
# define IS_CONSOLE(stdStream) 0
|
# define IS_CONSOLE(stdStream) 0
|
||||||
#endif
|
# endif
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
|
||||||
@@ -115,6 +115,8 @@ static int usage(const char* programName)
|
|||||||
DISPLAY( " -D file: use `file` as Dictionary \n");
|
DISPLAY( " -D file: use `file` as Dictionary \n");
|
||||||
DISPLAY( " -o file: result stored into `file` (only if 1 input file) \n");
|
DISPLAY( " -o file: result stored into `file` (only if 1 input file) \n");
|
||||||
DISPLAY( " -f : overwrite output without prompting \n");
|
DISPLAY( " -f : overwrite output without prompting \n");
|
||||||
|
DISPLAY( "--rm : remove source file(s) after successful de/compression \n");
|
||||||
|
DISPLAY( " -k : preserve source file(s) (default) \n");
|
||||||
DISPLAY( " -h/-H : display help/long help and exit\n");
|
DISPLAY( " -h/-H : display help/long help and exit\n");
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
@@ -132,7 +134,6 @@ static int usage_advanced(const char* programName)
|
|||||||
#ifdef UTIL_HAS_CREATEFILELIST
|
#ifdef UTIL_HAS_CREATEFILELIST
|
||||||
DISPLAY( " -r : operate recursively on directories\n");
|
DISPLAY( " -r : operate recursively on directories\n");
|
||||||
#endif
|
#endif
|
||||||
DISPLAY( "--rm : remove source files after successful de/compression \n");
|
|
||||||
#ifndef ZSTD_NOCOMPRESS
|
#ifndef ZSTD_NOCOMPRESS
|
||||||
DISPLAY( "--ultra : enable ultra modes (requires more memory to decompress)\n");
|
DISPLAY( "--ultra : enable ultra modes (requires more memory to decompress)\n");
|
||||||
DISPLAY( "--no-dictID : don't write dictID into header (dictionary compression)\n");
|
DISPLAY( "--no-dictID : don't write dictID into header (dictionary compression)\n");
|
||||||
@@ -169,7 +170,6 @@ static int badusage(const char* programName)
|
|||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
static void waitEnter(void)
|
static void waitEnter(void)
|
||||||
{
|
{
|
||||||
int unused;
|
int unused;
|
||||||
@@ -181,7 +181,7 @@ static void waitEnter(void)
|
|||||||
/*! readU32FromChar() :
|
/*! readU32FromChar() :
|
||||||
@return : unsigned integer value reach from input in `char` format
|
@return : unsigned integer value reach from input in `char` format
|
||||||
Will also modify `*stringPtr`, advancing it to position where it stopped reading.
|
Will also modify `*stringPtr`, advancing it to position where it stopped reading.
|
||||||
Note : this function can overflow if result > MAX_UNIT */
|
Note : this function can overflow if result > MAX_UINT */
|
||||||
static unsigned readU32FromChar(const char** stringPtr)
|
static unsigned readU32FromChar(const char** stringPtr)
|
||||||
{
|
{
|
||||||
unsigned result = 0;
|
unsigned result = 0;
|
||||||
@@ -205,7 +205,8 @@ int main(int argCount, const char** argv)
|
|||||||
dictBuild=0,
|
dictBuild=0,
|
||||||
nextArgumentIsOutFileName=0,
|
nextArgumentIsOutFileName=0,
|
||||||
nextArgumentIsMaxDict=0,
|
nextArgumentIsMaxDict=0,
|
||||||
nextArgumentIsDictID=0;
|
nextArgumentIsDictID=0,
|
||||||
|
nextArgumentIsFile=0;
|
||||||
unsigned cLevel = 1;
|
unsigned cLevel = 1;
|
||||||
unsigned cLevelLast = 1;
|
unsigned cLevelLast = 1;
|
||||||
unsigned recursive = 0;
|
unsigned recursive = 0;
|
||||||
@@ -229,7 +230,7 @@ int main(int argCount, const char** argv)
|
|||||||
(void)recursive; (void)cLevelLast; /* not used when ZSTD_NOBENCH set */
|
(void)recursive; (void)cLevelLast; /* not used when ZSTD_NOBENCH set */
|
||||||
(void)dictCLevel; (void)dictSelect; (void)dictID; /* not used when ZSTD_NODICT set */
|
(void)dictCLevel; (void)dictSelect; (void)dictID; /* not used when ZSTD_NODICT set */
|
||||||
(void)decode; (void)cLevel; /* not used when ZSTD_NOCOMPRESS set */
|
(void)decode; (void)cLevel; /* not used when ZSTD_NOCOMPRESS set */
|
||||||
if (filenameTable==NULL) { DISPLAY("not enough memory\n"); exit(1); }
|
if (filenameTable==NULL) { DISPLAY("zstd: %s \n", strerror(errno)); exit(1); }
|
||||||
filenameTable[0] = stdinmark;
|
filenameTable[0] = stdinmark;
|
||||||
displayOut = stderr;
|
displayOut = stderr;
|
||||||
/* Pick out program name from path. Don't rely on stdlib because of conflicting behavior */
|
/* Pick out program name from path. Don't rely on stdlib because of conflicting behavior */
|
||||||
@@ -247,142 +248,165 @@ int main(int argCount, const char** argv)
|
|||||||
const char* argument = argv[argNb];
|
const char* argument = argv[argNb];
|
||||||
if(!argument) continue; /* Protection if argument empty */
|
if(!argument) continue; /* Protection if argument empty */
|
||||||
|
|
||||||
/* long commands (--long-word) */
|
if (nextArgumentIsFile==0) {
|
||||||
if (!strcmp(argument, "--decompress")) { decode=1; continue; }
|
|
||||||
if (!strcmp(argument, "--force")) { FIO_overwriteMode(); continue; }
|
|
||||||
if (!strcmp(argument, "--version")) { displayOut=stdout; DISPLAY(WELCOME_MESSAGE); CLEAN_RETURN(0); }
|
|
||||||
if (!strcmp(argument, "--help")) { displayOut=stdout; CLEAN_RETURN(usage_advanced(programName)); }
|
|
||||||
if (!strcmp(argument, "--verbose")) { displayLevel=4; continue; }
|
|
||||||
if (!strcmp(argument, "--quiet")) { displayLevel--; continue; }
|
|
||||||
if (!strcmp(argument, "--stdout")) { forceStdout=1; outFileName=stdoutmark; displayLevel=1; continue; }
|
|
||||||
if (!strcmp(argument, "--ultra")) { FIO_setMaxWLog(0); continue; }
|
|
||||||
if (!strcmp(argument, "--check")) { FIO_setChecksumFlag(2); continue; }
|
|
||||||
if (!strcmp(argument, "--no-check")) { FIO_setChecksumFlag(0); continue; }
|
|
||||||
if (!strcmp(argument, "--no-dictID")) { FIO_setDictIDFlag(0); continue; }
|
|
||||||
if (!strcmp(argument, "--sparse")) { FIO_setSparseWrite(2); continue; }
|
|
||||||
if (!strcmp(argument, "--no-sparse")) { FIO_setSparseWrite(0); continue; }
|
|
||||||
if (!strcmp(argument, "--test")) { decode=1; outFileName=nulmark; FIO_overwriteMode(); continue; }
|
|
||||||
if (!strcmp(argument, "--train")) { dictBuild=1; outFileName=g_defaultDictName; continue; }
|
|
||||||
if (!strcmp(argument, "--maxdict")) { nextArgumentIsMaxDict=1; continue; }
|
|
||||||
if (!strcmp(argument, "--dictID")) { nextArgumentIsDictID=1; continue; }
|
|
||||||
if (!strcmp(argument, "--keep")) { continue; } /* does nothing, since preserving input is default; for gzip/xz compatibility */
|
|
||||||
if (!strcmp(argument, "--rm")) { FIO_setRemoveSrcFile(1); continue; }
|
|
||||||
|
|
||||||
/* '-' means stdin/stdout */
|
/* long commands (--long-word) */
|
||||||
if (!strcmp(argument, "-")){
|
if (!strcmp(argument, "--")) { nextArgumentIsFile=1; continue; }
|
||||||
if (!filenameIdx) { filenameIdx=1, filenameTable[0]=stdinmark; outFileName=stdoutmark; continue; }
|
if (!strcmp(argument, "--decompress")) { decode=1; continue; }
|
||||||
}
|
if (!strcmp(argument, "--force")) { FIO_overwriteMode(); continue; }
|
||||||
|
if (!strcmp(argument, "--version")) { displayOut=stdout; DISPLAY(WELCOME_MESSAGE); CLEAN_RETURN(0); }
|
||||||
|
if (!strcmp(argument, "--help")) { displayOut=stdout; CLEAN_RETURN(usage_advanced(programName)); }
|
||||||
|
if (!strcmp(argument, "--verbose")) { displayLevel=4; continue; }
|
||||||
|
if (!strcmp(argument, "--quiet")) { displayLevel--; continue; }
|
||||||
|
if (!strcmp(argument, "--stdout")) { forceStdout=1; outFileName=stdoutmark; displayLevel-=(displayLevel==2); continue; }
|
||||||
|
if (!strcmp(argument, "--ultra")) { FIO_setMaxWLog(0); continue; }
|
||||||
|
if (!strcmp(argument, "--check")) { FIO_setChecksumFlag(2); continue; }
|
||||||
|
if (!strcmp(argument, "--no-check")) { FIO_setChecksumFlag(0); continue; }
|
||||||
|
if (!strcmp(argument, "--no-dictID")) { FIO_setDictIDFlag(0); continue; }
|
||||||
|
if (!strcmp(argument, "--sparse")) { FIO_setSparseWrite(2); continue; }
|
||||||
|
if (!strcmp(argument, "--no-sparse")) { FIO_setSparseWrite(0); continue; }
|
||||||
|
if (!strcmp(argument, "--test")) { decode=1; outFileName=nulmark; FIO_overwriteMode(); continue; }
|
||||||
|
if (!strcmp(argument, "--train")) { dictBuild=1; outFileName=g_defaultDictName; continue; }
|
||||||
|
if (!strcmp(argument, "--maxdict")) { nextArgumentIsMaxDict=1; continue; }
|
||||||
|
if (!strcmp(argument, "--dictID")) { nextArgumentIsDictID=1; continue; }
|
||||||
|
if (!strcmp(argument, "--keep")) { FIO_setRemoveSrcFile(0); continue; }
|
||||||
|
if (!strcmp(argument, "--rm")) { FIO_setRemoveSrcFile(1); continue; }
|
||||||
|
|
||||||
/* Decode commands (note : aggregated commands are allowed) */
|
/* '-' means stdin/stdout */
|
||||||
if (argument[0]=='-') {
|
if (!strcmp(argument, "-")){
|
||||||
argument++;
|
if (!filenameIdx) {
|
||||||
|
filenameIdx=1, filenameTable[0]=stdinmark;
|
||||||
while (argument[0]!=0) {
|
outFileName=stdoutmark;
|
||||||
#ifndef ZSTD_NOCOMPRESS
|
displayLevel-=(displayLevel==2);
|
||||||
/* compression Level */
|
|
||||||
if ((*argument>='0') && (*argument<='9')) {
|
|
||||||
cLevel = readU32FromChar(&argument);
|
|
||||||
dictCLevel = cLevel;
|
|
||||||
if (dictCLevel > ZSTD_maxCLevel())
|
|
||||||
CLEAN_RETURN(badusage(programName));
|
|
||||||
continue;
|
continue;
|
||||||
}
|
} }
|
||||||
#endif
|
|
||||||
|
|
||||||
switch(argument[0])
|
/* Decode commands (note : aggregated commands are allowed) */
|
||||||
{
|
if (argument[0]=='-') {
|
||||||
/* Display help */
|
argument++;
|
||||||
case 'V': displayOut=stdout; DISPLAY(WELCOME_MESSAGE); CLEAN_RETURN(0); /* Version Only */
|
|
||||||
case 'H':
|
|
||||||
case 'h': displayOut=stdout; CLEAN_RETURN(usage_advanced(programName));
|
|
||||||
|
|
||||||
/* Decoding */
|
while (argument[0]!=0) {
|
||||||
case 'd': decode=1; argument++; break;
|
#ifndef ZSTD_NOCOMPRESS
|
||||||
|
/* compression Level */
|
||||||
|
if ((*argument>='0') && (*argument<='9')) {
|
||||||
|
cLevel = readU32FromChar(&argument);
|
||||||
|
dictCLevel = cLevel;
|
||||||
|
if (dictCLevel > ZSTD_maxCLevel())
|
||||||
|
CLEAN_RETURN(badusage(programName));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
/* Force stdout, even if stdout==console */
|
switch(argument[0])
|
||||||
case 'c': forceStdout=1; outFileName=stdoutmark; displayLevel=1; argument++; break;
|
{
|
||||||
|
/* Display help */
|
||||||
|
case 'V': displayOut=stdout; DISPLAY(WELCOME_MESSAGE); CLEAN_RETURN(0); /* Version Only */
|
||||||
|
case 'H':
|
||||||
|
case 'h': displayOut=stdout; CLEAN_RETURN(usage_advanced(programName));
|
||||||
|
|
||||||
/* Use file content as dictionary */
|
/* Decoding */
|
||||||
case 'D': nextEntryIsDictionary = 1; argument++; break;
|
case 'd': decode=1; argument++; break;
|
||||||
|
|
||||||
/* Overwrite */
|
/* Force stdout, even if stdout==console */
|
||||||
case 'f': FIO_overwriteMode(); forceStdout=1; argument++; break;
|
case 'c': forceStdout=1; outFileName=stdoutmark; displayLevel-=(displayLevel==2); argument++; break;
|
||||||
|
|
||||||
/* Verbose mode */
|
/* Use file content as dictionary */
|
||||||
case 'v': displayLevel=4; argument++; break;
|
case 'D': nextEntryIsDictionary = 1; argument++; break;
|
||||||
|
|
||||||
/* Quiet mode */
|
/* Overwrite */
|
||||||
case 'q': displayLevel--; argument++; break;
|
case 'f': FIO_overwriteMode(); forceStdout=1; argument++; break;
|
||||||
|
|
||||||
/* keep source file (default anyway, so useless; for gzip/xz compatibility) */
|
/* Verbose mode */
|
||||||
case 'k': argument++; break;
|
case 'v': displayLevel=4; argument++; break;
|
||||||
|
|
||||||
/* Checksum */
|
/* Quiet mode */
|
||||||
case 'C': argument++; FIO_setChecksumFlag(2); break;
|
case 'q': displayLevel--; argument++; break;
|
||||||
|
|
||||||
/* test compressed file */
|
/* keep source file (default); for gzip/xz compatibility */
|
||||||
case 't': decode=1; outFileName=nulmark; argument++; break;
|
case 'k': FIO_setRemoveSrcFile(0); argument++; break;
|
||||||
|
|
||||||
/* dictionary name */
|
/* Checksum */
|
||||||
case 'o': nextArgumentIsOutFileName=1; argument++; break;
|
case 'C': argument++; FIO_setChecksumFlag(2); break;
|
||||||
|
|
||||||
/* recursive */
|
/* test compressed file */
|
||||||
case 'r': recursive=1; argument++; break;
|
case 't': decode=1; outFileName=nulmark; argument++; break;
|
||||||
|
|
||||||
#ifndef ZSTD_NOBENCH
|
/* dictionary name */
|
||||||
/* Benchmark */
|
case 'o': nextArgumentIsOutFileName=1; argument++; break;
|
||||||
case 'b': bench=1; argument++; break;
|
|
||||||
|
|
||||||
/* range bench (benchmark only) */
|
/* recursive */
|
||||||
case 'e':
|
case 'r': recursive=1; argument++; break;
|
||||||
/* compression Level */
|
|
||||||
|
#ifndef ZSTD_NOBENCH
|
||||||
|
/* Benchmark */
|
||||||
|
case 'b': bench=1; argument++; break;
|
||||||
|
|
||||||
|
/* range bench (benchmark only) */
|
||||||
|
case 'e':
|
||||||
|
/* compression Level */
|
||||||
|
argument++;
|
||||||
|
cLevelLast = readU32FromChar(&argument);
|
||||||
|
break;
|
||||||
|
|
||||||
|
/* Modify Nb Iterations (benchmark only) */
|
||||||
|
case 'i':
|
||||||
argument++;
|
argument++;
|
||||||
cLevelLast = readU32FromChar(&argument);
|
{ U32 const iters = readU32FromChar(&argument);
|
||||||
|
BMK_setNotificationLevel(displayLevel);
|
||||||
|
BMK_SetNbIterations(iters);
|
||||||
|
}
|
||||||
break;
|
break;
|
||||||
|
|
||||||
/* Modify Nb Iterations (benchmark only) */
|
/* cut input into blocks (benchmark only) */
|
||||||
case 'i':
|
case 'B':
|
||||||
argument++;
|
argument++;
|
||||||
{ U32 const iters = readU32FromChar(&argument);
|
{ size_t bSize = readU32FromChar(&argument);
|
||||||
BMK_setNotificationLevel(displayLevel);
|
if (toupper(*argument)=='K') bSize<<=10, argument++; /* allows using KB notation */
|
||||||
BMK_SetNbIterations(iters);
|
if (toupper(*argument)=='M') bSize<<=20, argument++;
|
||||||
|
if (toupper(*argument)=='B') argument++;
|
||||||
|
BMK_setNotificationLevel(displayLevel);
|
||||||
|
BMK_SetBlockSize(bSize);
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
#endif /* ZSTD_NOBENCH */
|
||||||
|
|
||||||
|
/* Dictionary Selection level */
|
||||||
|
case 's':
|
||||||
|
argument++;
|
||||||
|
dictSelect = readU32FromChar(&argument);
|
||||||
|
break;
|
||||||
|
|
||||||
|
/* Pause at the end (-p) or set an additional param (-p#) (hidden option) */
|
||||||
|
case 'p': argument++;
|
||||||
|
#ifndef ZSTD_NOBENCH
|
||||||
|
if ((*argument>='0') && (*argument<='9')) {
|
||||||
|
BMK_setAdditionalParam(readU32FromChar(&argument));
|
||||||
|
} else
|
||||||
|
#endif
|
||||||
|
main_pause=1;
|
||||||
|
break;
|
||||||
|
/* unknown command */
|
||||||
|
default : CLEAN_RETURN(badusage(programName));
|
||||||
}
|
}
|
||||||
break;
|
|
||||||
|
|
||||||
/* cut input into blocks (benchmark only) */
|
|
||||||
case 'B':
|
|
||||||
argument++;
|
|
||||||
{ size_t bSize = readU32FromChar(&argument);
|
|
||||||
if (toupper(*argument)=='K') bSize<<=10, argument++; /* allows using KB notation */
|
|
||||||
if (toupper(*argument)=='M') bSize<<=20, argument++;
|
|
||||||
if (toupper(*argument)=='B') argument++;
|
|
||||||
BMK_setNotificationLevel(displayLevel);
|
|
||||||
BMK_SetBlockSize(bSize);
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
#endif /* ZSTD_NOBENCH */
|
|
||||||
|
|
||||||
/* Dictionary Selection level */
|
|
||||||
case 's':
|
|
||||||
argument++;
|
|
||||||
dictSelect = readU32FromChar(&argument);
|
|
||||||
break;
|
|
||||||
|
|
||||||
/* Pause at the end (-p) or set an additional param (-p#) (hidden option) */
|
|
||||||
case 'p': argument++;
|
|
||||||
#ifndef ZSTD_NOBENCH
|
|
||||||
if ((*argument>='0') && (*argument<='9')) {
|
|
||||||
BMK_setAdditionalParam(readU32FromChar(&argument));
|
|
||||||
} else
|
|
||||||
#endif
|
|
||||||
main_pause=1;
|
|
||||||
break;
|
|
||||||
/* unknown command */
|
|
||||||
default : CLEAN_RETURN(badusage(programName));
|
|
||||||
}
|
}
|
||||||
|
continue;
|
||||||
|
} /* if (argument[0]=='-') */
|
||||||
|
|
||||||
|
if (nextArgumentIsMaxDict) {
|
||||||
|
nextArgumentIsMaxDict = 0;
|
||||||
|
maxDictSize = readU32FromChar(&argument);
|
||||||
|
if (toupper(*argument)=='K') maxDictSize <<= 10;
|
||||||
|
if (toupper(*argument)=='M') maxDictSize <<= 20;
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
continue;
|
|
||||||
} /* if (argument[0]=='-') */
|
if (nextArgumentIsDictID) {
|
||||||
|
nextArgumentIsDictID = 0;
|
||||||
|
dictID = readU32FromChar(&argument);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
} /* if (nextArgumentIsAFile==0) */
|
||||||
|
|
||||||
if (nextEntryIsDictionary) {
|
if (nextEntryIsDictionary) {
|
||||||
nextEntryIsDictionary = 0;
|
nextEntryIsDictionary = 0;
|
||||||
@@ -397,20 +421,6 @@ int main(int argCount, const char** argv)
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (nextArgumentIsMaxDict) {
|
|
||||||
nextArgumentIsMaxDict = 0;
|
|
||||||
maxDictSize = readU32FromChar(&argument);
|
|
||||||
if (toupper(*argument)=='K') maxDictSize <<= 10;
|
|
||||||
if (toupper(*argument)=='M') maxDictSize <<= 20;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (nextArgumentIsDictID) {
|
|
||||||
nextArgumentIsDictID = 0;
|
|
||||||
dictID = readU32FromChar(&argument);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* add filename to list */
|
/* add filename to list */
|
||||||
filenameTable[filenameIdx++] = argument;
|
filenameTable[filenameIdx++] = argument;
|
||||||
}
|
}
|
||||||
|
|||||||
+2
-1
@@ -1,4 +1,4 @@
|
|||||||
projects for various integrated development environments (IDE)
|
projects for various integrated development environments (IDE)
|
||||||
================================
|
================================
|
||||||
|
|
||||||
#### Included projects
|
#### Included projects
|
||||||
@@ -7,3 +7,4 @@ The following projects are included with the zstd distribution:
|
|||||||
- cmake - CMake project contributed by Artyom Dymchenko
|
- cmake - CMake project contributed by Artyom Dymchenko
|
||||||
- VS2008 - Visual Studio 2008 project
|
- VS2008 - Visual Studio 2008 project
|
||||||
- VS2010 - Visual Studio 2010 project (which also works well with Visual Studio 2012, 2013, 2015)
|
- VS2010 - Visual Studio 2010 project (which also works well with Visual Studio 2012, 2013, 2015)
|
||||||
|
- build - command line scripts prepared for Visual Studio compilation without IDE
|
||||||
|
|||||||
@@ -0,0 +1,1155 @@
|
|||||||
|
Zstandard Compression Format
|
||||||
|
============================
|
||||||
|
|
||||||
|
### Notices
|
||||||
|
|
||||||
|
Copyright (c) 2016 Yann Collet
|
||||||
|
|
||||||
|
Permission is granted to copy and distribute this document
|
||||||
|
for any purpose and without charge,
|
||||||
|
including translations into other languages
|
||||||
|
and incorporation into compilations,
|
||||||
|
provided that the copyright notice and this notice are preserved,
|
||||||
|
and that any substantive changes or deletions from the original
|
||||||
|
are clearly marked.
|
||||||
|
Distribution of this document is unlimited.
|
||||||
|
|
||||||
|
### Version
|
||||||
|
|
||||||
|
0.1.0 (08/07/16)
|
||||||
|
|
||||||
|
|
||||||
|
Introduction
|
||||||
|
------------
|
||||||
|
|
||||||
|
The purpose of this document is to define a lossless compressed data format,
|
||||||
|
that is independent of CPU type, operating system,
|
||||||
|
file system and character set, suitable for
|
||||||
|
file compression, pipe and streaming compression,
|
||||||
|
using the [Zstandard algorithm](http://www.zstandard.org).
|
||||||
|
|
||||||
|
The data can be produced or consumed,
|
||||||
|
even for an arbitrarily long sequentially presented input data stream,
|
||||||
|
using only an a priori bounded amount of intermediate storage,
|
||||||
|
and hence can be used in data communications.
|
||||||
|
The format uses the Zstandard compression method,
|
||||||
|
and optional [xxHash-64 checksum method](http://www.xxhash.org),
|
||||||
|
for detection of data corruption.
|
||||||
|
|
||||||
|
The data format defined by this specification
|
||||||
|
does not attempt to allow random access to compressed data.
|
||||||
|
|
||||||
|
This specification is intended for use by implementers of software
|
||||||
|
to compress data into Zstandard format and/or decompress data from Zstandard format.
|
||||||
|
The text of the specification assumes a basic background in programming
|
||||||
|
at the level of bits and other primitive data representations.
|
||||||
|
|
||||||
|
Unless otherwise indicated below,
|
||||||
|
a compliant compressor must produce data sets
|
||||||
|
that conform to the specifications presented here.
|
||||||
|
It doesn’t need to support all options though.
|
||||||
|
|
||||||
|
A compliant decompressor must be able to decompress
|
||||||
|
at least one working set of parameters
|
||||||
|
that conforms to the specifications presented here.
|
||||||
|
It may also ignore informative fields, such as checksum.
|
||||||
|
Whenever it does not support a parameter defined in the compressed stream,
|
||||||
|
it must produce a non-ambiguous error code and associated error message
|
||||||
|
explaining which parameter is unsupported.
|
||||||
|
|
||||||
|
|
||||||
|
Definitions
|
||||||
|
-----------
|
||||||
|
A content compressed by Zstandard is transformed into a Zstandard __frame__.
|
||||||
|
Multiple frames can be appended into a single file or stream.
|
||||||
|
A frame is totally independent, has a defined beginning and end,
|
||||||
|
and a set of parameters which tells the decoder how to decompress it.
|
||||||
|
|
||||||
|
A frame encapsulates one or multiple __blocks__.
|
||||||
|
Each block can be compressed or not,
|
||||||
|
and has a guaranteed maximum content size, which depends on frame parameters.
|
||||||
|
Unlike frames, each block depends on previous blocks for proper decoding.
|
||||||
|
However, each block can be decompressed without waiting for its successor,
|
||||||
|
allowing streaming operations.
|
||||||
|
|
||||||
|
|
||||||
|
General Structure of Zstandard Frame format
|
||||||
|
-------------------------------------------
|
||||||
|
|
||||||
|
| MagicNb | Frame Header | Block | (More blocks) | EndMark |
|
||||||
|
|:-------:|:-------------:| ----- | ------------- | ------- |
|
||||||
|
| 4 bytes | 2-14 bytes | | | 3 bytes |
|
||||||
|
|
||||||
|
__Magic Number__
|
||||||
|
|
||||||
|
4 Bytes, Little endian format.
|
||||||
|
Value : 0xFD2FB527
|
||||||
|
|
||||||
|
__Frame Header__
|
||||||
|
|
||||||
|
2 to 14 Bytes, detailed in [next part](#frame-header).
|
||||||
|
|
||||||
|
__Data Blocks__
|
||||||
|
|
||||||
|
Detailed in [next chapter](#data-blocks).
|
||||||
|
That’s where compressed data is stored.
|
||||||
|
|
||||||
|
__EndMark__
|
||||||
|
|
||||||
|
The flow of blocks ends when the last block header brings an _end signal_ .
|
||||||
|
This last block header may optionally host a __Content Checksum__ .
|
||||||
|
|
||||||
|
##### __Content Checksum__
|
||||||
|
|
||||||
|
Content Checksum verify that frame content has been regenerated correctly.
|
||||||
|
The content checksum is the result
|
||||||
|
of [xxh64() hash function](https://www.xxHash.com)
|
||||||
|
digesting the original (decoded) data as input, and a seed of zero.
|
||||||
|
Bits from 11 to 32 (included) are extracted to form a 22 bits checksum
|
||||||
|
stored into the endmark body.
|
||||||
|
```
|
||||||
|
mask22bits = (1<<22)-1;
|
||||||
|
contentChecksum = (XXH64(content, size, 0) >> 11) & mask22bits;
|
||||||
|
```
|
||||||
|
Content checksum is only present when its associated flag
|
||||||
|
is set in the frame descriptor.
|
||||||
|
Its usage is optional.
|
||||||
|
|
||||||
|
__Frame Concatenation__
|
||||||
|
|
||||||
|
In some circumstances, it may be required to append multiple frames,
|
||||||
|
for example in order to add new data to an existing compressed file
|
||||||
|
without re-framing it.
|
||||||
|
|
||||||
|
In such case, each frame brings its own set of descriptor flags.
|
||||||
|
Each frame is considered independent.
|
||||||
|
The only relation between frames is their sequential order.
|
||||||
|
|
||||||
|
The ability to decode multiple concatenated frames
|
||||||
|
within a single stream or file is left outside of this specification.
|
||||||
|
As an example, the reference `zstd` command line utility is able
|
||||||
|
to decode all concatenated frames in their sequential order,
|
||||||
|
delivering the final decompressed result as if it was a single content.
|
||||||
|
|
||||||
|
|
||||||
|
Frame Header
|
||||||
|
-------------
|
||||||
|
|
||||||
|
| FHD | (WD) | (dictID) | (Content Size) |
|
||||||
|
| ------- | --------- | --------- |:--------------:|
|
||||||
|
| 1 byte | 0-1 byte | 0-4 bytes | 0 - 8 bytes |
|
||||||
|
|
||||||
|
Frame header has a variable size, which uses a minimum of 2 bytes,
|
||||||
|
and up to 14 bytes depending on optional parameters.
|
||||||
|
|
||||||
|
__FHD byte__ (Frame Header Descriptor)
|
||||||
|
|
||||||
|
The first Header's byte is called the Frame Header Descriptor.
|
||||||
|
It tells which other fields are present.
|
||||||
|
Decoding this byte is enough to tell the size of Frame Header.
|
||||||
|
|
||||||
|
| BitNb | 7-6 | 5 | 4 | 3 | 2 | 1-0 |
|
||||||
|
| ------- | ------ | ------- | ------ | -------- | -------- | ------ |
|
||||||
|
|FieldName| FCSize | Segment | Unused | Reserved | Checksum | dictID |
|
||||||
|
|
||||||
|
In this table, bit 7 is highest bit, while bit 0 is lowest.
|
||||||
|
|
||||||
|
__Frame Content Size flag__
|
||||||
|
|
||||||
|
This is a 2-bits flag (`= FHD >> 6`),
|
||||||
|
specifying if decompressed data size is provided within the header.
|
||||||
|
|
||||||
|
| Value | 0 | 1 | 2 | 3 |
|
||||||
|
| ------- | --- | --- | --- | --- |
|
||||||
|
|FieldSize| 0-1 | 2 | 4 | 8 |
|
||||||
|
|
||||||
|
Value 0 meaning depends on _single segment_ mode :
|
||||||
|
it either means `0` (size not provided) _if_ the `WD` byte is present,
|
||||||
|
or `1` (frame content size <= 255 bytes) otherwise.
|
||||||
|
|
||||||
|
__Single Segment__
|
||||||
|
|
||||||
|
If this flag is set,
|
||||||
|
data shall be regenerated within a single continuous memory segment.
|
||||||
|
|
||||||
|
In which case, `WD` byte __is not present__,
|
||||||
|
but `Frame Content Size` field necessarily is.
|
||||||
|
As a consequence, the decoder must allocate a memory segment
|
||||||
|
of size `>= Frame Content Size`.
|
||||||
|
|
||||||
|
In order to preserve the decoder from unreasonable memory requirement,
|
||||||
|
a decoder can reject a compressed frame
|
||||||
|
which requests a memory size beyond decoder's authorized range.
|
||||||
|
|
||||||
|
For broader compatibility, decoders are recommended to support
|
||||||
|
memory sizes of at least 8 MB.
|
||||||
|
This is just a recommendation,
|
||||||
|
each decoder is free to support higher or lower limits,
|
||||||
|
depending on local limitations.
|
||||||
|
|
||||||
|
__Unused bit__
|
||||||
|
|
||||||
|
The value of this bit is unimportant
|
||||||
|
and not interpreted by a decoder compliant with this specification version.
|
||||||
|
It may be used in a future revision,
|
||||||
|
to signal a property which is not required to properly decode the frame.
|
||||||
|
|
||||||
|
__Reserved bit__
|
||||||
|
|
||||||
|
This bit is reserved for some future feature.
|
||||||
|
Its value _must be zero_.
|
||||||
|
A decoder compliant with this specification version must ensure it is not set.
|
||||||
|
This bit may be used in a future revision,
|
||||||
|
to signal a feature that must be interpreted in order to decode the frame.
|
||||||
|
|
||||||
|
__Content checksum flag__
|
||||||
|
|
||||||
|
If this flag is set, a content checksum will be present into the EndMark.
|
||||||
|
The checksum is a 22 bits value extracted from the XXH64() of data,
|
||||||
|
and stored into endMark. See [__Content Checksum__](#content-checksum) .
|
||||||
|
|
||||||
|
__Dictionary ID flag__
|
||||||
|
|
||||||
|
This is a 2-bits flag (`= FHD & 3`),
|
||||||
|
telling if a dictionary ID is provided within the header.
|
||||||
|
It also specifies the size of this field.
|
||||||
|
|
||||||
|
| Value | 0 | 1 | 2 | 3 |
|
||||||
|
| ------- | --- | --- | --- | --- |
|
||||||
|
|FieldSize| 0 | 1 | 2 | 4 |
|
||||||
|
|
||||||
|
__WD byte__ (Window Descriptor)
|
||||||
|
|
||||||
|
Provides guarantees on maximum back-reference distance
|
||||||
|
that will be present within compressed data.
|
||||||
|
This information is useful for decoders to allocate enough memory.
|
||||||
|
|
||||||
|
`WD` byte is optional. It's not present in `single segment` mode.
|
||||||
|
In which case, the maximum back-reference distance is the content size itself,
|
||||||
|
which can be any value from 1 to 2^64-1 bytes (16 EB).
|
||||||
|
|
||||||
|
| BitNb | 7-3 | 0-2 |
|
||||||
|
| --------- | -------- | -------- |
|
||||||
|
| FieldName | Exponent | Mantissa |
|
||||||
|
|
||||||
|
Maximum distance is given by the following formulae :
|
||||||
|
```
|
||||||
|
windowLog = 10 + Exponent;
|
||||||
|
windowBase = 1 << windowLog;
|
||||||
|
windowAdd = (windowBase / 8) * Mantissa;
|
||||||
|
windowSize = windowBase + windowAdd;
|
||||||
|
```
|
||||||
|
The minimum window size is 1 KB.
|
||||||
|
The maximum size is `15*(1<<38)` bytes, which is 1.875 TB.
|
||||||
|
|
||||||
|
To properly decode compressed data,
|
||||||
|
a decoder will need to allocate a buffer of at least `windowSize` bytes.
|
||||||
|
|
||||||
|
In order to preserve decoder from unreasonable memory requirements,
|
||||||
|
a decoder can refuse a compressed frame
|
||||||
|
which requests a memory size beyond decoder's authorized range.
|
||||||
|
|
||||||
|
For improved interoperability,
|
||||||
|
decoders are recommended to be compatible with window sizes of 8 MB.
|
||||||
|
Encoders are recommended to not request more than 8 MB.
|
||||||
|
It's merely a recommendation though,
|
||||||
|
decoders are free to support larger or lower limits,
|
||||||
|
depending on local limitations.
|
||||||
|
|
||||||
|
__Dictionary ID__
|
||||||
|
|
||||||
|
This is a variable size field, which contains an ID.
|
||||||
|
It checks if the correct dictionary is used for decoding.
|
||||||
|
Note that this field is optional. If it's not present,
|
||||||
|
it's up to the caller to make sure it uses the correct dictionary.
|
||||||
|
|
||||||
|
Field size depends on __Dictionary ID flag__.
|
||||||
|
1 byte can represent an ID 0-255.
|
||||||
|
2 bytes can represent an ID 0-65535.
|
||||||
|
4 bytes can represent an ID 0-4294967295.
|
||||||
|
|
||||||
|
It's allowed to represent a small ID (for example `13`)
|
||||||
|
with a large 4-bytes dictionary ID, losing some compacity in the process.
|
||||||
|
|
||||||
|
__Frame Content Size__
|
||||||
|
|
||||||
|
This is the original (uncompressed) size.
|
||||||
|
This information is optional, and only present if associated flag is set.
|
||||||
|
Content size is provided using 1, 2, 4 or 8 Bytes.
|
||||||
|
Format is Little endian.
|
||||||
|
|
||||||
|
| Field Size | Range |
|
||||||
|
| ---------- | ---------- |
|
||||||
|
| 0 | 0 |
|
||||||
|
| 1 | 0 - 255 |
|
||||||
|
| 2 | 256 - 65791|
|
||||||
|
| 4 | 0 - 2^32-1 |
|
||||||
|
| 8 | 0 - 2^64-1 |
|
||||||
|
|
||||||
|
When field size is 1, 4 or 8 bytes, the value is read directly.
|
||||||
|
When field size is 2, _an offset of 256 is added_.
|
||||||
|
It's allowed to represent a small size (ex: `18`) using any compatible variant.
|
||||||
|
A size of `0` means `content size is unknown`.
|
||||||
|
In which case, the `WD` byte will necessarily be present,
|
||||||
|
and becomes the only hint to guide memory allocation.
|
||||||
|
|
||||||
|
In order to preserve decoder from unreasonable memory requirement,
|
||||||
|
a decoder can refuse a compressed frame
|
||||||
|
which requests a memory size beyond decoder's authorized range.
|
||||||
|
|
||||||
|
|
||||||
|
Data Blocks
|
||||||
|
-----------
|
||||||
|
|
||||||
|
| B. Header | data |
|
||||||
|
|:---------:| ------ |
|
||||||
|
| 3 bytes | |
|
||||||
|
|
||||||
|
|
||||||
|
__Block Header__
|
||||||
|
|
||||||
|
This field uses 3-bytes, format is __big-endian__.
|
||||||
|
|
||||||
|
The 2 highest bits represent the `block type`,
|
||||||
|
while the remaining 22 bits represent the (compressed) block size.
|
||||||
|
|
||||||
|
There are 4 block types :
|
||||||
|
|
||||||
|
| Value | 0 | 1 | 2 | 3 |
|
||||||
|
| ---------- | ---------- | --- | --- | ------- |
|
||||||
|
| Block Type | Compressed | Raw | RLE | EndMark |
|
||||||
|
|
||||||
|
- Compressed : this is a [Zstandard compressed block](#compressed-block-format),
|
||||||
|
detailed in another section of this specification.
|
||||||
|
"block size" is the compressed size.
|
||||||
|
Decompressed size is unknown,
|
||||||
|
but its maximum possible value is guaranteed (see below)
|
||||||
|
- Raw : this is an uncompressed block.
|
||||||
|
"block size" is the number of bytes to read and copy.
|
||||||
|
- RLE : this is a single byte, repeated N times.
|
||||||
|
In which case, "block size" is the size to regenerate,
|
||||||
|
while the "compressed" block is just 1 byte (the byte to repeat).
|
||||||
|
- EndMark : this is not a block. Signal the end of the frame.
|
||||||
|
The rest of the field may be optionally filled by a checksum
|
||||||
|
(see [Content Checksum](#content-checksum)).
|
||||||
|
|
||||||
|
Block sizes must respect a few rules :
|
||||||
|
- In compressed mode, compressed size if always strictly `< decompressed size`.
|
||||||
|
- Block decompressed size is always <= maximum back-reference distance .
|
||||||
|
- Block decompressed size is always <= 128 KB
|
||||||
|
|
||||||
|
|
||||||
|
__Data__
|
||||||
|
|
||||||
|
Where the actual data to decode stands.
|
||||||
|
It might be compressed or not, depending on previous field indications.
|
||||||
|
A data block is not necessarily "full" :
|
||||||
|
since an arbitrary “flush” may happen anytime,
|
||||||
|
block decompressed content can be any size,
|
||||||
|
up to Block Maximum Decompressed Size, which is the smallest of :
|
||||||
|
- Maximum back-reference distance
|
||||||
|
- 128 KB
|
||||||
|
|
||||||
|
|
||||||
|
Skippable Frames
|
||||||
|
----------------
|
||||||
|
|
||||||
|
| Magic Number | Frame Size | User Data |
|
||||||
|
|:------------:|:----------:| --------- |
|
||||||
|
| 4 bytes | 4 bytes | |
|
||||||
|
|
||||||
|
Skippable frames allow the insertion of user-defined data
|
||||||
|
into a flow of concatenated frames.
|
||||||
|
Its design is pretty straightforward,
|
||||||
|
with the sole objective to allow the decoder to quickly skip
|
||||||
|
over user-defined data and continue decoding.
|
||||||
|
|
||||||
|
Skippable frames defined in this specification are compatible with [LZ4] ones.
|
||||||
|
|
||||||
|
[LZ4]:http://www.lz4.org
|
||||||
|
|
||||||
|
__Magic Number__ :
|
||||||
|
|
||||||
|
4 Bytes, Little endian format.
|
||||||
|
Value : 0x184D2A5X, which means any value from 0x184D2A50 to 0x184D2A5F.
|
||||||
|
All 16 values are valid to identify a skippable frame.
|
||||||
|
|
||||||
|
__Frame Size__ :
|
||||||
|
|
||||||
|
This is the size, in bytes, of the following User Data
|
||||||
|
(without including the magic number nor the size field itself).
|
||||||
|
4 Bytes, Little endian format, unsigned 32-bits.
|
||||||
|
This means User Data can’t be bigger than (2^32-1) Bytes.
|
||||||
|
|
||||||
|
__User Data__ :
|
||||||
|
|
||||||
|
User Data can be anything. Data will just be skipped by the decoder.
|
||||||
|
|
||||||
|
|
||||||
|
Compressed block format
|
||||||
|
-----------------------
|
||||||
|
This specification details the content of a _compressed block_.
|
||||||
|
A compressed block has a size, which must be known.
|
||||||
|
It also has a guaranteed maximum regenerated size,
|
||||||
|
in order to properly allocate destination buffer.
|
||||||
|
See [Data Blocks](#data-blocks) for more details.
|
||||||
|
|
||||||
|
A compressed block consists of 2 sections :
|
||||||
|
- [Literals section](#literals-section)
|
||||||
|
- [Sequences section](#sequences-section)
|
||||||
|
|
||||||
|
### Prerequisites
|
||||||
|
To decode a compressed block, the following elements are necessary :
|
||||||
|
- Previous decoded blocks, up to a distance of `windowSize`,
|
||||||
|
or all previous blocks in "single segment" mode.
|
||||||
|
- List of "recent offsets" from previous compressed block.
|
||||||
|
- Decoding tables of previous compressed block for each symbol type
|
||||||
|
(literals, litLength, matchLength, offset).
|
||||||
|
|
||||||
|
|
||||||
|
### Literals section
|
||||||
|
|
||||||
|
Literals are compressed using Huffman prefix codes.
|
||||||
|
During sequence phase, literals will be entangled with match copy operations.
|
||||||
|
All literals are regrouped in the first part of the block.
|
||||||
|
They can be decoded first, and then copied during sequence operations,
|
||||||
|
or they can be decoded on the flow, as needed by sequence commands.
|
||||||
|
|
||||||
|
| Header | (Tree Description) | Stream1 | (Stream2) | (Stream3) | (Stream4) |
|
||||||
|
| ------ | ------------------ | ------- | --------- | --------- | --------- |
|
||||||
|
|
||||||
|
Literals can be compressed, or uncompressed.
|
||||||
|
When compressed, an optional tree description can be present,
|
||||||
|
followed by 1 or 4 streams.
|
||||||
|
|
||||||
|
#### Literals section header
|
||||||
|
|
||||||
|
Header is in charge of describing how literals are packed.
|
||||||
|
It's a byte-aligned variable-size bitfield, ranging from 1 to 5 bytes,
|
||||||
|
using big-endian convention.
|
||||||
|
|
||||||
|
| BlockType | sizes format | (compressed size) | regenerated size |
|
||||||
|
| --------- | ------------ | ----------------- | ---------------- |
|
||||||
|
| 2 bits | 1 - 2 bits | 0 - 18 bits | 5 - 20 bits |
|
||||||
|
|
||||||
|
__Block Type__ :
|
||||||
|
|
||||||
|
This is a 2-bits field, describing 4 different block types :
|
||||||
|
|
||||||
|
| Value | 0 | 1 | 2 | 3 |
|
||||||
|
| ---------- | ---------- | ------ | --- | ------- |
|
||||||
|
| Block Type | Compressed | Repeat | Raw | RLE |
|
||||||
|
|
||||||
|
- Compressed : This is a standard huffman-compressed block,
|
||||||
|
starting with a huffman tree description.
|
||||||
|
See details below.
|
||||||
|
- Repeat Stats : This is a huffman-compressed block,
|
||||||
|
using huffman tree _from previous huffman-compressed literals block_.
|
||||||
|
Huffman tree description will be skipped.
|
||||||
|
- Raw : Literals are stored uncompressed.
|
||||||
|
- RLE : Literals consist of a single byte value repeated N times.
|
||||||
|
|
||||||
|
__Sizes format__ :
|
||||||
|
|
||||||
|
Sizes format are divided into 2 families :
|
||||||
|
|
||||||
|
- For compressed block, it requires to decode both the compressed size
|
||||||
|
and the decompressed size. It will also decode the number of streams.
|
||||||
|
- For Raw or RLE blocks, it's enough to decode the size to regenerate.
|
||||||
|
|
||||||
|
For values spanning several bytes, convention is Big-endian.
|
||||||
|
|
||||||
|
__Sizes format for Raw or RLE literals block__ :
|
||||||
|
|
||||||
|
- Value : 0x : Regenerated size uses 5 bits (0-31).
|
||||||
|
Total literal header size is 1 byte.
|
||||||
|
`size = h[0] & 31;`
|
||||||
|
- Value : 10 : Regenerated size uses 12 bits (0-4095).
|
||||||
|
Total literal header size is 2 bytes.
|
||||||
|
`size = ((h[0] & 15) << 8) + h[1];`
|
||||||
|
- Value : 11 : Regenerated size uses 20 bits (0-1048575).
|
||||||
|
Total literal header size is 3 bytes.
|
||||||
|
`size = ((h[0] & 15) << 16) + (h[1]<<8) + h[2];`
|
||||||
|
|
||||||
|
Note : it's allowed to represent a short value (ex : `13`)
|
||||||
|
using a long format, accepting the reduced compacity.
|
||||||
|
|
||||||
|
__Sizes format for Compressed literals block__ :
|
||||||
|
|
||||||
|
Note : also applicable to "repeat-stats" blocks.
|
||||||
|
- Value : 00 : 4 streams.
|
||||||
|
Compressed and regenerated sizes use 10 bits (0-1023).
|
||||||
|
Total literal header size is 3 bytes.
|
||||||
|
- Value : 01 : _Single stream_.
|
||||||
|
Compressed and regenerated sizes use 10 bits (0-1023).
|
||||||
|
Total literal header size is 3 bytes.
|
||||||
|
- Value : 10 : 4 streams.
|
||||||
|
Compressed and regenerated sizes use 14 bits (0-16383).
|
||||||
|
Total literal header size is 4 bytes.
|
||||||
|
- Value : 10 : 4 streams.
|
||||||
|
Compressed and regenerated sizes use 18 bits (0-262143).
|
||||||
|
Total literal header size is 5 bytes.
|
||||||
|
|
||||||
|
Compressed and regenerated size fields follow big endian convention.
|
||||||
|
|
||||||
|
#### Huffman Tree description
|
||||||
|
|
||||||
|
This section is only present when literals block type is `Compressed` (`0`).
|
||||||
|
|
||||||
|
Prefix coding represents symbols from an a priori known alphabet
|
||||||
|
by bit sequences (codes), one code for each symbol,
|
||||||
|
in a manner such that different symbols may be represented
|
||||||
|
by bit sequences of different lengths,
|
||||||
|
but a parser can always parse an encoded string
|
||||||
|
unambiguously symbol-by-symbol.
|
||||||
|
|
||||||
|
Given an alphabet with known symbol frequencies,
|
||||||
|
the Huffman algorithm allows the construction of an optimal prefix code
|
||||||
|
using the fewest bits of any possible prefix codes for that alphabet.
|
||||||
|
Such a code is called a Huffman code.
|
||||||
|
|
||||||
|
Prefix code must not exceed a maximum code length.
|
||||||
|
More bits improve accuracy but cost more header size,
|
||||||
|
and require more memory for decoding operations.
|
||||||
|
|
||||||
|
The current format limits the maximum depth to 15 bits.
|
||||||
|
The reference decoder goes further, by limiting it to 11 bits.
|
||||||
|
It is recommended to remain compatible with reference decoder.
|
||||||
|
|
||||||
|
|
||||||
|
##### Representation
|
||||||
|
|
||||||
|
All literal values from zero (included) to last present one (excluded)
|
||||||
|
are represented by `weight` values, from 0 to `maxBits`.
|
||||||
|
Transformation from `weight` to `nbBits` follows this formulae :
|
||||||
|
`nbBits = weight ? maxBits + 1 - weight : 0;` .
|
||||||
|
The last symbol's weight is deduced from previously decoded ones,
|
||||||
|
by completing to the nearest power of 2.
|
||||||
|
This power of 2 gives `maxBits`, the depth of the current tree.
|
||||||
|
|
||||||
|
__Example__ :
|
||||||
|
Let's presume the following huffman tree must be described :
|
||||||
|
|
||||||
|
| literal | 0 | 1 | 2 | 3 | 4 | 5 |
|
||||||
|
| ------- | --- | --- | --- | --- | --- | --- |
|
||||||
|
| nbBits | 1 | 2 | 3 | 0 | 4 | 4 |
|
||||||
|
|
||||||
|
The tree depth is 4, since its smallest element uses 4 bits.
|
||||||
|
Value `5` will not be listed, nor will values above `5`.
|
||||||
|
Values from `0` to `4` will be listed using `weight` instead of `nbBits`.
|
||||||
|
Weight formula is : `weight = nbBits ? maxBits + 1 - nbBits : 0;`
|
||||||
|
It gives the following serie of weights :
|
||||||
|
|
||||||
|
| weights | 4 | 3 | 2 | 0 | 1 |
|
||||||
|
| ------- | --- | --- | --- | --- | --- |
|
||||||
|
| literal | 0 | 1 | 2 | 3 | 4 |
|
||||||
|
|
||||||
|
The decoder will do the inverse operation :
|
||||||
|
having collected weights of literals from `0` to `4`,
|
||||||
|
it knows the last literal, `5`, is present with a non-zero weight.
|
||||||
|
The weight of `5` can be deducted by joining to the nearest power of 2.
|
||||||
|
Sum of 2^(weight-1) (excluding 0) is :
|
||||||
|
`8 + 4 + 2 + 0 + 1 = 15`
|
||||||
|
Nearest power of 2 is 16.
|
||||||
|
Therefore, `maxBits = 4` and `weight[5] = 1`.
|
||||||
|
|
||||||
|
##### Huffman Tree header
|
||||||
|
|
||||||
|
This is a single byte value (0-255),
|
||||||
|
which tells how to decode the list of weights.
|
||||||
|
|
||||||
|
- if headerByte >= 242 : this is one of 14 pre-defined weight distributions :
|
||||||
|
|
||||||
|
| value |242|243|244|245|246|247|248|249|250|251|252|253|254|255|
|
||||||
|
| -------- |---|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||||
|
| Nb of 1s | 1 | 2 | 3 | 4 | 7 | 8 | 15| 16| 31| 32| 63| 64|127|128|
|
||||||
|
|Complement| 1 | 2 | 1 | 4 | 1 | 8 | 1 | 16| 1 | 32| 1 | 64| 1 |128|
|
||||||
|
|
||||||
|
_Note_ : complement is found by using "join to nearest power of 2" rule.
|
||||||
|
|
||||||
|
- if headerByte >= 128 : this is a direct representation,
|
||||||
|
where each weight is written directly as a 4 bits field (0-15).
|
||||||
|
The full representation occupies `((nbSymbols+1)/2)` bytes,
|
||||||
|
meaning it uses a last full byte even if nbSymbols is odd.
|
||||||
|
`nbSymbols = headerByte - 127;`.
|
||||||
|
Note that maximum nbSymbols is 241-127 = 114.
|
||||||
|
A larger serie must necessarily use FSE compression.
|
||||||
|
|
||||||
|
- if headerByte < 128 :
|
||||||
|
the serie of weights is compressed by FSE.
|
||||||
|
The length of the FSE-compressed serie is `headerByte` (0-127).
|
||||||
|
|
||||||
|
##### FSE (Finite State Entropy) compression of huffman weights
|
||||||
|
|
||||||
|
The serie of weights is compressed using FSE compression.
|
||||||
|
It's a single bitstream with 2 interleaved states,
|
||||||
|
sharing a single distribution table.
|
||||||
|
|
||||||
|
To decode an FSE bitstream, it is necessary to know its compressed size.
|
||||||
|
Compressed size is provided by `headerByte`.
|
||||||
|
It's also necessary to know its maximum decompressed size,
|
||||||
|
which is `255`, since literal values span from `0` to `255`,
|
||||||
|
and last symbol value is not represented.
|
||||||
|
|
||||||
|
An FSE bitstream starts by a header, describing probabilities distribution.
|
||||||
|
It will create a Decoding Table.
|
||||||
|
Table must be pre-allocated, which requires to support a maximum accuracy.
|
||||||
|
For a list of huffman weights, recommended maximum is 7 bits.
|
||||||
|
|
||||||
|
FSE header is [described in relevant chapter](#fse-distribution-table--condensed-format),
|
||||||
|
and so is [FSE bitstream](#bitstream).
|
||||||
|
The main difference is that Huffman header compression uses 2 states,
|
||||||
|
which share the same FSE distribution table.
|
||||||
|
Bitstream contains only FSE symbols, there are no interleaved "raw bitfields".
|
||||||
|
The number of symbols to decode is discovered
|
||||||
|
by tracking bitStream overflow condition.
|
||||||
|
When both states have overflowed the bitstream, end is reached.
|
||||||
|
|
||||||
|
|
||||||
|
##### Conversion from weights to huffman prefix codes
|
||||||
|
|
||||||
|
All present symbols shall now have a `weight` value.
|
||||||
|
A `weight` directly represents a `range` of prefix codes,
|
||||||
|
following the formulae : `range = weight ? 1 << (weight-1) : 0 ;`
|
||||||
|
Symbols are sorted by weight.
|
||||||
|
Within same weight, symbols keep natural order.
|
||||||
|
Starting from lowest weight,
|
||||||
|
symbols are being allocated to a range of prefix codes.
|
||||||
|
Symbols with a weight of zero are not present.
|
||||||
|
|
||||||
|
It is then possible to transform weights into nbBits :
|
||||||
|
`nbBits = nbBits ? maxBits + 1 - weight : 0;` .
|
||||||
|
|
||||||
|
|
||||||
|
__Example__ :
|
||||||
|
Let's presume the following huffman tree has been decoded :
|
||||||
|
|
||||||
|
| Literal | 0 | 1 | 2 | 3 | 4 | 5 |
|
||||||
|
| ------- | --- | --- | --- | --- | --- | --- |
|
||||||
|
| weight | 4 | 3 | 2 | 0 | 1 | 1 |
|
||||||
|
|
||||||
|
Sorted by weight and then natural order,
|
||||||
|
it gives the following distribution :
|
||||||
|
|
||||||
|
| Literal | 3 | 4 | 5 | 2 | 1 | 0 |
|
||||||
|
| ------------ | --- | --- | --- | --- | --- | ---- |
|
||||||
|
| weight | 0 | 1 | 1 | 2 | 3 | 4 |
|
||||||
|
| range | 0 | 1 | 1 | 2 | 4 | 8 |
|
||||||
|
| prefix codes | N/A | 0 | 1 | 2-3 | 4-7 | 8-15 |
|
||||||
|
| nb bits | 0 | 4 | 4 | 3 | 2 | 1 |
|
||||||
|
|
||||||
|
|
||||||
|
#### Literals bitstreams
|
||||||
|
|
||||||
|
##### Bitstreams sizes
|
||||||
|
|
||||||
|
As seen in a previous paragraph,
|
||||||
|
there are 2 flavors of huffman-compressed literals :
|
||||||
|
single stream, and 4-streams.
|
||||||
|
|
||||||
|
4-streams is useful for CPU with multiple execution units and OoO operations.
|
||||||
|
Since each stream can be decoded independently,
|
||||||
|
it's possible to decode them up to 4x faster than a single stream,
|
||||||
|
presuming the CPU has enough parallelism available.
|
||||||
|
|
||||||
|
For single stream, header provides both the compressed and regenerated size.
|
||||||
|
For 4-streams though,
|
||||||
|
header only provides compressed and regenerated size of all 4 streams combined.
|
||||||
|
In order to properly decode the 4 streams,
|
||||||
|
it's necessary to know the compressed and regenerated size of each stream.
|
||||||
|
|
||||||
|
Regenerated size is easiest :
|
||||||
|
each stream has a size of `(totalSize+3)/4`,
|
||||||
|
except the last one, which is up to 3 bytes smaller, to reach `totalSize`.
|
||||||
|
|
||||||
|
Compressed size must be provided explicitly : in the 4-streams variant,
|
||||||
|
bitstreams are preceded by 3 unsigned Little Endian 16-bits values.
|
||||||
|
Each value represents the compressed size of one stream, in order.
|
||||||
|
The last stream size is deducted from total compressed size
|
||||||
|
and from already known stream sizes :
|
||||||
|
`stream4CSize = totalCSize - 6 - stream1CSize - stream2CSize - stream3CSize;`
|
||||||
|
|
||||||
|
##### Bitstreams read and decode
|
||||||
|
|
||||||
|
Each bitstream must be read _backward_,
|
||||||
|
that is starting from the end down to the beginning.
|
||||||
|
Therefore it's necessary to know the size of each bitstream.
|
||||||
|
|
||||||
|
It's also necessary to know exactly which _bit_ is the latest.
|
||||||
|
This is detected by a final bit flag :
|
||||||
|
the highest bit of latest byte is a final-bit-flag.
|
||||||
|
Consequently, a last byte of `0` is not possible.
|
||||||
|
And the final-bit-flag itself is not part of the useful bitstream.
|
||||||
|
Hence, the last byte contain between 0 and 7 useful bits.
|
||||||
|
|
||||||
|
Starting from the end,
|
||||||
|
it's possible to read the bitstream in a little-endian fashion,
|
||||||
|
keeping track of already used bits.
|
||||||
|
|
||||||
|
Reading the last `maxBits` bits,
|
||||||
|
it's then possible to compare extracted value to the prefix codes table,
|
||||||
|
determining the symbol to decode and number of bits to discard.
|
||||||
|
|
||||||
|
The process continues up to reading the required number of symbols per stream.
|
||||||
|
If a bitstream is not entirely and exactly consumed,
|
||||||
|
hence reaching exactly its beginning position with all bits consumed,
|
||||||
|
the decoding process is considered faulty.
|
||||||
|
|
||||||
|
|
||||||
|
### Sequences section
|
||||||
|
|
||||||
|
A compressed block is a succession of _sequences_ .
|
||||||
|
A sequence is a literal copy command, followed by a match copy command.
|
||||||
|
A literal copy command specifies a length.
|
||||||
|
It is the number of bytes to be copied (or extracted) from the literal section.
|
||||||
|
A match copy command specifies an offset and a length.
|
||||||
|
The offset gives the position to copy from,
|
||||||
|
which can stand within a previous block.
|
||||||
|
|
||||||
|
There are 3 symbol types, `literalLength`, `matchLength` and `offset`,
|
||||||
|
which are encoded together, interleaved in a single _bitstream_.
|
||||||
|
|
||||||
|
Each symbol is a _code_ in its own context,
|
||||||
|
which specifies a baseline and a number of bits to add.
|
||||||
|
_Codes_ are FSE compressed,
|
||||||
|
and interleaved with raw additional bits in the same bitstream.
|
||||||
|
|
||||||
|
The Sequences section starts by a header,
|
||||||
|
followed by optional Probability tables for each symbol type,
|
||||||
|
followed by the bitstream.
|
||||||
|
|
||||||
|
| Header | (LitLengthTable) | (OffsetTable) | (MatchLengthTable) | bitStream |
|
||||||
|
| ------ | ---------------- | ------------- | ------------------ | --------- |
|
||||||
|
|
||||||
|
To decode the Sequence section, it's required to know its size.
|
||||||
|
This size is deducted from `blockSize - literalSectionSize`.
|
||||||
|
|
||||||
|
|
||||||
|
#### Sequences section header
|
||||||
|
|
||||||
|
Consists in 2 items :
|
||||||
|
- Nb of Sequences
|
||||||
|
- Flags providing Symbol compression types
|
||||||
|
|
||||||
|
__Nb of Sequences__
|
||||||
|
|
||||||
|
This is a variable size field, `nbSeqs`, using between 1 and 3 bytes.
|
||||||
|
Let's call its first byte `byte0`.
|
||||||
|
- `if (byte0 == 0)` : there are no sequences.
|
||||||
|
The sequence section stops there.
|
||||||
|
Regenerated content is defined entirely by literals section.
|
||||||
|
- `if (byte0 < 128)` : `nbSeqs = byte0;` . Uses 1 byte.
|
||||||
|
- `if (byte0 < 255)` : `nbSeqs = ((byte0-128) << 8) + byte1;` . Uses 2 bytes.
|
||||||
|
- `if (byte0 == 255)`: `nbSeqs = byte1 + (byte2<<8) + 0x7F00;` . Uses 3 bytes.
|
||||||
|
|
||||||
|
__Symbol compression modes__
|
||||||
|
|
||||||
|
This is a single byte, defining the compression mode of each symbol type.
|
||||||
|
|
||||||
|
| BitNb | 7-6 | 5-4 | 3-2 | 1-0 |
|
||||||
|
| ------- | ------ | ------ | ------ | -------- |
|
||||||
|
|FieldName| LLtype | OFType | MLType | Reserved |
|
||||||
|
|
||||||
|
The last field, `Reserved`, must be all-zeroes.
|
||||||
|
|
||||||
|
`LLtype`, `OFType` and `MLType` define the compression mode of
|
||||||
|
Literal Lengths, Offsets and Match Lengths respectively.
|
||||||
|
|
||||||
|
They follow the same enumeration :
|
||||||
|
|
||||||
|
| Value | 0 | 1 | 2 | 3 |
|
||||||
|
| ---------------- | ------ | --- | ------ | --- |
|
||||||
|
| Compression Mode | predef | RLE | Repeat | FSE |
|
||||||
|
|
||||||
|
- "predef" : uses a pre-defined distribution table.
|
||||||
|
- "RLE" : it's a single code, repeated `nbSeqs` times.
|
||||||
|
- "Repeat" : re-use distribution table from previous compressed block.
|
||||||
|
- "FSE" : standard FSE compression.
|
||||||
|
A distribution table will be present.
|
||||||
|
It will be described in [next part](#distribution-tables).
|
||||||
|
|
||||||
|
#### Symbols decoding
|
||||||
|
|
||||||
|
##### Literal Lengths codes
|
||||||
|
|
||||||
|
Literal lengths codes are values ranging from `0` to `35` included.
|
||||||
|
They define lengths from 0 to 131071 bytes.
|
||||||
|
|
||||||
|
| Code | 0-15 |
|
||||||
|
| ------ | ---- |
|
||||||
|
| length | Code |
|
||||||
|
| nbBits | 0 |
|
||||||
|
|
||||||
|
|
||||||
|
| Code | 16 | 17 | 18 | 19 | 20 | 21 | 22 | 23 |
|
||||||
|
| -------- | ---- | ---- | ---- | ---- | ---- | ---- | ---- | ---- |
|
||||||
|
| Baseline | 16 | 18 | 20 | 22 | 24 | 28 | 32 | 40 |
|
||||||
|
| nb Bits | 1 | 1 | 1 | 1 | 2 | 2 | 3 | 3 |
|
||||||
|
|
||||||
|
| Code | 24 | 25 | 26 | 27 | 28 | 29 | 30 | 31 |
|
||||||
|
| -------- | ---- | ---- | ---- | ---- | ---- | ---- | ---- | ---- |
|
||||||
|
| Baseline | 48 | 64 | 128 | 256 | 512 | 1024 | 2048 | 4096 |
|
||||||
|
| nb Bits | 4 | 6 | 7 | 8 | 9 | 10 | 11 | 12 |
|
||||||
|
|
||||||
|
| Code | 32 | 33 | 34 | 35 |
|
||||||
|
| -------- | ---- | ---- | ---- | ---- |
|
||||||
|
| Baseline | 8192 |16384 |32768 |65536 |
|
||||||
|
| nb Bits | 13 | 14 | 15 | 16 |
|
||||||
|
|
||||||
|
__Default distribution__
|
||||||
|
|
||||||
|
When "compression mode" is "predef"",
|
||||||
|
a pre-defined distribution is used for FSE compression.
|
||||||
|
|
||||||
|
Below is its definition. It uses an accuracy of 6 bits (64 states).
|
||||||
|
```
|
||||||
|
short literalLengths_defaultDistribution[36] =
|
||||||
|
{ 4, 3, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 1, 1, 1,
|
||||||
|
2, 2, 2, 2, 2, 2, 2, 2, 2, 3, 2, 1, 1, 1, 1, 1,
|
||||||
|
-1,-1,-1,-1 };
|
||||||
|
```
|
||||||
|
|
||||||
|
##### Match Lengths codes
|
||||||
|
|
||||||
|
Match lengths codes are values ranging from `0` to `52` included.
|
||||||
|
They define lengths from 3 to 131074 bytes.
|
||||||
|
|
||||||
|
| Code | 0-31 |
|
||||||
|
| ------ | -------- |
|
||||||
|
| value | Code + 3 |
|
||||||
|
| nbBits | 0 |
|
||||||
|
|
||||||
|
| Code | 32 | 33 | 34 | 35 | 36 | 37 | 38 | 39 |
|
||||||
|
| -------- | ---- | ---- | ---- | ---- | ---- | ---- | ---- | ---- |
|
||||||
|
| Baseline | 35 | 37 | 39 | 41 | 43 | 47 | 51 | 59 |
|
||||||
|
| nb Bits | 1 | 1 | 1 | 1 | 2 | 2 | 3 | 3 |
|
||||||
|
|
||||||
|
| Code | 40 | 41 | 42 | 43 | 44 | 45 | 46 | 47 |
|
||||||
|
| -------- | ---- | ---- | ---- | ---- | ---- | ---- | ---- | ---- |
|
||||||
|
| Baseline | 67 | 83 | 99 | 131 | 258 | 514 | 1026 | 2050 |
|
||||||
|
| nb Bits | 4 | 4 | 5 | 7 | 8 | 9 | 10 | 11 |
|
||||||
|
|
||||||
|
| Code | 48 | 49 | 50 | 51 | 52 |
|
||||||
|
| -------- | ---- | ---- | ---- | ---- | ---- |
|
||||||
|
| Baseline | 4098 | 8194 |16486 |32770 |65538 |
|
||||||
|
| nb Bits | 12 | 13 | 14 | 15 | 16 |
|
||||||
|
|
||||||
|
__Default distribution__
|
||||||
|
|
||||||
|
When "compression mode" is defined as "predef",
|
||||||
|
a pre-defined distribution is used for FSE compression.
|
||||||
|
|
||||||
|
Here is its definition. It uses an accuracy of 6 bits (64 states).
|
||||||
|
```
|
||||||
|
short matchLengths_defaultDistribution[53] =
|
||||||
|
{ 1, 4, 3, 2, 2, 2, 2, 2, 2, 1, 1, 1, 1, 1, 1, 1,
|
||||||
|
1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
|
||||||
|
1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,-1,-1,
|
||||||
|
-1,-1,-1,-1,-1 };
|
||||||
|
```
|
||||||
|
|
||||||
|
##### Offset codes
|
||||||
|
|
||||||
|
Offset codes are values ranging from `0` to `N`,
|
||||||
|
with `N` being limited by maximum backreference distance.
|
||||||
|
|
||||||
|
A decoder is free to limit its maximum `N` supported.
|
||||||
|
Recommendation is to support at least up to `22`.
|
||||||
|
For information, at the time of this writing.
|
||||||
|
the reference decoder supports a maximum `N` value of `28` in 64-bits mode.
|
||||||
|
|
||||||
|
An offset code is also the nb of additional bits to read,
|
||||||
|
and can be translated into an `OFValue` using the following formulae :
|
||||||
|
|
||||||
|
```
|
||||||
|
OFValue = (1 << offsetCode) + readNBits(offsetCode);
|
||||||
|
if (OFValue > 3) offset = OFValue - 3;
|
||||||
|
```
|
||||||
|
|
||||||
|
OFValue from 1 to 3 are special : they define "repeat codes",
|
||||||
|
which means one of the previous offsets will be repeated.
|
||||||
|
They are sorted in recency order, with 1 meaning the most recent one.
|
||||||
|
See [Repeat offsets](#repeat-offsets) paragraph.
|
||||||
|
|
||||||
|
__Default distribution__
|
||||||
|
|
||||||
|
When "compression mode" is defined as "predef",
|
||||||
|
a pre-defined distribution is used for FSE compression.
|
||||||
|
|
||||||
|
Here is its definition. It uses an accuracy of 5 bits (32 states),
|
||||||
|
and supports a maximum `N` of 28, allowing offset values up to 536,870,908 .
|
||||||
|
|
||||||
|
If any sequence in the compressed block requires an offset larger than this,
|
||||||
|
it's not possible to use the default distribution to represent it.
|
||||||
|
|
||||||
|
```
|
||||||
|
short offsetCodes_defaultDistribution[53] =
|
||||||
|
{ 1, 1, 1, 1, 1, 1, 2, 2, 2, 1, 1, 1, 1, 1, 1, 1,
|
||||||
|
1, 1, 1, 1, 1, 1, 1, 1,-1,-1,-1,-1,-1 };
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Distribution tables
|
||||||
|
|
||||||
|
Following the header, up to 3 distribution tables can be described.
|
||||||
|
They are, in order :
|
||||||
|
- Literal lengthes
|
||||||
|
- Offsets
|
||||||
|
- Match Lengthes
|
||||||
|
|
||||||
|
The content to decode depends on their respective compression mode :
|
||||||
|
- Repeat mode : no content. Re-use distribution from previous compressed block.
|
||||||
|
- Predef : no content. Use pre-defined distribution table.
|
||||||
|
- RLE : 1 byte. This is the only code to use across the whole compressed block.
|
||||||
|
- FSE : A distribution table is present.
|
||||||
|
|
||||||
|
##### FSE distribution table : condensed format
|
||||||
|
|
||||||
|
An FSE distribution table describes the probabilities of all symbols
|
||||||
|
from `0` to the last present one (included)
|
||||||
|
on a normalized scale of `1 << AccuracyLog` .
|
||||||
|
|
||||||
|
It's a bitstream which is read forward, in little-endian fashion.
|
||||||
|
It's not necessary to know its exact size,
|
||||||
|
since it will be discovered and reported by the decoding process.
|
||||||
|
|
||||||
|
The bitstream starts by reporting on which scale it operates.
|
||||||
|
`AccuracyLog = low4bits + 5;`
|
||||||
|
In theory, it can define a scale from 5 to 20.
|
||||||
|
In practice, decoders are allowed to limit the maximum supported `AccuracyLog`.
|
||||||
|
Recommended maximum are `9` for literal and match lengthes, and `8` for offsets.
|
||||||
|
The reference decoder uses these limits.
|
||||||
|
|
||||||
|
Then follow each symbol value, from `0` to last present one.
|
||||||
|
The nb of bits used by each field is variable.
|
||||||
|
It depends on :
|
||||||
|
|
||||||
|
- Remaining probabilities + 1 :
|
||||||
|
__example__ :
|
||||||
|
Presuming an AccuracyLog of 8,
|
||||||
|
and presuming 100 probabilities points have already been distributed,
|
||||||
|
the decoder may read any value from `0` to `255 - 100 + 1 == 156` (included).
|
||||||
|
Therefore, it must read `log2sup(156) == 8` bits.
|
||||||
|
|
||||||
|
- Value decoded : small values use 1 less bit :
|
||||||
|
__example__ :
|
||||||
|
Presuming values from 0 to 156 (included) are possible,
|
||||||
|
255-156 = 99 values are remaining in an 8-bits field.
|
||||||
|
They are used this way :
|
||||||
|
first 99 values (hence from 0 to 98) use only 7 bits,
|
||||||
|
values from 99 to 156 use 8 bits.
|
||||||
|
This is achieved through this scheme :
|
||||||
|
|
||||||
|
| Value read | Value decoded | nb Bits used |
|
||||||
|
| ---------- | ------------- | ------------ |
|
||||||
|
| 0 - 98 | 0 - 98 | 7 |
|
||||||
|
| 99 - 127 | 99 - 127 | 8 |
|
||||||
|
| 128 - 226 | 0 - 98 | 7 |
|
||||||
|
| 227 - 255 | 128 - 156 | 8 |
|
||||||
|
|
||||||
|
Symbols probabilities are read one by one, in order.
|
||||||
|
|
||||||
|
Probability is obtained from Value decoded by following formulae :
|
||||||
|
`Proba = value - 1;`
|
||||||
|
|
||||||
|
It means value `0` becomes negative probability `-1`.
|
||||||
|
`-1` is a special probability, which means `less than 1`.
|
||||||
|
Its effect on distribution table is described in [next paragraph].
|
||||||
|
For the purpose of calculating cumulated distribution, it counts as one.
|
||||||
|
|
||||||
|
[next paragraph]:#fse-decoding--from-normalized-distribution-to-decoding-tables
|
||||||
|
|
||||||
|
When a symbol has a probability of `zero`,
|
||||||
|
it is followed by a 2-bits repeat flag.
|
||||||
|
This repeat flag tells how many probabilities of zeroes follow the current one.
|
||||||
|
It provides a number ranging from 0 to 3.
|
||||||
|
If it is a 3, another 2-bits repeat flag follows, and so on.
|
||||||
|
|
||||||
|
When last symbol reaches cumulated total of `1 << AccuracyLog`,
|
||||||
|
decoding is complete.
|
||||||
|
Then the decoder can tell how many bytes were used in this process,
|
||||||
|
and how many symbols are present.
|
||||||
|
|
||||||
|
The bitstream consumes a round number of bytes.
|
||||||
|
Any remaining bit within the last byte is just unused.
|
||||||
|
|
||||||
|
If the last symbol makes cumulated total go above `1 << AccuracyLog`,
|
||||||
|
distribution is considered corrupted.
|
||||||
|
|
||||||
|
##### FSE decoding : from normalized distribution to decoding tables
|
||||||
|
|
||||||
|
The distribution of normalized probabilities is enough
|
||||||
|
to create a unique decoding table.
|
||||||
|
|
||||||
|
It follows the following build rule :
|
||||||
|
|
||||||
|
The table has a size of `tableSize = 1 << AccuracyLog;`.
|
||||||
|
Each cell describes the symbol decoded,
|
||||||
|
and instructions to get the next state.
|
||||||
|
|
||||||
|
Symbols are scanned in their natural order for `less than 1` probabilities.
|
||||||
|
Symbols with this probability are being attributed a single cell,
|
||||||
|
starting from the end of the table.
|
||||||
|
These symbols define a full state reset, reading `AccuracyLog` bits.
|
||||||
|
|
||||||
|
All remaining symbols are sorted in their natural order.
|
||||||
|
Starting from symbol `0` and table position `0`,
|
||||||
|
each symbol gets attributed as many cells as its probability.
|
||||||
|
Cell allocation is spreaded, not linear :
|
||||||
|
each successor position follow this rule :
|
||||||
|
|
||||||
|
```
|
||||||
|
position += (tableSize>>1) + (tableSize>>3) + 3;
|
||||||
|
position &= tableSize-1;
|
||||||
|
```
|
||||||
|
|
||||||
|
A position is skipped if already occupied,
|
||||||
|
typically by a "less than 1" probability symbol.
|
||||||
|
|
||||||
|
The result is a list of state values.
|
||||||
|
Each state will decode the current symbol.
|
||||||
|
|
||||||
|
To get the Number of bits and baseline required for next state,
|
||||||
|
it's first necessary to sort all states in their natural order.
|
||||||
|
The lower states will need 1 more bit than higher ones.
|
||||||
|
|
||||||
|
__Example__ :
|
||||||
|
Presuming a symbol has a probability of 5.
|
||||||
|
It receives 5 state values. States are sorted in natural order.
|
||||||
|
|
||||||
|
Next power of 2 is 8.
|
||||||
|
Space of probabilities is divided into 8 equal parts.
|
||||||
|
Presuming the AccuracyLog is 7, it defines 128 states.
|
||||||
|
Divided by 8, each share is 16 large.
|
||||||
|
|
||||||
|
In order to reach 8, 8-5=3 lowest states will count "double",
|
||||||
|
taking shares twice larger,
|
||||||
|
requiring one more bit in the process.
|
||||||
|
|
||||||
|
Numbering starts from higher states using less bits.
|
||||||
|
|
||||||
|
| state order | 0 | 1 | 2 | 3 | 4 |
|
||||||
|
| ----------- | ----- | ----- | ------ | ---- | ----- |
|
||||||
|
| width | 32 | 32 | 32 | 16 | 16 |
|
||||||
|
| nb Bits | 5 | 5 | 5 | 4 | 4 |
|
||||||
|
| range nb | 2 | 4 | 6 | 0 | 1 |
|
||||||
|
| baseline | 32 | 64 | 96 | 0 | 16 |
|
||||||
|
| range | 32-63 | 64-95 | 96-127 | 0-15 | 16-31 |
|
||||||
|
|
||||||
|
Next state is determined from current state
|
||||||
|
by reading the required number of bits, and adding the specified baseline.
|
||||||
|
|
||||||
|
|
||||||
|
#### Bitstream
|
||||||
|
|
||||||
|
All sequences are stored in a single bitstream, read _backward_.
|
||||||
|
It is therefore necessary to know the bitstream size,
|
||||||
|
which is deducted from compressed block size.
|
||||||
|
|
||||||
|
The last useful bit of the stream is followed by an end-bit-flag.
|
||||||
|
Highest bit of last byte is this flag.
|
||||||
|
It does not belong to the useful part of the bitstream.
|
||||||
|
Therefore, last byte has 0-7 useful bits.
|
||||||
|
Note that it also means that last byte cannot be `0`.
|
||||||
|
|
||||||
|
##### Starting states
|
||||||
|
|
||||||
|
The bitstream starts with initial state values,
|
||||||
|
each using the required number of bits in their respective _accuracy_,
|
||||||
|
decoded previously from their normalized distribution.
|
||||||
|
|
||||||
|
It starts by `Literal Length State`,
|
||||||
|
followed by `Offset State`,
|
||||||
|
and finally `Match Length State`.
|
||||||
|
|
||||||
|
Reminder : always keep in mind that all values are read _backward_.
|
||||||
|
|
||||||
|
##### Decoding a sequence
|
||||||
|
|
||||||
|
A state gives a code.
|
||||||
|
A code provides a baseline and number of bits to add.
|
||||||
|
See [Symbol Decoding] section for details on each symbol.
|
||||||
|
|
||||||
|
Decoding starts by reading the nb of bits required to decode offset.
|
||||||
|
It then does the same for match length,
|
||||||
|
and then for literal length.
|
||||||
|
|
||||||
|
Offset / matchLength / litLength define a sequence.
|
||||||
|
It starts by inserting the number of literals defined by `litLength`,
|
||||||
|
then continue by copying `matchLength` bytes from `currentPos - offset`.
|
||||||
|
|
||||||
|
The next operation is to update states.
|
||||||
|
Using rules pre-calculated in the decoding tables,
|
||||||
|
`Literal Length State` is updated,
|
||||||
|
followed by `Match Length State`,
|
||||||
|
and then `Offset State`.
|
||||||
|
|
||||||
|
This operation will be repeated `NbSeqs` times.
|
||||||
|
At the end, the bitstream shall be entirely consumed,
|
||||||
|
otherwise bitstream is considered corrupted.
|
||||||
|
|
||||||
|
[Symbol Decoding]:#symbols-decoding
|
||||||
|
|
||||||
|
##### Repeat offsets
|
||||||
|
|
||||||
|
As seen in [Offset Codes], the first 3 values define a repeated offset.
|
||||||
|
They are sorted in recency order, with 1 meaning "most recent one".
|
||||||
|
|
||||||
|
There is an exception though, when current sequence's literal length is `0`.
|
||||||
|
In which case, 1 would just make previous match longer.
|
||||||
|
Therefore, in such case, 1 means in fact 2, and 2 is impossible.
|
||||||
|
Meaning of 3 is unmodified.
|
||||||
|
|
||||||
|
Repeat offsets start with the following values : 1, 4 and 8 (in order).
|
||||||
|
|
||||||
|
Then each block receives its start value from previous compressed block.
|
||||||
|
Note that non-compressed blocks are skipped,
|
||||||
|
they do not contribute to offset history.
|
||||||
|
|
||||||
|
[Offset Codes]: #offset-codes
|
||||||
|
|
||||||
|
###### Offset updates rules
|
||||||
|
|
||||||
|
When the new offset is a normal one,
|
||||||
|
offset history is simply translated by one position,
|
||||||
|
with the new offset taking first spot.
|
||||||
|
|
||||||
|
- When repeat offset 1 (most recent) is used, history is unmodified.
|
||||||
|
- When repeat offset 2 is used, it's swapped with offset 1.
|
||||||
|
- When repeat offset 3 is used, it takes first spot,
|
||||||
|
pushing the other ones by one position.
|
||||||
|
|
||||||
|
|
||||||
|
Dictionary format
|
||||||
|
-----------------
|
||||||
|
|
||||||
|
`zstd` is compatible with "pure content" dictionaries, free of any format restriction.
|
||||||
|
But dictionaries created by `zstd --train` follow a format, described here.
|
||||||
|
|
||||||
|
__Pre-requisites__ : a dictionary has a known length,
|
||||||
|
defined either by a buffer limit, or a file size.
|
||||||
|
|
||||||
|
| Header | DictID | Stats | Content |
|
||||||
|
| ------ | ------ | ----- | ------- |
|
||||||
|
|
||||||
|
__Header__ : 4 bytes ID, value 0xEC30A437, Little Endian format
|
||||||
|
|
||||||
|
__Dict_ID__ : 4 bytes, stored in Little Endian format.
|
||||||
|
DictID can be any value, except 0 (which means no DictID).
|
||||||
|
It's used by decoders to check if they use the correct dictionary.
|
||||||
|
|
||||||
|
__Stats__ : Entropy tables, following the same format as a [compressed blocks].
|
||||||
|
They are stored in following order :
|
||||||
|
Huffman tables for literals, FSE table for offset,
|
||||||
|
FSE table for matchLenth, and FSE table for litLength.
|
||||||
|
It's finally followed by 3 offset values, populating recent offsets,
|
||||||
|
stored in order, 4-bytes little endian each, for a total of 12 bytes.
|
||||||
|
|
||||||
|
__Content__ : Where the actual dictionary content is.
|
||||||
|
Content size depends on Dictionary size.
|
||||||
|
|
||||||
|
[compressed blocks]: #compressed-block-format
|
||||||
|
|
||||||
|
|
||||||
|
Version changes
|
||||||
|
---------------
|
||||||
|
0.1.0 initial release
|
||||||
Reference in New Issue
Block a user