diff --git a/CHANGELOG.md b/CHANGELOG.md index eb5bab9..e602d5d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- `wpMur3HasherIO` to setup the hash IO struct +- `wpX64Mur3Hasher128` implementation for the x64 128-bit variant of the MurmurHash3 algorithm +- `wpMiscUtilsRotl64` utility + ## [2.4.0] - 2026-08-30 ### Added diff --git a/src/base/hash/hasher/hasher.h b/src/base/hash/hasher/hasher.h new file mode 100644 index 0000000..c03301a --- /dev/null +++ b/src/base/hash/hasher/hasher.h @@ -0,0 +1,18 @@ +// vim:fileencoding=utf-8:foldmethod=marker + +#ifndef HASHER_H +#define HASHER_H + +#include "../../stream/stream.h" + +#ifdef WP_PLATFORM_CPP +BEGIN_C_LINKAGE +#endif // !WP_PLATFORM_CPP + +typedef void (*WpHasher)(WpU8Stream *stream, void *hasher_io); + +#ifdef WP_PLATFORM_CPP +END_C_LINKAGE +#endif // !WP_PLATFORM_CPP + +#endif // !HASHER_H diff --git a/src/base/hash/hasher/murmur3.c b/src/base/hash/hasher/murmur3.c new file mode 100644 index 0000000..b581baf --- /dev/null +++ b/src/base/hash/hasher/murmur3.c @@ -0,0 +1,105 @@ +// vim:fileencoding=utf-8:foldmethod=marker + +#include "murmur3.h" +#include "../../../common/aliases/aliases.h" +#include "../../../common/misc/misc_utils.h" + +#define _bigConstant(x) (x##LLU) + +wp_intern inline u64 fmix64(u64 k); + +void wpX64Mur3Hasher128(WpU8Stream *bytes, void *hasher_io) { + WpMur3HasherIO *io = (WpMur3HasherIO *)hasher_io; + + const i32 nblocks = (bytes->count * bytes->item_size) / 16; + + u64 h1 = io->seed; + u64 h2 = io->seed; + + const u64 c1 = _bigConstant(0x87c37b91114253d5); + const u64 c2 = _bigConstant(0x4cf5ad432745937f); + + //---------- + // body + + for(int i = 0; i < nblocks; ++i) { + u64 k1 = *((u64 *)wpStreamConsumeCount(u8, bytes, sizeof(u64))); + u64 k2 = *((u64 *)wpStreamConsumeCount(u8, bytes, sizeof(u64))); + + k1 *= c1; + k1 = wpMiscUtilsRotl64(k1, 31); + k1 *= c2; + h1 ^= k1; + + h1 = wpMiscUtilsRotl64(h1, 27); + h1 += h2; + h1 = h1 * 5 +0x52dce729; + + k2 *= c2; + k2 = wpMiscUtilsRotl64(k2, 33); + k2 *= c1; + h2 ^= k2; + + h2 = wpMiscUtilsRotl64(h2, 31); + h2 += h1; + h2 = h2*5+0x38495ab5; + } + + //---------- + // tail + + const u8 *tail = wpStreamConsumeCount(u8, bytes, bytes->count - bytes->position); + + u64 k1 = 0; + u64 k2 = 0; + + u64 len = bytes->count * bytes->item_size; + + switch(len & 15) { + case 15: k2 ^= ((u64)tail[14]) << 48; + case 14: k2 ^= ((u64)tail[13]) << 40; + case 13: k2 ^= ((u64)tail[12]) << 32; + case 12: k2 ^= ((u64)tail[11]) << 24; + case 11: k2 ^= ((u64)tail[10]) << 16; + case 10: k2 ^= ((u64)tail[ 9]) << 8; + case 9: k2 ^= ((u64)tail[ 8]) << 0; + k2 *= c2; k2 = wpMiscUtilsRotl64(k2,33); k2 *= c1; h2 ^= k2; + + case 8: k1 ^= ((u64)tail[ 7]) << 56; + case 7: k1 ^= ((u64)tail[ 6]) << 48; + case 6: k1 ^= ((u64)tail[ 5]) << 40; + case 5: k1 ^= ((u64)tail[ 4]) << 32; + case 4: k1 ^= ((u64)tail[ 3]) << 24; + case 3: k1 ^= ((u64)tail[ 2]) << 16; + case 2: k1 ^= ((u64)tail[ 1]) << 8; + case 1: k1 ^= ((u64)tail[ 0]) << 0; + k1 *= c1; k1 = wpMiscUtilsRotl64(k1,31); k1 *= c2; h1 ^= k1; + }; + + //---------- + // finalization + + h1 ^= len; h2 ^= len; + + h1 += h2; + h2 += h1; + + h1 = fmix64(h1); + h2 = fmix64(h2); + + h1 += h2; + h2 += h1; + + io->out_hash1 = h1; + io->out_hash2 = h2; +} + +wp_intern inline u64 fmix64(u64 k) { + k ^= k >> 33; + k *= _bigConstant(0xff51afd7ed558ccd); + k ^= k >> 33; + k *= _bigConstant(0xc4ceb9fe1a85ec53); + k ^= k >> 33; + + return k; +} diff --git a/src/base/hash/hasher/murmur3.h b/src/base/hash/hasher/murmur3.h new file mode 100644 index 0000000..dfc8d38 --- /dev/null +++ b/src/base/hash/hasher/murmur3.h @@ -0,0 +1,40 @@ +// vim:fileencoding=utf-8:foldmethod=marker + +/** + * An implementation of the x64 128-bit variant of the MurmurHash3 algorithm + * based on https://github.com/aappleby/smhasher/blob/master/src/MurmurHash3.cpp + */ + +#ifndef MURMUR3_H +#define MURMUR3_H + +#include "../../stream/stream.h" +#include "../../../common/aliases/aliases.h" +#include "../../../common/platform/platform.h" + +#ifdef WP_PLATFORM_CPP +BEGIN_C_LINKAGE +#endif // !WP_PLATFORM_CPP + +#define WP_MUR3_HASHER_MAGIC ((u64)0x57504d4841534833) + +typedef struct { + u64 magic; + u64 out_hash1; + u64 out_hash2; + u32 seed; +} WpMur3HasherIO; + +#ifdef WP_PLATFORM_CPP +#define wpMur3HasherIO(SEED) (WpMur3HasherIO{WP_MUR3_HASHER_MAGIC, 0, 0, SEED}) +#else +#define wpMur3HasherIO(SEED) ((WpMur3HasherIO){ .magic = WP_MUR3_HASHER_MAGIC, .seed = SEED }) +#endif + +void wpX64Mur3Hasher128(WpU8Stream *bytes, void *hasher_io); + +#ifdef WP_PLATFORM_CPP +END_C_LINKAGE +#endif // !WP_PLATFORM_CPP + +#endif // !MURMUR3_H diff --git a/src/base/wapp_base.c b/src/base/wapp_base.c index eeb3369..57a90d2 100644 --- a/src/base/wapp_base.c +++ b/src/base/wapp_base.c @@ -6,6 +6,7 @@ #include "wapp_base.h" #include "array/array.c" #include "dbl_list/dbl_list.c" +#include "hash/hasher/murmur3.c" #include "queue/queue.c" #include "stream/stream.c" #include "mem/allocator/mem_allocator.c" diff --git a/src/base/wapp_base.h b/src/base/wapp_base.h index c7be92f..1fdc9e6 100644 --- a/src/base/wapp_base.h +++ b/src/base/wapp_base.h @@ -5,6 +5,8 @@ #include "array/array.h" #include "dbl_list/dbl_list.h" +#include "hash/hasher/hasher.h" +#include "hash/hasher/murmur3.h" #include "queue/queue.h" #include "stream/stream.h" #include "mem/allocator/mem_allocator.h" diff --git a/src/common/misc/misc_utils.h b/src/common/misc/misc_utils.h index 1b9f34f..e1e375e 100644 --- a/src/common/misc/misc_utils.h +++ b/src/common/misc/misc_utils.h @@ -50,6 +50,10 @@ BEGIN_C_LINKAGE #define wpMiscUtilsIsPowerOfTwo(NUM) ((NUM & (NUM - 1)) == 0) #define wpMiscUtilsOffsetPointer(PTR, OFFSET) ((void *)((uptr)(PTR) + (OFFSET))) +wp_intern inline u64 wpMiscUtilsRotl64(u64 x, i8 rotation) { + return (x << rotation) | (x >> (64 - rotation)); +} + #ifdef WP_PLATFORM_CPP END_C_LINKAGE diff --git a/tests/hasher/MurmurHash3.c b/tests/hasher/MurmurHash3.c new file mode 100644 index 0000000..2393fa9 --- /dev/null +++ b/tests/hasher/MurmurHash3.c @@ -0,0 +1,335 @@ +//----------------------------------------------------------------------------- +// MurmurHash3 was written by Austin Appleby, and is placed in the public +// domain. The author hereby disclaims copyright to this source code. + +// Note - The x86 and x64 versions do _not_ produce the same results, as the +// algorithms are optimized for their respective platforms. You can still +// compile and run any of them on any platform, but your performance with the +// non-native version will be less than optimal. + +#include "MurmurHash3.h" + +//----------------------------------------------------------------------------- +// Platform-specific functions and macros + +// Microsoft Visual Studio + +#if defined(_MSC_VER) + +#define FORCE_INLINE __forceinline + +#include + +#define ROTL32(x,y) _rotl(x,y) +#define ROTL64(x,y) _rotl64(x,y) + +#define BIG_CONSTANT(x) (x) + +// Other compilers + +#else // defined(_MSC_VER) + +#define FORCE_INLINE inline __attribute__((always_inline)) + +static inline uint32_t rotl32 ( uint32_t x, int8_t r ) +{ + return (x << r) | (x >> (32 - r)); +} + +static inline uint64_t rotl64 ( uint64_t x, int8_t r ) +{ + return (x << r) | (x >> (64 - r)); +} + +#define ROTL32(x,y) rotl32(x,y) +#define ROTL64(x,y) rotl64(x,y) + +#define BIG_CONSTANT(x) (x##LLU) + +#endif // !defined(_MSC_VER) + +//----------------------------------------------------------------------------- +// Block read - if your platform needs to do endian-swapping or can only +// handle aligned reads, do the conversion here + +FORCE_INLINE uint32_t getblock32 ( const uint32_t * p, int i ) +{ + return p[i]; +} + +FORCE_INLINE uint64_t getblock64 ( const uint64_t * p, int i ) +{ + return p[i]; +} + +//----------------------------------------------------------------------------- +// Finalization mix - force all bits of a hash block to avalanche + +FORCE_INLINE uint32_t fmix32 ( uint32_t h ) +{ + h ^= h >> 16; + h *= 0x85ebca6b; + h ^= h >> 13; + h *= 0xc2b2ae35; + h ^= h >> 16; + + return h; +} + +//---------- + +FORCE_INLINE uint64_t fmix64 ( uint64_t k ) +{ + k ^= k >> 33; + k *= BIG_CONSTANT(0xff51afd7ed558ccd); + k ^= k >> 33; + k *= BIG_CONSTANT(0xc4ceb9fe1a85ec53); + k ^= k >> 33; + + return k; +} + +//----------------------------------------------------------------------------- + +void MurmurHash3_x86_32 ( const void * key, int len, + uint32_t seed, void * out ) +{ + const uint8_t * data = (const uint8_t*)key; + const int nblocks = len / 4; + + uint32_t h1 = seed; + + const uint32_t c1 = 0xcc9e2d51; + const uint32_t c2 = 0x1b873593; + + //---------- + // body + + const uint32_t * blocks = (const uint32_t *)(data + nblocks*4); + + for(int i = -nblocks; i; i++) + { + uint32_t k1 = getblock32(blocks,i); + + k1 *= c1; + k1 = ROTL32(k1,15); + k1 *= c2; + + h1 ^= k1; + h1 = ROTL32(h1,13); + h1 = h1*5+0xe6546b64; + } + + //---------- + // tail + + const uint8_t * tail = (const uint8_t*)(data + nblocks*4); + + uint32_t k1 = 0; + + switch(len & 3) + { + case 3: k1 ^= tail[2] << 16; + case 2: k1 ^= tail[1] << 8; + case 1: k1 ^= tail[0]; + k1 *= c1; k1 = ROTL32(k1,15); k1 *= c2; h1 ^= k1; + }; + + //---------- + // finalization + + h1 ^= len; + + h1 = fmix32(h1); + + *(uint32_t*)out = h1; +} + +//----------------------------------------------------------------------------- + +void MurmurHash3_x86_128 ( const void * key, const int len, + uint32_t seed, void * out ) +{ + const uint8_t * data = (const uint8_t*)key; + const int nblocks = len / 16; + + uint32_t h1 = seed; + uint32_t h2 = seed; + uint32_t h3 = seed; + uint32_t h4 = seed; + + const uint32_t c1 = 0x239b961b; + const uint32_t c2 = 0xab0e9789; + const uint32_t c3 = 0x38b34ae5; + const uint32_t c4 = 0xa1e38b93; + + //---------- + // body + + const uint32_t * blocks = (const uint32_t *)(data + nblocks*16); + + for(int i = -nblocks; i; i++) + { + uint32_t k1 = getblock32(blocks,i*4+0); + uint32_t k2 = getblock32(blocks,i*4+1); + uint32_t k3 = getblock32(blocks,i*4+2); + uint32_t k4 = getblock32(blocks,i*4+3); + + k1 *= c1; k1 = ROTL32(k1,15); k1 *= c2; h1 ^= k1; + + h1 = ROTL32(h1,19); h1 += h2; h1 = h1*5+0x561ccd1b; + + k2 *= c2; k2 = ROTL32(k2,16); k2 *= c3; h2 ^= k2; + + h2 = ROTL32(h2,17); h2 += h3; h2 = h2*5+0x0bcaa747; + + k3 *= c3; k3 = ROTL32(k3,17); k3 *= c4; h3 ^= k3; + + h3 = ROTL32(h3,15); h3 += h4; h3 = h3*5+0x96cd1c35; + + k4 *= c4; k4 = ROTL32(k4,18); k4 *= c1; h4 ^= k4; + + h4 = ROTL32(h4,13); h4 += h1; h4 = h4*5+0x32ac3b17; + } + + //---------- + // tail + + const uint8_t * tail = (const uint8_t*)(data + nblocks*16); + + uint32_t k1 = 0; + uint32_t k2 = 0; + uint32_t k3 = 0; + uint32_t k4 = 0; + + switch(len & 15) + { + case 15: k4 ^= tail[14] << 16; + case 14: k4 ^= tail[13] << 8; + case 13: k4 ^= tail[12] << 0; + k4 *= c4; k4 = ROTL32(k4,18); k4 *= c1; h4 ^= k4; + + case 12: k3 ^= tail[11] << 24; + case 11: k3 ^= tail[10] << 16; + case 10: k3 ^= tail[ 9] << 8; + case 9: k3 ^= tail[ 8] << 0; + k3 *= c3; k3 = ROTL32(k3,17); k3 *= c4; h3 ^= k3; + + case 8: k2 ^= tail[ 7] << 24; + case 7: k2 ^= tail[ 6] << 16; + case 6: k2 ^= tail[ 5] << 8; + case 5: k2 ^= tail[ 4] << 0; + k2 *= c2; k2 = ROTL32(k2,16); k2 *= c3; h2 ^= k2; + + case 4: k1 ^= tail[ 3] << 24; + case 3: k1 ^= tail[ 2] << 16; + case 2: k1 ^= tail[ 1] << 8; + case 1: k1 ^= tail[ 0] << 0; + k1 *= c1; k1 = ROTL32(k1,15); k1 *= c2; h1 ^= k1; + }; + + //---------- + // finalization + + h1 ^= len; h2 ^= len; h3 ^= len; h4 ^= len; + + h1 += h2; h1 += h3; h1 += h4; + h2 += h1; h3 += h1; h4 += h1; + + h1 = fmix32(h1); + h2 = fmix32(h2); + h3 = fmix32(h3); + h4 = fmix32(h4); + + h1 += h2; h1 += h3; h1 += h4; + h2 += h1; h3 += h1; h4 += h1; + + ((uint32_t*)out)[0] = h1; + ((uint32_t*)out)[1] = h2; + ((uint32_t*)out)[2] = h3; + ((uint32_t*)out)[3] = h4; +} + +//----------------------------------------------------------------------------- + +void MurmurHash3_x64_128 ( const void * key, const int len, + const uint32_t seed, void * out ) +{ + const uint8_t * data = (const uint8_t*)key; + const int nblocks = len / 16; + + uint64_t h1 = seed; + uint64_t h2 = seed; + + const uint64_t c1 = BIG_CONSTANT(0x87c37b91114253d5); + const uint64_t c2 = BIG_CONSTANT(0x4cf5ad432745937f); + + //---------- + // body + + const uint64_t * blocks = (const uint64_t *)(data); + + for(int i = 0; i < nblocks; i++) + { + uint64_t k1 = getblock64(blocks,i*2+0); + uint64_t k2 = getblock64(blocks,i*2+1); + + k1 *= c1; k1 = ROTL64(k1,31); k1 *= c2; h1 ^= k1; + + h1 = ROTL64(h1,27); h1 += h2; h1 = h1*5+0x52dce729; + + k2 *= c2; k2 = ROTL64(k2,33); k2 *= c1; h2 ^= k2; + + h2 = ROTL64(h2,31); h2 += h1; h2 = h2*5+0x38495ab5; + } + + //---------- + // tail + + const uint8_t * tail = (const uint8_t*)(data + nblocks*16); + + uint64_t k1 = 0; + uint64_t k2 = 0; + + switch(len & 15) + { + case 15: k2 ^= ((uint64_t)tail[14]) << 48; + case 14: k2 ^= ((uint64_t)tail[13]) << 40; + case 13: k2 ^= ((uint64_t)tail[12]) << 32; + case 12: k2 ^= ((uint64_t)tail[11]) << 24; + case 11: k2 ^= ((uint64_t)tail[10]) << 16; + case 10: k2 ^= ((uint64_t)tail[ 9]) << 8; + case 9: k2 ^= ((uint64_t)tail[ 8]) << 0; + k2 *= c2; k2 = ROTL64(k2,33); k2 *= c1; h2 ^= k2; + + case 8: k1 ^= ((uint64_t)tail[ 7]) << 56; + case 7: k1 ^= ((uint64_t)tail[ 6]) << 48; + case 6: k1 ^= ((uint64_t)tail[ 5]) << 40; + case 5: k1 ^= ((uint64_t)tail[ 4]) << 32; + case 4: k1 ^= ((uint64_t)tail[ 3]) << 24; + case 3: k1 ^= ((uint64_t)tail[ 2]) << 16; + case 2: k1 ^= ((uint64_t)tail[ 1]) << 8; + case 1: k1 ^= ((uint64_t)tail[ 0]) << 0; + k1 *= c1; k1 = ROTL64(k1,31); k1 *= c2; h1 ^= k1; + }; + + //---------- + // finalization + + h1 ^= len; h2 ^= len; + + h1 += h2; + h2 += h1; + + h1 = fmix64(h1); + h2 = fmix64(h2); + + h1 += h2; + h2 += h1; + + ((uint64_t*)out)[0] = h1; + ((uint64_t*)out)[1] = h2; +} + +//----------------------------------------------------------------------------- + diff --git a/tests/hasher/MurmurHash3.cc b/tests/hasher/MurmurHash3.cc new file mode 100644 index 0000000..aa7982d --- /dev/null +++ b/tests/hasher/MurmurHash3.cc @@ -0,0 +1,335 @@ +//----------------------------------------------------------------------------- +// MurmurHash3 was written by Austin Appleby, and is placed in the public +// domain. The author hereby disclaims copyright to this source code. + +// Note - The x86 and x64 versions do _not_ produce the same results, as the +// algorithms are optimized for their respective platforms. You can still +// compile and run any of them on any platform, but your performance with the +// non-native version will be less than optimal. + +#include "MurmurHash3.h" + +//----------------------------------------------------------------------------- +// Platform-specific functions and macros + +// Microsoft Visual Studio + +#if defined(_MSC_VER) + +#define FORCE_INLINE __forceinline + +#include + +#define ROTL32(x,y) _rotl(x,y) +#define ROTL64(x,y) _rotl64(x,y) + +#define BIG_CONSTANT(x) (x) + +// Other compilers + +#else // defined(_MSC_VER) + +#define FORCE_INLINE inline __attribute__((always_inline)) + +inline uint32_t rotl32 ( uint32_t x, int8_t r ) +{ + return (x << r) | (x >> (32 - r)); +} + +inline uint64_t rotl64 ( uint64_t x, int8_t r ) +{ + return (x << r) | (x >> (64 - r)); +} + +#define ROTL32(x,y) rotl32(x,y) +#define ROTL64(x,y) rotl64(x,y) + +#define BIG_CONSTANT(x) (x##LLU) + +#endif // !defined(_MSC_VER) + +//----------------------------------------------------------------------------- +// Block read - if your platform needs to do endian-swapping or can only +// handle aligned reads, do the conversion here + +FORCE_INLINE uint32_t getblock32 ( const uint32_t * p, int i ) +{ + return p[i]; +} + +FORCE_INLINE uint64_t getblock64 ( const uint64_t * p, int i ) +{ + return p[i]; +} + +//----------------------------------------------------------------------------- +// Finalization mix - force all bits of a hash block to avalanche + +FORCE_INLINE uint32_t fmix32 ( uint32_t h ) +{ + h ^= h >> 16; + h *= 0x85ebca6b; + h ^= h >> 13; + h *= 0xc2b2ae35; + h ^= h >> 16; + + return h; +} + +//---------- + +FORCE_INLINE uint64_t fmix64 ( uint64_t k ) +{ + k ^= k >> 33; + k *= BIG_CONSTANT(0xff51afd7ed558ccd); + k ^= k >> 33; + k *= BIG_CONSTANT(0xc4ceb9fe1a85ec53); + k ^= k >> 33; + + return k; +} + +//----------------------------------------------------------------------------- + +void MurmurHash3_x86_32 ( const void * key, int len, + uint32_t seed, void * out ) +{ + const uint8_t * data = (const uint8_t*)key; + const int nblocks = len / 4; + + uint32_t h1 = seed; + + const uint32_t c1 = 0xcc9e2d51; + const uint32_t c2 = 0x1b873593; + + //---------- + // body + + const uint32_t * blocks = (const uint32_t *)(data + nblocks*4); + + for(int i = -nblocks; i; i++) + { + uint32_t k1 = getblock32(blocks,i); + + k1 *= c1; + k1 = ROTL32(k1,15); + k1 *= c2; + + h1 ^= k1; + h1 = ROTL32(h1,13); + h1 = h1*5+0xe6546b64; + } + + //---------- + // tail + + const uint8_t * tail = (const uint8_t*)(data + nblocks*4); + + uint32_t k1 = 0; + + switch(len & 3) + { + case 3: k1 ^= tail[2] << 16; + case 2: k1 ^= tail[1] << 8; + case 1: k1 ^= tail[0]; + k1 *= c1; k1 = ROTL32(k1,15); k1 *= c2; h1 ^= k1; + }; + + //---------- + // finalization + + h1 ^= len; + + h1 = fmix32(h1); + + *(uint32_t*)out = h1; +} + +//----------------------------------------------------------------------------- + +void MurmurHash3_x86_128 ( const void * key, const int len, + uint32_t seed, void * out ) +{ + const uint8_t * data = (const uint8_t*)key; + const int nblocks = len / 16; + + uint32_t h1 = seed; + uint32_t h2 = seed; + uint32_t h3 = seed; + uint32_t h4 = seed; + + const uint32_t c1 = 0x239b961b; + const uint32_t c2 = 0xab0e9789; + const uint32_t c3 = 0x38b34ae5; + const uint32_t c4 = 0xa1e38b93; + + //---------- + // body + + const uint32_t * blocks = (const uint32_t *)(data + nblocks*16); + + for(int i = -nblocks; i; i++) + { + uint32_t k1 = getblock32(blocks,i*4+0); + uint32_t k2 = getblock32(blocks,i*4+1); + uint32_t k3 = getblock32(blocks,i*4+2); + uint32_t k4 = getblock32(blocks,i*4+3); + + k1 *= c1; k1 = ROTL32(k1,15); k1 *= c2; h1 ^= k1; + + h1 = ROTL32(h1,19); h1 += h2; h1 = h1*5+0x561ccd1b; + + k2 *= c2; k2 = ROTL32(k2,16); k2 *= c3; h2 ^= k2; + + h2 = ROTL32(h2,17); h2 += h3; h2 = h2*5+0x0bcaa747; + + k3 *= c3; k3 = ROTL32(k3,17); k3 *= c4; h3 ^= k3; + + h3 = ROTL32(h3,15); h3 += h4; h3 = h3*5+0x96cd1c35; + + k4 *= c4; k4 = ROTL32(k4,18); k4 *= c1; h4 ^= k4; + + h4 = ROTL32(h4,13); h4 += h1; h4 = h4*5+0x32ac3b17; + } + + //---------- + // tail + + const uint8_t * tail = (const uint8_t*)(data + nblocks*16); + + uint32_t k1 = 0; + uint32_t k2 = 0; + uint32_t k3 = 0; + uint32_t k4 = 0; + + switch(len & 15) + { + case 15: k4 ^= tail[14] << 16; + case 14: k4 ^= tail[13] << 8; + case 13: k4 ^= tail[12] << 0; + k4 *= c4; k4 = ROTL32(k4,18); k4 *= c1; h4 ^= k4; + + case 12: k3 ^= tail[11] << 24; + case 11: k3 ^= tail[10] << 16; + case 10: k3 ^= tail[ 9] << 8; + case 9: k3 ^= tail[ 8] << 0; + k3 *= c3; k3 = ROTL32(k3,17); k3 *= c4; h3 ^= k3; + + case 8: k2 ^= tail[ 7] << 24; + case 7: k2 ^= tail[ 6] << 16; + case 6: k2 ^= tail[ 5] << 8; + case 5: k2 ^= tail[ 4] << 0; + k2 *= c2; k2 = ROTL32(k2,16); k2 *= c3; h2 ^= k2; + + case 4: k1 ^= tail[ 3] << 24; + case 3: k1 ^= tail[ 2] << 16; + case 2: k1 ^= tail[ 1] << 8; + case 1: k1 ^= tail[ 0] << 0; + k1 *= c1; k1 = ROTL32(k1,15); k1 *= c2; h1 ^= k1; + }; + + //---------- + // finalization + + h1 ^= len; h2 ^= len; h3 ^= len; h4 ^= len; + + h1 += h2; h1 += h3; h1 += h4; + h2 += h1; h3 += h1; h4 += h1; + + h1 = fmix32(h1); + h2 = fmix32(h2); + h3 = fmix32(h3); + h4 = fmix32(h4); + + h1 += h2; h1 += h3; h1 += h4; + h2 += h1; h3 += h1; h4 += h1; + + ((uint32_t*)out)[0] = h1; + ((uint32_t*)out)[1] = h2; + ((uint32_t*)out)[2] = h3; + ((uint32_t*)out)[3] = h4; +} + +//----------------------------------------------------------------------------- + +void MurmurHash3_x64_128 ( const void * key, const int len, + const uint32_t seed, void * out ) +{ + const uint8_t * data = (const uint8_t*)key; + const int nblocks = len / 16; + + uint64_t h1 = seed; + uint64_t h2 = seed; + + const uint64_t c1 = BIG_CONSTANT(0x87c37b91114253d5); + const uint64_t c2 = BIG_CONSTANT(0x4cf5ad432745937f); + + //---------- + // body + + const uint64_t * blocks = (const uint64_t *)(data); + + for(int i = 0; i < nblocks; i++) + { + uint64_t k1 = getblock64(blocks,i*2+0); + uint64_t k2 = getblock64(blocks,i*2+1); + + k1 *= c1; k1 = ROTL64(k1,31); k1 *= c2; h1 ^= k1; + + h1 = ROTL64(h1,27); h1 += h2; h1 = h1*5+0x52dce729; + + k2 *= c2; k2 = ROTL64(k2,33); k2 *= c1; h2 ^= k2; + + h2 = ROTL64(h2,31); h2 += h1; h2 = h2*5+0x38495ab5; + } + + //---------- + // tail + + const uint8_t * tail = (const uint8_t*)(data + nblocks*16); + + uint64_t k1 = 0; + uint64_t k2 = 0; + + switch(len & 15) + { + case 15: k2 ^= ((uint64_t)tail[14]) << 48; + case 14: k2 ^= ((uint64_t)tail[13]) << 40; + case 13: k2 ^= ((uint64_t)tail[12]) << 32; + case 12: k2 ^= ((uint64_t)tail[11]) << 24; + case 11: k2 ^= ((uint64_t)tail[10]) << 16; + case 10: k2 ^= ((uint64_t)tail[ 9]) << 8; + case 9: k2 ^= ((uint64_t)tail[ 8]) << 0; + k2 *= c2; k2 = ROTL64(k2,33); k2 *= c1; h2 ^= k2; + + case 8: k1 ^= ((uint64_t)tail[ 7]) << 56; + case 7: k1 ^= ((uint64_t)tail[ 6]) << 48; + case 6: k1 ^= ((uint64_t)tail[ 5]) << 40; + case 5: k1 ^= ((uint64_t)tail[ 4]) << 32; + case 4: k1 ^= ((uint64_t)tail[ 3]) << 24; + case 3: k1 ^= ((uint64_t)tail[ 2]) << 16; + case 2: k1 ^= ((uint64_t)tail[ 1]) << 8; + case 1: k1 ^= ((uint64_t)tail[ 0]) << 0; + k1 *= c1; k1 = ROTL64(k1,31); k1 *= c2; h1 ^= k1; + }; + + //---------- + // finalization + + h1 ^= len; h2 ^= len; + + h1 += h2; + h2 += h1; + + h1 = fmix64(h1); + h2 = fmix64(h2); + + h1 += h2; + h2 += h1; + + ((uint64_t*)out)[0] = h1; + ((uint64_t*)out)[1] = h2; +} + +//----------------------------------------------------------------------------- + diff --git a/tests/hasher/MurmurHash3.h b/tests/hasher/MurmurHash3.h new file mode 100644 index 0000000..e1c6d34 --- /dev/null +++ b/tests/hasher/MurmurHash3.h @@ -0,0 +1,37 @@ +//----------------------------------------------------------------------------- +// MurmurHash3 was written by Austin Appleby, and is placed in the public +// domain. The author hereby disclaims copyright to this source code. + +#ifndef _MURMURHASH3_H_ +#define _MURMURHASH3_H_ + +//----------------------------------------------------------------------------- +// Platform-specific functions and macros + +// Microsoft Visual Studio + +#if defined(_MSC_VER) && (_MSC_VER < 1600) + +typedef unsigned char uint8_t; +typedef unsigned int uint32_t; +typedef unsigned __int64 uint64_t; + +// Other compilers + +#else // defined(_MSC_VER) + +#include + +#endif // !defined(_MSC_VER) + +//----------------------------------------------------------------------------- + +void MurmurHash3_x86_32 ( const void * key, int len, uint32_t seed, void * out ); + +void MurmurHash3_x86_128 ( const void * key, int len, uint32_t seed, void * out ); + +void MurmurHash3_x64_128 ( const void * key, int len, uint32_t seed, void * out ); + +//----------------------------------------------------------------------------- + +#endif // _MURMURHASH3_H_ diff --git a/tests/hasher/test_hasher.c b/tests/hasher/test_hasher.c new file mode 100644 index 0000000..8c390d6 --- /dev/null +++ b/tests/hasher/test_hasher.c @@ -0,0 +1,41 @@ +// vim:fileencoding=utf-8:foldmethod=marker + +#include "test_hasher.h" +#include "MurmurHash3.h" +#include "wapp.h" + +WpTestFuncResult test_murmur3(void) { + b8 result = true; + + WpStr8 str = wpStr8Lit("Hello world"); + WpI32Array arr = wpArray(i32, 1, 2, 3, 4, 5); + u64 num = 287324; + + WpMur3HasherIO io = wpMur3HasherIO(873923); + + WpU8Stream str_bytes = wpStream(u8, str.buf, str.size); + WpU8Stream arr_bytes = wpStream(u8, (u8 *)arr, wpArrayCount(arr) * wpArrayItemSize(arr)); + WpU8Stream num_bytes = wpStream(u8, (u8 *)&num, sizeof(u64)); + + u64 hash[2] = {0}; + + // Sring hash + wpX64Mur3Hasher128(&str_bytes, (void *)&io); + MurmurHash3_x64_128(str_bytes.data, str_bytes.count * str_bytes.item_size, io.seed, (void *)hash); + + result = result && io.out_hash1 == hash[0] && io.out_hash2 == hash[1]; + + // Array hash + wpX64Mur3Hasher128(&arr_bytes, (void *)&io); + MurmurHash3_x64_128(arr_bytes.data, arr_bytes.count * arr_bytes.item_size, io.seed, (void *)hash); + + result = result && io.out_hash1 == hash[0] && io.out_hash2 == hash[1]; + + // Number hash + wpX64Mur3Hasher128(&num_bytes, (void *)&io); + MurmurHash3_x64_128(num_bytes.data, num_bytes.count * num_bytes.item_size, io.seed, (void *)hash); + + result = result && io.out_hash1 == hash[0] && io.out_hash2 == hash[1]; + + return wpTesterResult(result); +} diff --git a/tests/hasher/test_hasher.cc b/tests/hasher/test_hasher.cc new file mode 100644 index 0000000..8c390d6 --- /dev/null +++ b/tests/hasher/test_hasher.cc @@ -0,0 +1,41 @@ +// vim:fileencoding=utf-8:foldmethod=marker + +#include "test_hasher.h" +#include "MurmurHash3.h" +#include "wapp.h" + +WpTestFuncResult test_murmur3(void) { + b8 result = true; + + WpStr8 str = wpStr8Lit("Hello world"); + WpI32Array arr = wpArray(i32, 1, 2, 3, 4, 5); + u64 num = 287324; + + WpMur3HasherIO io = wpMur3HasherIO(873923); + + WpU8Stream str_bytes = wpStream(u8, str.buf, str.size); + WpU8Stream arr_bytes = wpStream(u8, (u8 *)arr, wpArrayCount(arr) * wpArrayItemSize(arr)); + WpU8Stream num_bytes = wpStream(u8, (u8 *)&num, sizeof(u64)); + + u64 hash[2] = {0}; + + // Sring hash + wpX64Mur3Hasher128(&str_bytes, (void *)&io); + MurmurHash3_x64_128(str_bytes.data, str_bytes.count * str_bytes.item_size, io.seed, (void *)hash); + + result = result && io.out_hash1 == hash[0] && io.out_hash2 == hash[1]; + + // Array hash + wpX64Mur3Hasher128(&arr_bytes, (void *)&io); + MurmurHash3_x64_128(arr_bytes.data, arr_bytes.count * arr_bytes.item_size, io.seed, (void *)hash); + + result = result && io.out_hash1 == hash[0] && io.out_hash2 == hash[1]; + + // Number hash + wpX64Mur3Hasher128(&num_bytes, (void *)&io); + MurmurHash3_x64_128(num_bytes.data, num_bytes.count * num_bytes.item_size, io.seed, (void *)hash); + + result = result && io.out_hash1 == hash[0] && io.out_hash2 == hash[1]; + + return wpTesterResult(result); +} diff --git a/tests/hasher/test_hasher.h b/tests/hasher/test_hasher.h new file mode 100644 index 0000000..209d4f8 --- /dev/null +++ b/tests/hasher/test_hasher.h @@ -0,0 +1,11 @@ +// vim:fileencoding=utf-8:foldmethod=marker + +#ifndef TEST_HASHER_H +#define TEST_HASHER_H + +#include "wapp.h" + +// Test Murmur3 hash implementation against the reference implementation +WpTestFuncResult test_murmur3(void); + +#endif // !TEST_HASHER_H diff --git a/tests/wapptest.c b/tests/wapptest.c index 38ad6ef..c376ddb 100644 --- a/tests/wapptest.c +++ b/tests/wapptest.c @@ -8,6 +8,7 @@ #include "test_stream.h" #include "test_cpath.h" #include "test_file.h" +#include "test_hasher.h" #include "test_shell_commander.h" #include "wapp.h" #include @@ -101,6 +102,7 @@ int main(void) { test_wapp_file_close, test_wapp_file_rename, test_wapp_file_remove, + test_murmur3, test_commander_cmd_success, test_commander_cmd_failure, test_commander_cmd_out_buf_success, diff --git a/tests/wapptest.cc b/tests/wapptest.cc index 38ad6ef..c376ddb 100644 --- a/tests/wapptest.cc +++ b/tests/wapptest.cc @@ -8,6 +8,7 @@ #include "test_stream.h" #include "test_cpath.h" #include "test_file.h" +#include "test_hasher.h" #include "test_shell_commander.h" #include "wapp.h" #include @@ -101,6 +102,7 @@ int main(void) { test_wapp_file_close, test_wapp_file_rename, test_wapp_file_remove, + test_murmur3, test_commander_cmd_success, test_commander_cmd_failure, test_commander_cmd_out_buf_success,