From 437461d8090cfdef523a9003e00f1ad83a84ee5f Mon Sep 17 00:00:00 2001 From: Joey Yakimowich-Payne Date: Wed, 12 Feb 2020 13:40:58 -0700 Subject: [PATCH 1/3] Add more robust hashing based on a paper The paper in question can be referenced [here](https://courses.cs.washington.edu/courses/cse521/15sp/refs/thorup1.pdf) This will compute a hash that is almost just as performant as the previous hashing, but handles the case where if lower numbered bits are all 0 and the highest bit of the length of the container is less than the lowest bit of the key, the hash will always be 0. --- lib/pure/collections/hashcommon.nim | 10 ++++- lib/pure/hashes.nim | 60 ++++++++++++++++++++++------- 2 files changed, 56 insertions(+), 14 deletions(-) diff --git a/lib/pure/collections/hashcommon.nim b/lib/pure/collections/hashcommon.nim index d9a914afc..b97b67cb2 100644 --- a/lib/pure/collections/hashcommon.nim +++ b/lib/pure/collections/hashcommon.nim @@ -50,8 +50,16 @@ template rawGetKnownHCImpl() {.dirty.} = proc rawGetKnownHC[X, A](t: X, key: A, hc: Hash): int {.inline.} = rawGetKnownHCImpl() +template bits(n): uint32 = + var temp = n + var bits = 0.uint32 + while temp != 0: + temp = temp shr 1 + bits += 1 + bits + template genHashImpl(key, hc: typed) = - hc = hash(key) + hc = hash(key, targetBits=bits(maxHash(t))) if hc == 0: # This almost never taken branch should be very predictable. hc = 314159265 # Value doesn't matter; Any non-zero favorite is fine. diff --git a/lib/pure/hashes.nim b/lib/pure/hashes.nim index ac8498517..111e25382 100644 --- a/lib/pure/hashes.nim +++ b/lib/pure/hashes.nim @@ -113,31 +113,65 @@ proc hash*[T: proc](x: T): Hash {.inline.} = result = hash(pointer(x)) const - prime = uint(11) + defaultSeedA = (17316035218449499591'u64).uint + defaultSeedB = (1734880652122947187'u64).uint -proc hash*(x: int): Hash {.inline.} = +template computeIntegerHash(x, numBits, seedA, seedB): Hash = + cast[Hash]((seedA*cast[uint](x) + seedB) shr (sizeof(x)*8 - targetBits*8)) + +proc hash*( + x: int, + targetBits = sizeof(uint).uint32, + seedA = defaultSeedA, + seedB = defaultSeedB +): Hash {.inline.} = ## Efficient hashing of integers. - result = cast[Hash](cast[uint](x) * prime) + result = computeIntegerHash(x, targetBits, seedA, seedB) -proc hash*(x: int64): Hash {.inline.} = +proc hash*( + x: int64, + targetBits = sizeof(uint).uint32, + seedA = defaultSeedA, + seedB = defaultSeedB +): Hash {.inline.} = ## Efficient hashing of `int64` integers. - result = cast[Hash](cast[uint](x) * prime) + result = computeIntegerHash(x, targetBits, seedA, seedB) -proc hash*(x: uint): Hash {.inline.} = +proc hash*( + x: uint, + targetBits = sizeof(uint).uint32, + seedA = defaultSeedA, + seedB = defaultSeedB +): Hash {.inline.} = ## Efficient hashing of unsigned integers. - result = cast[Hash](x * prime) + result = computeIntegerHash(x, targetBits, seedA, seedB) -proc hash*(x: uint64): Hash {.inline.} = +proc hash*( + x: uint64, + targetBits = sizeof(uint).uint32, + seedA = defaultSeedA, + seedB = defaultSeedB +): Hash {.inline.} = ## Efficient hashing of `uint64` integers. - result = cast[Hash](cast[uint](x) * prime) + result = computeIntegerHash(x, targetBits, seedA, seedB) -proc hash*(x: char): Hash {.inline.} = +proc hash*( + x: char, + targetBits = sizeof(uint).uint32, + seedA = defaultSeedA, + seedB = defaultSeedB +): Hash {.inline.} = ## Efficient hashing of characters. - result = cast[Hash](cast[uint](ord(x)) * prime) + result = computeIntegerHash(ord(x), targetBits, seedA, seedB) -proc hash*[T: Ordinal](x: T): Hash {.inline.} = +proc hash*[T: Ordinal]( + x: T, + targetBits = sizeof(uint).uint32, + seedA = defaultSeedA, + seedB = defaultSeedB +): Hash {.inline.} = ## Efficient hashing of other ordinal types (e.g. enums). - result = cast[Hash](cast[uint](ord(x)) * prime) + result = computeIntegerHash(x, targetBits, seedA, seedB) proc hash*(x: float): Hash {.inline.} = ## Efficient hashing of floats. From b3dad3007c5c534b79edcfbe2438bed389f50a03 Mon Sep 17 00:00:00 2001 From: Joey Yakimowich-Payne Date: Thu, 13 Feb 2020 06:28:16 -0700 Subject: [PATCH 2/3] WIP 3 --- lib/pure/bitops.nim | 9 +++++++++ lib/pure/collections/hashcommon.nim | 11 ++--------- 2 files changed, 11 insertions(+), 9 deletions(-) diff --git a/lib/pure/bitops.nim b/lib/pure/bitops.nim index 9ebdabb7b..e00ea3109 100644 --- a/lib/pure/bitops.nim +++ b/lib/pure/bitops.nim @@ -355,6 +355,15 @@ proc firstSetBit*(x: SomeInteger): int {.inline, noSideEffect.} = when sizeof(x) <= 4: result = firstSetBitNim(x.uint32) else: result = firstSetBitNim(x.uint64) +proc lastSetBit*(x: SomeUnsignedInt): uint {.inline, noSideEffect.} = + ## Returns the 1-based index of the most significant set bit of x. + ## If x == 0, the result will also be 0 + var temp = x + result = 0 + while temp != 0: + temp = temp shr 1 + result += 1 + proc fastLog2*(x: SomeInteger): int {.inline, noSideEffect.} = ## Quickly find the log base 2 of an integer. ## If `x` is zero, when ``noUndefinedBitOpts`` is set, result is -1, diff --git a/lib/pure/collections/hashcommon.nim b/lib/pure/collections/hashcommon.nim index b97b67cb2..ce5ab5809 100644 --- a/lib/pure/collections/hashcommon.nim +++ b/lib/pure/collections/hashcommon.nim @@ -9,6 +9,7 @@ # An ``include`` file which contains common code for # hash sets and tables. +from bitops import lastSetBit const growthFactor = 2 @@ -50,16 +51,8 @@ template rawGetKnownHCImpl() {.dirty.} = proc rawGetKnownHC[X, A](t: X, key: A, hc: Hash): int {.inline.} = rawGetKnownHCImpl() -template bits(n): uint32 = - var temp = n - var bits = 0.uint32 - while temp != 0: - temp = temp shr 1 - bits += 1 - bits - template genHashImpl(key, hc: typed) = - hc = hash(key, targetBits=bits(maxHash(t))) + hc = hash(key, targetBits=lastSetBit(maxHash(t))) if hc == 0: # This almost never taken branch should be very predictable. hc = 314159265 # Value doesn't matter; Any non-zero favorite is fine. From ae108f7a567a87188b144d3fe4f99ddf808750de Mon Sep 17 00:00:00 2001 From: Joey Yakimowich-Payne Date: Thu, 13 Feb 2020 08:02:20 -0700 Subject: [PATCH 3/3] WIP 4 --- lib/pure/collections/hashcommon.nim | 2 +- lib/pure/hashes.nim | 14 +++++++------- tests/collections/thashsets.nim | 12 +++++++++++- 3 files changed, 19 insertions(+), 9 deletions(-) diff --git a/lib/pure/collections/hashcommon.nim b/lib/pure/collections/hashcommon.nim index ce5ab5809..6281074af 100644 --- a/lib/pure/collections/hashcommon.nim +++ b/lib/pure/collections/hashcommon.nim @@ -52,7 +52,7 @@ proc rawGetKnownHC[X, A](t: X, key: A, hc: Hash): int {.inline.} = rawGetKnownHCImpl() template genHashImpl(key, hc: typed) = - hc = hash(key, targetBits=lastSetBit(maxHash(t))) + hc = hash(key, targetBits=lastSetBit(maxHash(t).uint)) if hc == 0: # This almost never taken branch should be very predictable. hc = 314159265 # Value doesn't matter; Any non-zero favorite is fine. diff --git a/lib/pure/hashes.nim b/lib/pure/hashes.nim index 111e25382..9a511ce9d 100644 --- a/lib/pure/hashes.nim +++ b/lib/pure/hashes.nim @@ -117,11 +117,11 @@ const defaultSeedB = (1734880652122947187'u64).uint template computeIntegerHash(x, numBits, seedA, seedB): Hash = - cast[Hash]((seedA*cast[uint](x) + seedB) shr (sizeof(x)*8 - targetBits*8)) + cast[Hash]((seedA*cast[uint](x) + seedB) shr (sizeof(x)*8 - numBits*8)) proc hash*( x: int, - targetBits = sizeof(uint).uint32, + targetBits: SomeUnsignedInt = sizeof(uint), seedA = defaultSeedA, seedB = defaultSeedB ): Hash {.inline.} = @@ -130,7 +130,7 @@ proc hash*( proc hash*( x: int64, - targetBits = sizeof(uint).uint32, + targetBits: SomeUnsignedInt = sizeof(uint), seedA = defaultSeedA, seedB = defaultSeedB ): Hash {.inline.} = @@ -139,7 +139,7 @@ proc hash*( proc hash*( x: uint, - targetBits = sizeof(uint).uint32, + targetBits: SomeUnsignedInt = sizeof(uint), seedA = defaultSeedA, seedB = defaultSeedB ): Hash {.inline.} = @@ -148,7 +148,7 @@ proc hash*( proc hash*( x: uint64, - targetBits = sizeof(uint).uint32, + targetBits: SomeUnsignedInt = sizeof(uint), seedA = defaultSeedA, seedB = defaultSeedB ): Hash {.inline.} = @@ -157,7 +157,7 @@ proc hash*( proc hash*( x: char, - targetBits = sizeof(uint).uint32, + targetBits: SomeUnsignedInt = sizeof(uint), seedA = defaultSeedA, seedB = defaultSeedB ): Hash {.inline.} = @@ -166,7 +166,7 @@ proc hash*( proc hash*[T: Ordinal]( x: T, - targetBits = sizeof(uint).uint32, + targetBits: SomeUnsignedInt = sizeof(uint), seedA = defaultSeedA, seedB = defaultSeedB ): Hash {.inline.} = diff --git a/tests/collections/thashsets.nim b/tests/collections/thashsets.nim index cd4401511..bccf4654f 100644 --- a/tests/collections/thashsets.nim +++ b/tests/collections/thashsets.nim @@ -1,4 +1,4 @@ -import sets, hashes, algorithm +import sets, hashes, algorithm, bitops block setEquality: @@ -120,3 +120,13 @@ block hashForOrderdSet: reversed = !$reversed doAssert hash(r) == reversed doAssert hash(s1) != reversed + +block hashDoesntHaveLowerBitsCleared: + let num32 = 0xFFFF.uint32 + let bits = num32.lastBitSet + + let h1 = hash(0xFFFFFFFF00000000'u64, targetBits=bits) + let h2 = hash(0x2345123400000000'u64, targetBits=bits) + + + doAssert (h1 and num32) != (h2 and num32)