Add more robust hashing based on a paper

The paper in question can be referenced [here](https://courses.cs.washington.edu/courses/cse521/15sp/refs/thorup1.pdf)

This will compute a hash that is almost just as performant as the
previous hashing, but handles the case where if lower numbered bits
are all 0 and the highest bit of the length of the container is less
than the lowest bit of the key, the hash will always be 0.
This commit is contained in:
Joey Yakimowich-Payne 2020-02-12 13:40:58 -07:00
commit 437461d809
2 changed files with 56 additions and 14 deletions

View file

@ -50,8 +50,16 @@ template rawGetKnownHCImpl() {.dirty.} =
proc rawGetKnownHC[X, A](t: X, key: A, hc: Hash): int {.inline.} = proc rawGetKnownHC[X, A](t: X, key: A, hc: Hash): int {.inline.} =
rawGetKnownHCImpl() rawGetKnownHCImpl()
template bits(n): uint32 =
var temp = n
var bits = 0.uint32
while temp != 0:
temp = temp shr 1
bits += 1
bits
template genHashImpl(key, hc: typed) = template genHashImpl(key, hc: typed) =
hc = hash(key) hc = hash(key, targetBits=bits(maxHash(t)))
if hc == 0: # This almost never taken branch should be very predictable. if hc == 0: # This almost never taken branch should be very predictable.
hc = 314159265 # Value doesn't matter; Any non-zero favorite is fine. hc = 314159265 # Value doesn't matter; Any non-zero favorite is fine.

View file

@ -113,31 +113,65 @@ proc hash*[T: proc](x: T): Hash {.inline.} =
result = hash(pointer(x)) result = hash(pointer(x))
const const
prime = uint(11) defaultSeedA = (17316035218449499591'u64).uint
defaultSeedB = (1734880652122947187'u64).uint
proc hash*(x: int): Hash {.inline.} = template computeIntegerHash(x, numBits, seedA, seedB): Hash =
cast[Hash]((seedA*cast[uint](x) + seedB) shr (sizeof(x)*8 - targetBits*8))
proc hash*(
x: int,
targetBits = sizeof(uint).uint32,
seedA = defaultSeedA,
seedB = defaultSeedB
): Hash {.inline.} =
## Efficient hashing of integers. ## Efficient hashing of integers.
result = cast[Hash](cast[uint](x) * prime) result = computeIntegerHash(x, targetBits, seedA, seedB)
proc hash*(x: int64): Hash {.inline.} = proc hash*(
x: int64,
targetBits = sizeof(uint).uint32,
seedA = defaultSeedA,
seedB = defaultSeedB
): Hash {.inline.} =
## Efficient hashing of `int64` integers. ## Efficient hashing of `int64` integers.
result = cast[Hash](cast[uint](x) * prime) result = computeIntegerHash(x, targetBits, seedA, seedB)
proc hash*(x: uint): Hash {.inline.} = proc hash*(
x: uint,
targetBits = sizeof(uint).uint32,
seedA = defaultSeedA,
seedB = defaultSeedB
): Hash {.inline.} =
## Efficient hashing of unsigned integers. ## Efficient hashing of unsigned integers.
result = cast[Hash](x * prime) result = computeIntegerHash(x, targetBits, seedA, seedB)
proc hash*(x: uint64): Hash {.inline.} = proc hash*(
x: uint64,
targetBits = sizeof(uint).uint32,
seedA = defaultSeedA,
seedB = defaultSeedB
): Hash {.inline.} =
## Efficient hashing of `uint64` integers. ## Efficient hashing of `uint64` integers.
result = cast[Hash](cast[uint](x) * prime) result = computeIntegerHash(x, targetBits, seedA, seedB)
proc hash*(x: char): Hash {.inline.} = proc hash*(
x: char,
targetBits = sizeof(uint).uint32,
seedA = defaultSeedA,
seedB = defaultSeedB
): Hash {.inline.} =
## Efficient hashing of characters. ## Efficient hashing of characters.
result = cast[Hash](cast[uint](ord(x)) * prime) result = computeIntegerHash(ord(x), targetBits, seedA, seedB)
proc hash*[T: Ordinal](x: T): Hash {.inline.} = proc hash*[T: Ordinal](
x: T,
targetBits = sizeof(uint).uint32,
seedA = defaultSeedA,
seedB = defaultSeedB
): Hash {.inline.} =
## Efficient hashing of other ordinal types (e.g. enums). ## Efficient hashing of other ordinal types (e.g. enums).
result = cast[Hash](cast[uint](ord(x)) * prime) result = computeIntegerHash(x, targetBits, seedA, seedB)
proc hash*(x: float): Hash {.inline.} = proc hash*(x: float): Hash {.inline.} =
## Efficient hashing of floats. ## Efficient hashing of floats.