Add more robust hashing based on a paper
The paper in question can be referenced [here](https://courses.cs.washington.edu/courses/cse521/15sp/refs/thorup1.pdf) This will compute a hash that is almost just as performant as the previous hashing, but handles the case where if lower numbered bits are all 0 and the highest bit of the length of the container is less than the lowest bit of the key, the hash will always be 0.
This commit is contained in:
parent
1e303100f8
commit
437461d809
2 changed files with 56 additions and 14 deletions
|
|
@ -50,8 +50,16 @@ template rawGetKnownHCImpl() {.dirty.} =
|
|||
proc rawGetKnownHC[X, A](t: X, key: A, hc: Hash): int {.inline.} =
|
||||
rawGetKnownHCImpl()
|
||||
|
||||
template bits(n): uint32 =
|
||||
var temp = n
|
||||
var bits = 0.uint32
|
||||
while temp != 0:
|
||||
temp = temp shr 1
|
||||
bits += 1
|
||||
bits
|
||||
|
||||
template genHashImpl(key, hc: typed) =
|
||||
hc = hash(key)
|
||||
hc = hash(key, targetBits=bits(maxHash(t)))
|
||||
if hc == 0: # This almost never taken branch should be very predictable.
|
||||
hc = 314159265 # Value doesn't matter; Any non-zero favorite is fine.
|
||||
|
||||
|
|
|
|||
|
|
@ -113,31 +113,65 @@ proc hash*[T: proc](x: T): Hash {.inline.} =
|
|||
result = hash(pointer(x))
|
||||
|
||||
const
|
||||
prime = uint(11)
|
||||
defaultSeedA = (17316035218449499591'u64).uint
|
||||
defaultSeedB = (1734880652122947187'u64).uint
|
||||
|
||||
proc hash*(x: int): Hash {.inline.} =
|
||||
template computeIntegerHash(x, numBits, seedA, seedB): Hash =
|
||||
cast[Hash]((seedA*cast[uint](x) + seedB) shr (sizeof(x)*8 - targetBits*8))
|
||||
|
||||
proc hash*(
|
||||
x: int,
|
||||
targetBits = sizeof(uint).uint32,
|
||||
seedA = defaultSeedA,
|
||||
seedB = defaultSeedB
|
||||
): Hash {.inline.} =
|
||||
## Efficient hashing of integers.
|
||||
result = cast[Hash](cast[uint](x) * prime)
|
||||
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||
|
||||
proc hash*(x: int64): Hash {.inline.} =
|
||||
proc hash*(
|
||||
x: int64,
|
||||
targetBits = sizeof(uint).uint32,
|
||||
seedA = defaultSeedA,
|
||||
seedB = defaultSeedB
|
||||
): Hash {.inline.} =
|
||||
## Efficient hashing of `int64` integers.
|
||||
result = cast[Hash](cast[uint](x) * prime)
|
||||
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||
|
||||
proc hash*(x: uint): Hash {.inline.} =
|
||||
proc hash*(
|
||||
x: uint,
|
||||
targetBits = sizeof(uint).uint32,
|
||||
seedA = defaultSeedA,
|
||||
seedB = defaultSeedB
|
||||
): Hash {.inline.} =
|
||||
## Efficient hashing of unsigned integers.
|
||||
result = cast[Hash](x * prime)
|
||||
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||
|
||||
proc hash*(x: uint64): Hash {.inline.} =
|
||||
proc hash*(
|
||||
x: uint64,
|
||||
targetBits = sizeof(uint).uint32,
|
||||
seedA = defaultSeedA,
|
||||
seedB = defaultSeedB
|
||||
): Hash {.inline.} =
|
||||
## Efficient hashing of `uint64` integers.
|
||||
result = cast[Hash](cast[uint](x) * prime)
|
||||
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||
|
||||
proc hash*(x: char): Hash {.inline.} =
|
||||
proc hash*(
|
||||
x: char,
|
||||
targetBits = sizeof(uint).uint32,
|
||||
seedA = defaultSeedA,
|
||||
seedB = defaultSeedB
|
||||
): Hash {.inline.} =
|
||||
## Efficient hashing of characters.
|
||||
result = cast[Hash](cast[uint](ord(x)) * prime)
|
||||
result = computeIntegerHash(ord(x), targetBits, seedA, seedB)
|
||||
|
||||
proc hash*[T: Ordinal](x: T): Hash {.inline.} =
|
||||
proc hash*[T: Ordinal](
|
||||
x: T,
|
||||
targetBits = sizeof(uint).uint32,
|
||||
seedA = defaultSeedA,
|
||||
seedB = defaultSeedB
|
||||
): Hash {.inline.} =
|
||||
## Efficient hashing of other ordinal types (e.g. enums).
|
||||
result = cast[Hash](cast[uint](ord(x)) * prime)
|
||||
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||
|
||||
proc hash*(x: float): Hash {.inline.} =
|
||||
## Efficient hashing of floats.
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue