Add more robust hashing based on a paper
The paper in question can be referenced [here](https://courses.cs.washington.edu/courses/cse521/15sp/refs/thorup1.pdf) This will compute a hash that is almost just as performant as the previous hashing, but handles the case where if lower numbered bits are all 0 and the highest bit of the length of the container is less than the lowest bit of the key, the hash will always be 0.
This commit is contained in:
parent
1e303100f8
commit
437461d809
2 changed files with 56 additions and 14 deletions
|
|
@ -50,8 +50,16 @@ template rawGetKnownHCImpl() {.dirty.} =
|
||||||
proc rawGetKnownHC[X, A](t: X, key: A, hc: Hash): int {.inline.} =
|
proc rawGetKnownHC[X, A](t: X, key: A, hc: Hash): int {.inline.} =
|
||||||
rawGetKnownHCImpl()
|
rawGetKnownHCImpl()
|
||||||
|
|
||||||
|
template bits(n): uint32 =
|
||||||
|
var temp = n
|
||||||
|
var bits = 0.uint32
|
||||||
|
while temp != 0:
|
||||||
|
temp = temp shr 1
|
||||||
|
bits += 1
|
||||||
|
bits
|
||||||
|
|
||||||
template genHashImpl(key, hc: typed) =
|
template genHashImpl(key, hc: typed) =
|
||||||
hc = hash(key)
|
hc = hash(key, targetBits=bits(maxHash(t)))
|
||||||
if hc == 0: # This almost never taken branch should be very predictable.
|
if hc == 0: # This almost never taken branch should be very predictable.
|
||||||
hc = 314159265 # Value doesn't matter; Any non-zero favorite is fine.
|
hc = 314159265 # Value doesn't matter; Any non-zero favorite is fine.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -113,31 +113,65 @@ proc hash*[T: proc](x: T): Hash {.inline.} =
|
||||||
result = hash(pointer(x))
|
result = hash(pointer(x))
|
||||||
|
|
||||||
const
|
const
|
||||||
prime = uint(11)
|
defaultSeedA = (17316035218449499591'u64).uint
|
||||||
|
defaultSeedB = (1734880652122947187'u64).uint
|
||||||
|
|
||||||
proc hash*(x: int): Hash {.inline.} =
|
template computeIntegerHash(x, numBits, seedA, seedB): Hash =
|
||||||
|
cast[Hash]((seedA*cast[uint](x) + seedB) shr (sizeof(x)*8 - targetBits*8))
|
||||||
|
|
||||||
|
proc hash*(
|
||||||
|
x: int,
|
||||||
|
targetBits = sizeof(uint).uint32,
|
||||||
|
seedA = defaultSeedA,
|
||||||
|
seedB = defaultSeedB
|
||||||
|
): Hash {.inline.} =
|
||||||
## Efficient hashing of integers.
|
## Efficient hashing of integers.
|
||||||
result = cast[Hash](cast[uint](x) * prime)
|
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||||
|
|
||||||
proc hash*(x: int64): Hash {.inline.} =
|
proc hash*(
|
||||||
|
x: int64,
|
||||||
|
targetBits = sizeof(uint).uint32,
|
||||||
|
seedA = defaultSeedA,
|
||||||
|
seedB = defaultSeedB
|
||||||
|
): Hash {.inline.} =
|
||||||
## Efficient hashing of `int64` integers.
|
## Efficient hashing of `int64` integers.
|
||||||
result = cast[Hash](cast[uint](x) * prime)
|
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||||
|
|
||||||
proc hash*(x: uint): Hash {.inline.} =
|
proc hash*(
|
||||||
|
x: uint,
|
||||||
|
targetBits = sizeof(uint).uint32,
|
||||||
|
seedA = defaultSeedA,
|
||||||
|
seedB = defaultSeedB
|
||||||
|
): Hash {.inline.} =
|
||||||
## Efficient hashing of unsigned integers.
|
## Efficient hashing of unsigned integers.
|
||||||
result = cast[Hash](x * prime)
|
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||||
|
|
||||||
proc hash*(x: uint64): Hash {.inline.} =
|
proc hash*(
|
||||||
|
x: uint64,
|
||||||
|
targetBits = sizeof(uint).uint32,
|
||||||
|
seedA = defaultSeedA,
|
||||||
|
seedB = defaultSeedB
|
||||||
|
): Hash {.inline.} =
|
||||||
## Efficient hashing of `uint64` integers.
|
## Efficient hashing of `uint64` integers.
|
||||||
result = cast[Hash](cast[uint](x) * prime)
|
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||||
|
|
||||||
proc hash*(x: char): Hash {.inline.} =
|
proc hash*(
|
||||||
|
x: char,
|
||||||
|
targetBits = sizeof(uint).uint32,
|
||||||
|
seedA = defaultSeedA,
|
||||||
|
seedB = defaultSeedB
|
||||||
|
): Hash {.inline.} =
|
||||||
## Efficient hashing of characters.
|
## Efficient hashing of characters.
|
||||||
result = cast[Hash](cast[uint](ord(x)) * prime)
|
result = computeIntegerHash(ord(x), targetBits, seedA, seedB)
|
||||||
|
|
||||||
proc hash*[T: Ordinal](x: T): Hash {.inline.} =
|
proc hash*[T: Ordinal](
|
||||||
|
x: T,
|
||||||
|
targetBits = sizeof(uint).uint32,
|
||||||
|
seedA = defaultSeedA,
|
||||||
|
seedB = defaultSeedB
|
||||||
|
): Hash {.inline.} =
|
||||||
## Efficient hashing of other ordinal types (e.g. enums).
|
## Efficient hashing of other ordinal types (e.g. enums).
|
||||||
result = cast[Hash](cast[uint](ord(x)) * prime)
|
result = computeIntegerHash(x, targetBits, seedA, seedB)
|
||||||
|
|
||||||
proc hash*(x: float): Hash {.inline.} =
|
proc hash*(x: float): Hash {.inline.} =
|
||||||
## Efficient hashing of floats.
|
## Efficient hashing of floats.
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue