Unwind just the "pseudorandom probing" part of recent sets,tables changes (#13816)
* Unwind just the "pseudorandom probing" (whole hash-code-keyed variable stride double hashing) part of recent sets & tables changes (which has still been causing bugs over a month later (e.g., two days ago https://github.com/nim-lang/Nim/issues/13794) as well as still having several "figure this out" implementation question comments in them (see just diffs of this PR). This topic has been discussed in many places: https://github.com/nim-lang/Nim/issues/13393 https://github.com/nim-lang/Nim/pull/13418 https://github.com/nim-lang/Nim/pull/13440 https://github.com/nim-lang/Nim/issues/13794 Alternative/non-mandatory stronger integer hashes (or vice-versa opt-in identity hashes) are a better solution that is more general (no illusion of one hard-coded sequence solving all problems) while retaining the virtues of linear probing such as cache obliviousness and age-less tables under delete-heavy workloads (still untested after a month of this change). The only real solution for truly adversarial keys is a hash keyed off of data unobservable to attackers. That all fits better with a few families of user-pluggable/define-switchable hashes which can be provided in a separate PR more about `hashes.nim`. This PR carefully preserves the better (but still hard coded!) probing of the `intsets` and other recent fixes like `move` annotations, hash order invariant tests, `intsets.missingOrExcl` fixing, and the move of `rightSize` into `hashcommon.nim`. * Fix `data.len` -> `dataLen` problem.
This commit is contained in:
parent
7abeba6aeb
commit
b1aa3b1eea
8 changed files with 99 additions and 175 deletions
|
|
@ -14,20 +14,13 @@ include hashcommon
|
|||
template rawGetDeepImpl() {.dirty.} = # Search algo for unconditional add
|
||||
genHashImpl(key, hc)
|
||||
var h: Hash = hc and maxHash(t)
|
||||
var perturb = t.getPerturb(hc)
|
||||
while true:
|
||||
let hcode = t.data[h].hcode
|
||||
if hcode == deletedMarker or hcode == freeMarker:
|
||||
break
|
||||
else:
|
||||
h = nextTry(h, maxHash(t), perturb)
|
||||
while isFilled(t.data[h].hcode):
|
||||
h = nextTry(h, maxHash(t))
|
||||
result = h
|
||||
|
||||
template rawInsertImpl(t) {.dirty.} =
|
||||
template rawInsertImpl() {.dirty.} =
|
||||
data[h].key = key
|
||||
data[h].val = val
|
||||
if data[h].hcode == deletedMarker:
|
||||
t.countDeleted.dec
|
||||
data[h].hcode = hc
|
||||
|
||||
proc rawGetDeep[X, A](t: X, key: A, hc: var Hash): int {.inline.} =
|
||||
|
|
@ -35,7 +28,7 @@ proc rawGetDeep[X, A](t: X, key: A, hc: var Hash): int {.inline.} =
|
|||
|
||||
proc rawInsert[X, A, B](t: var X, data: var KeyValuePairSeq[A, B],
|
||||
key: A, val: B, hc: Hash, h: Hash) =
|
||||
rawInsertImpl(t)
|
||||
rawInsertImpl()
|
||||
|
||||
template checkIfInitialized() =
|
||||
when compiles(defaultInitialSize):
|
||||
|
|
@ -51,6 +44,7 @@ template addImpl(enlarge) {.dirty.} =
|
|||
inc(t.counter)
|
||||
|
||||
template maybeRehashPutImpl(enlarge) {.dirty.} =
|
||||
checkIfInitialized()
|
||||
if mustRehash(t):
|
||||
enlarge(t)
|
||||
index = rawGetKnownHC(t, key, hc)
|
||||
|
|
@ -88,11 +82,24 @@ template delImplIdx(t, i) =
|
|||
let msk = maxHash(t)
|
||||
if i >= 0:
|
||||
dec(t.counter)
|
||||
inc(t.countDeleted)
|
||||
t.data[i].hcode = deletedMarker
|
||||
t.data[i].key = default(type(t.data[i].key))
|
||||
t.data[i].val = default(type(t.data[i].val))
|
||||
# mustRehash + enlarge not needed because counter+countDeleted doesn't change
|
||||
block outer:
|
||||
while true: # KnuthV3 Algo6.4R adapted for i=i+1 instead of i=i-1
|
||||
var j = i # The correctness of this depends on (h+1) in nextTry,
|
||||
var r = j # though may be adaptable to other simple sequences.
|
||||
t.data[i].hcode = 0 # mark current EMPTY
|
||||
t.data[i].key = default(type(t.data[i].key))
|
||||
t.data[i].val = default(type(t.data[i].val))
|
||||
while true:
|
||||
i = (i + 1) and msk # increment mod table size
|
||||
if isEmpty(t.data[i].hcode): # end of collision cluster; So all done
|
||||
break outer
|
||||
r = t.data[i].hcode and msk # "home" location of key@i
|
||||
if not ((i >= r and r > j) or (r > j and j > i) or (j > i and i >= r)):
|
||||
break
|
||||
when defined(js):
|
||||
t.data[j] = t.data[i]
|
||||
else:
|
||||
t.data[j] = move(t.data[i]) # data[j] will be marked EMPTY next loop
|
||||
|
||||
template delImpl() {.dirty.} =
|
||||
var hc: Hash
|
||||
|
|
@ -108,7 +115,6 @@ template clearImpl() {.dirty.} =
|
|||
t.counter = 0
|
||||
|
||||
template ctAnd(a, b): bool =
|
||||
# pending https://github.com/nim-lang/Nim/issues/13502
|
||||
when a:
|
||||
when b: true
|
||||
else: false
|
||||
|
|
@ -126,7 +132,7 @@ template initImpl(result: typed, size: int) =
|
|||
result.last = -1
|
||||
|
||||
template insertImpl() = # for CountTable
|
||||
checkIfInitialized()
|
||||
if t.dataLen == 0: initImpl(t, defaultInitialSize)
|
||||
if mustRehash(t): enlarge(t)
|
||||
ctRawInsert(t, t.data, key, val)
|
||||
inc(t.counter)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue