Unwind just the "pseudorandom probing" part of recent sets,tables changes (#13816)
* Unwind just the "pseudorandom probing" (whole hash-code-keyed variable stride double hashing) part of recent sets & tables changes (which has still been causing bugs over a month later (e.g., two days ago https://github.com/nim-lang/Nim/issues/13794) as well as still having several "figure this out" implementation question comments in them (see just diffs of this PR). This topic has been discussed in many places: https://github.com/nim-lang/Nim/issues/13393 https://github.com/nim-lang/Nim/pull/13418 https://github.com/nim-lang/Nim/pull/13440 https://github.com/nim-lang/Nim/issues/13794 Alternative/non-mandatory stronger integer hashes (or vice-versa opt-in identity hashes) are a better solution that is more general (no illusion of one hard-coded sequence solving all problems) while retaining the virtues of linear probing such as cache obliviousness and age-less tables under delete-heavy workloads (still untested after a month of this change). The only real solution for truly adversarial keys is a hash keyed off of data unobservable to attackers. That all fits better with a few families of user-pluggable/define-switchable hashes which can be provided in a separate PR more about `hashes.nim`. This PR carefully preserves the better (but still hard coded!) probing of the `intsets` and other recent fixes like `move` annotations, hash order invariant tests, `intsets.missingOrExcl` fixing, and the move of `rightSize` into `hashcommon.nim`. * Fix `data.len` -> `dataLen` problem.
This commit is contained in:
parent
7abeba6aeb
commit
b1aa3b1eea
8 changed files with 99 additions and 175 deletions
|
|
@ -68,7 +68,6 @@ type
|
|||
## before calling other procs on it.
|
||||
data: KeyValuePairSeq[A]
|
||||
counter: int
|
||||
countDeleted: int
|
||||
|
||||
type
|
||||
OrderedKeyValuePair[A] = tuple[
|
||||
|
|
@ -81,7 +80,6 @@ type
|
|||
## <#initOrderedSet,int>`_ before calling other procs on it.
|
||||
data: OrderedKeyValuePairSeq[A]
|
||||
counter, first, last: int
|
||||
countDeleted: int
|
||||
|
||||
const
|
||||
defaultInitialSize* = 64
|
||||
|
|
@ -249,7 +247,7 @@ iterator items*[A](s: HashSet[A]): A =
|
|||
## echo b
|
||||
## # --> {(a: 1, b: 3), (a: 0, b: 4)}
|
||||
for h in 0 .. high(s.data):
|
||||
if isFilledAndValid(s.data[h].hcode): yield s.data[h].key
|
||||
if isFilled(s.data[h].hcode): yield s.data[h].key
|
||||
|
||||
proc containsOrIncl*[A](s: var HashSet[A], key: A): bool =
|
||||
## Includes `key` in the set `s` and tells if `key` was already in `s`.
|
||||
|
|
@ -341,7 +339,7 @@ proc pop*[A](s: var HashSet[A]): A =
|
|||
doAssertRaises(KeyError, echo s.pop)
|
||||
|
||||
for h in 0 .. high(s.data):
|
||||
if isFilledAndValid(s.data[h].hcode):
|
||||
if isFilled(s.data[h].hcode):
|
||||
result = s.data[h].key
|
||||
excl(s, result)
|
||||
return result
|
||||
|
|
@ -574,8 +572,7 @@ proc map*[A, B](data: HashSet[A], op: proc (x: A): B {.closure.}): HashSet[B] =
|
|||
proc hash*[A](s: HashSet[A]): Hash =
|
||||
## Hashing of HashSet.
|
||||
for h in 0 .. high(s.data):
|
||||
if isFilledAndValid(s.data[h].hcode):
|
||||
result = result xor s.data[h].hcode
|
||||
result = result xor s.data[h].hcode
|
||||
result = !$result
|
||||
|
||||
proc `$`*[A](s: HashSet[A]): string =
|
||||
|
|
@ -593,6 +590,7 @@ proc `$`*[A](s: HashSet[A]): string =
|
|||
## # --> {no, esc'aping, is " provided}
|
||||
dollarImpl()
|
||||
|
||||
|
||||
proc initSet*[A](initialSize = defaultInitialSize): HashSet[A] {.deprecated:
|
||||
"Deprecated since v0.20, use 'initHashSet'".} = initHashSet[A](initialSize)
|
||||
|
||||
|
|
@ -624,7 +622,7 @@ template forAllOrderedPairs(yieldStmt: untyped) {.dirty.} =
|
|||
var idx = 0
|
||||
while h >= 0:
|
||||
var nxt = s.data[h].next
|
||||
if isFilledAndValid(s.data[h].hcode):
|
||||
if isFilled(s.data[h].hcode):
|
||||
yieldStmt
|
||||
inc(idx)
|
||||
h = nxt
|
||||
|
|
@ -858,7 +856,7 @@ proc `==`*[A](s, t: OrderedSet[A]): bool =
|
|||
while h >= 0 and g >= 0:
|
||||
var nxh = s.data[h].next
|
||||
var nxg = t.data[g].next
|
||||
if isFilledAndValid(s.data[h].hcode) and isFilledAndValid(t.data[g].hcode):
|
||||
if isFilled(s.data[h].hcode) and isFilled(t.data[g].hcode):
|
||||
if s.data[h].key == t.data[g].key:
|
||||
inc compared
|
||||
else:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue