Unwind just the "pseudorandom probing" part of recent sets,tables changes (#13816)
* Unwind just the "pseudorandom probing" (whole hash-code-keyed variable stride double hashing) part of recent sets & tables changes (which has still been causing bugs over a month later (e.g., two days ago https://github.com/nim-lang/Nim/issues/13794) as well as still having several "figure this out" implementation question comments in them (see just diffs of this PR). This topic has been discussed in many places: https://github.com/nim-lang/Nim/issues/13393 https://github.com/nim-lang/Nim/pull/13418 https://github.com/nim-lang/Nim/pull/13440 https://github.com/nim-lang/Nim/issues/13794 Alternative/non-mandatory stronger integer hashes (or vice-versa opt-in identity hashes) are a better solution that is more general (no illusion of one hard-coded sequence solving all problems) while retaining the virtues of linear probing such as cache obliviousness and age-less tables under delete-heavy workloads (still untested after a month of this change). The only real solution for truly adversarial keys is a hash keyed off of data unobservable to attackers. That all fits better with a few families of user-pluggable/define-switchable hashes which can be provided in a separate PR more about `hashes.nim`. This PR carefully preserves the better (but still hard coded!) probing of the `intsets` and other recent fixes like `move` annotations, hash order invariant tests, `intsets.missingOrExcl` fixing, and the move of `rightSize` into `hashcommon.nim`. * Fix `data.len` -> `dataLen` problem.
This commit is contained in:
parent
7abeba6aeb
commit
b1aa3b1eea
8 changed files with 99 additions and 175 deletions
|
|
@ -233,7 +233,6 @@ type
|
|||
## For creating an empty Table, use `initTable proc<#initTable,int>`_.
|
||||
data: KeyValuePairSeq[A, B]
|
||||
counter: int
|
||||
countDeleted: int
|
||||
TableRef*[A, B] = ref Table[A, B] ## Ref version of `Table<#Table>`_.
|
||||
##
|
||||
## For creating a new empty TableRef, use `newTable proc
|
||||
|
|
@ -266,16 +265,14 @@ template get(t, key): untyped =
|
|||
|
||||
proc enlarge[A, B](t: var Table[A, B]) =
|
||||
var n: KeyValuePairSeq[A, B]
|
||||
newSeq(n, t.counter.rightSize)
|
||||
newSeq(n, len(t.data) * growthFactor)
|
||||
swap(t.data, n)
|
||||
t.countDeleted = 0
|
||||
for i in countup(0, high(n)):
|
||||
let eh = n[i].hcode
|
||||
if isFilledAndValid(eh):
|
||||
if isFilled(eh):
|
||||
var j: Hash = eh and maxHash(t)
|
||||
var perturb = t.getPerturb(eh)
|
||||
while isFilled(t.data[j].hcode):
|
||||
j = nextTry(j, maxHash(t), perturb)
|
||||
j = nextTry(j, maxHash(t))
|
||||
when defined(js):
|
||||
rawInsert(t, t.data, n[i].key, n[i].val, eh, j)
|
||||
else:
|
||||
|
|
@ -662,7 +659,7 @@ iterator pairs*[A, B](t: Table[A, B]): (A, B) =
|
|||
## # value: [1, 5, 7, 9]
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield (t.data[h].key, t.data[h].val)
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -684,7 +681,7 @@ iterator mpairs*[A, B](t: var Table[A, B]): (A, var B) =
|
|||
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield (t.data[h].key, t.data[h].val)
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -705,7 +702,7 @@ iterator keys*[A, B](t: Table[A, B]): A =
|
|||
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield t.data[h].key
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -726,7 +723,7 @@ iterator values*[A, B](t: Table[A, B]): B =
|
|||
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield t.data[h].val
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -748,27 +745,10 @@ iterator mvalues*[A, B](t: var Table[A, B]): var B =
|
|||
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield t.data[h].val
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
template hasKeyOrPutCache(cache, h): bool =
|
||||
# using `IntSet` would be an option worth considering to avoid quadratic
|
||||
# behavior in case user misuses Table with lots of duplicate keys; but it
|
||||
# has overhead in the common case of small number of duplicates.
|
||||
# However: when lots of duplicates are used, all operations would be slow
|
||||
# anyway because the full `hash(key)` is identical for these, which makes
|
||||
# `nextTry` follow the exact same path for each key, resulting in large
|
||||
# collision clusters. Alternatives could involve modifying the hash/retrieval
|
||||
# based on duplicate key count.
|
||||
var ret = false
|
||||
for hi in cache:
|
||||
if hi == h:
|
||||
ret = true
|
||||
break
|
||||
if not ret: cache.add h
|
||||
ret
|
||||
|
||||
iterator allValues*[A, B](t: Table[A, B]; key: A): B =
|
||||
## Iterates over any value in the table ``t`` that belongs to the given ``key``.
|
||||
##
|
||||
|
|
@ -782,20 +762,13 @@ iterator allValues*[A, B](t: Table[A, B]; key: A): B =
|
|||
for i in 1..3: a.add('z', 10*i)
|
||||
doAssert toSeq(a.pairs).sorted == @[('a', 3), ('b', 5), ('z', 10), ('z', 20), ('z', 30)]
|
||||
doAssert sorted(toSeq(a.allValues('z'))) == @[10, 20, 30]
|
||||
|
||||
let hc = genHash(key)
|
||||
var h: Hash = hc and high(t.data)
|
||||
var h: Hash = genHash(key) and high(t.data)
|
||||
let L = len(t)
|
||||
var perturb = t.getPerturb(hc)
|
||||
|
||||
var num = 0
|
||||
var cache: seq[Hash]
|
||||
while isFilled(t.data[h].hcode): # `isFilledAndValid` would be incorrect, see test for `allValues`
|
||||
if t.data[h].hcode == hc and t.data[h].key == key:
|
||||
if not hasKeyOrPutCache(cache, h):
|
||||
yield t.data[h].val
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
h = nextTry(h, high(t.data), perturb)
|
||||
while isFilled(t.data[h].hcode):
|
||||
if t.data[h].key == key:
|
||||
yield t.data[h].val
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
h = nextTry(h, high(t.data))
|
||||
|
||||
|
||||
|
||||
|
|
@ -1118,7 +1091,7 @@ iterator pairs*[A, B](t: TableRef[A, B]): (A, B) =
|
|||
## # value: [1, 5, 7, 9]
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield (t.data[h].key, t.data[h].val)
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -1140,7 +1113,7 @@ iterator mpairs*[A, B](t: TableRef[A, B]): (A, var B) =
|
|||
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield (t.data[h].key, t.data[h].val)
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -1161,7 +1134,7 @@ iterator keys*[A, B](t: TableRef[A, B]): A =
|
|||
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield t.data[h].key
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -1182,7 +1155,7 @@ iterator values*[A, B](t: TableRef[A, B]): B =
|
|||
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield t.data[h].val
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -1203,7 +1176,7 @@ iterator mvalues*[A, B](t: TableRef[A, B]): var B =
|
|||
|
||||
let L = len(t)
|
||||
for h in 0 .. high(t.data):
|
||||
if isFilledAndValid(t.data[h].hcode):
|
||||
if isFilled(t.data[h].hcode):
|
||||
yield t.data[h].val
|
||||
assert(len(t) == L, "the length of the table changed while iterating over it")
|
||||
|
||||
|
|
@ -1229,8 +1202,6 @@ type
|
|||
## <#initOrderedTable,int>`_.
|
||||
data: OrderedKeyValuePairSeq[A, B]
|
||||
counter, first, last: int
|
||||
countDeleted: int
|
||||
|
||||
OrderedTableRef*[A, B] = ref OrderedTable[A, B] ## Ref version of
|
||||
## `OrderedTable<#OrderedTable>`_.
|
||||
##
|
||||
|
|
@ -1252,7 +1223,7 @@ proc rawGet[A, B](t: OrderedTable[A, B], key: A, hc: var Hash): int =
|
|||
proc rawInsert[A, B](t: var OrderedTable[A, B],
|
||||
data: var OrderedKeyValuePairSeq[A, B],
|
||||
key: A, val: B, hc: Hash, h: Hash) =
|
||||
rawInsertImpl(t)
|
||||
rawInsertImpl()
|
||||
data[h].next = -1
|
||||
if t.first < 0: t.first = h
|
||||
if t.last >= 0: data[t.last].next = h
|
||||
|
|
@ -1260,20 +1231,18 @@ proc rawInsert[A, B](t: var OrderedTable[A, B],
|
|||
|
||||
proc enlarge[A, B](t: var OrderedTable[A, B]) =
|
||||
var n: OrderedKeyValuePairSeq[A, B]
|
||||
newSeq(n, t.counter.rightSize)
|
||||
newSeq(n, len(t.data) * growthFactor)
|
||||
var h = t.first
|
||||
t.first = -1
|
||||
t.last = -1
|
||||
swap(t.data, n)
|
||||
t.countDeleted = 0
|
||||
while h >= 0:
|
||||
var nxt = n[h].next
|
||||
let eh = n[h].hcode
|
||||
if isFilledAndValid(eh):
|
||||
if isFilled(eh):
|
||||
var j: Hash = eh and maxHash(t)
|
||||
var perturb = t.getPerturb(eh)
|
||||
while isFilled(t.data[j].hcode):
|
||||
j = nextTry(j, maxHash(t), perturb)
|
||||
j = nextTry(j, maxHash(t))
|
||||
rawInsert(t, t.data, move n[h].key, move n[h].val, n[h].hcode, j)
|
||||
h = nxt
|
||||
|
||||
|
|
@ -1282,10 +1251,6 @@ template forAllOrderedPairs(yieldStmt: untyped) {.dirty.} =
|
|||
var h = t.first
|
||||
while h >= 0:
|
||||
var nxt = t.data[h].next
|
||||
# For OrderedTable/OrderedTableRef, isFilled is ok because `del` is O(n)
|
||||
# and doesn't create tombsones, but if it does start using tombstones,
|
||||
# carefully replace `isFilled` by `isFilledAndValid` as appropriate for these
|
||||
# table types only, ditto with `OrderedSet`.
|
||||
if isFilled(t.data[h].hcode):
|
||||
yieldStmt
|
||||
h = nxt
|
||||
|
|
@ -2229,7 +2194,6 @@ type
|
|||
## <#initCountTable,int>`_.
|
||||
data: seq[tuple[key: A, val: int]]
|
||||
counter: int
|
||||
countDeleted: int
|
||||
isSorted: bool
|
||||
CountTableRef*[A] = ref CountTable[A] ## Ref version of
|
||||
## `CountTable<#CountTable>`_.
|
||||
|
|
@ -2242,10 +2206,8 @@ type
|
|||
|
||||
proc ctRawInsert[A](t: CountTable[A], data: var seq[tuple[key: A, val: int]],
|
||||
key: A, val: int) =
|
||||
let hc = hash(key)
|
||||
var perturb = t.getPerturb(hc)
|
||||
var h: Hash = hc and high(data)
|
||||
while data[h].val != 0: h = nextTry(h, high(data), perturb) # TODO: handle deletedMarker
|
||||
var h: Hash = hash(key) and high(data)
|
||||
while data[h].val != 0: h = nextTry(h, high(data))
|
||||
data[h].key = key
|
||||
data[h].val = val
|
||||
|
||||
|
|
@ -2272,12 +2234,10 @@ proc remove[A](t: var CountTable[A], key: A) =
|
|||
proc rawGet[A](t: CountTable[A], key: A): int =
|
||||
if t.data.len == 0:
|
||||
return -1
|
||||
let hc = hash(key)
|
||||
var perturb = t.getPerturb(hc)
|
||||
var h: Hash = hc and high(t.data) # start with real hash value
|
||||
while t.data[h].val != 0: # TODO: may need to handle t.data[h].hcode == deletedMarker?
|
||||
var h: Hash = hash(key) and high(t.data) # start with real hash value
|
||||
while t.data[h].val != 0:
|
||||
if t.data[h].key == key: return h
|
||||
h = nextTry(h, high(t.data), perturb)
|
||||
h = nextTry(h, high(t.data))
|
||||
result = -1 - h # < 0 => MISSING; insert idx = -1 - result
|
||||
|
||||
template ctget(t, key, default: untyped): untyped =
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue