featurebase/shardwidth/helper.go
Seebs dab8dff9c2 make "subdivide list by shardwidth" available for reuse
We keep wanting this, and it's shardwidth-dependent code, and we keep rewriting it.
It should be in the shardwidth package.
2021-08-18 13:45:20 -05:00

70 lines
2.5 KiB
Go

// Copyright 2021 Molecula Corp.
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
package shardwidth
import (
"math/bits"
)
// FindNextShard returns the index of the first item which is not in the
// same shard as i. The index it returns may be equal to the length of the
// haystack, indicatincg that the rest of the list is in the same shard.
func FindNextShard(i int, haystack []uint64) int {
// compute the last thing that's in the same shard as haystack[i].
if i >= len(haystack) {
return i
}
// current shard:
shard := (haystack[i] >> Exponent)
// last value in shard:
shardEnd := ((shard + 1) << Exponent) - 1
j := i
// We want to do a binary search of the haystack. For any length of
// haystack, its topmost bit gives us a reasonable halfway point; it may
// not actually be halfway, but the number of steps it'll take to search
// it will be the same as if it were. sort.Search has interface overhead
// and makes us sad.
for incr := 1 << (bits.Len64(uint64(len(haystack) - i))); incr > 0; incr >>= 1 {
if j+incr < len(haystack) {
if haystack[j+incr] <= shardEnd {
j += incr
}
}
}
// we've found the last item that is in the same shard as i, so...
return j + 1
}
// FindShards finds the shards in a given haystack
func FindShards(haystack []uint64) (shards []uint64, endIndexes []int) {
if len(haystack) == 0 {
return nil, nil
}
index := 0
// the steady state of this loop is that shards contains the current
// shard, but not its ending index; each time we find a new ending
// index, we record that index as the end for the current shard, and
// the new shard, until we reach the end and append len(haystack)
// as the last index.
shards = []uint64{haystack[index] >> Exponent}
index = FindNextShard(index, haystack)
for index < len(haystack) {
shards = append(shards, haystack[index]>>Exponent)
endIndexes = append(endIndexes, index)
index = FindNextShard(index, haystack)
}
endIndexes = append(endIndexes, index)
return shards, endIndexes
}