mirror of
https://github.com/featurebasedb/featurebase.git
synced 2026-09-05 08:10:50 +00:00
- all tests green on RoaringTx
- RoaringTx on by default
- blueGreenTx testing framework available for A-vs-B comparison
of Tx implementations
- flag -tx added to server command line but not wired to
change NewIndex() selection yet.
- 918 green tests, 14 tests red on BadgerTx.
A full list of the 14 red tests on BadgerTx follows.
Note that these red tests represent not defects in BadgerDB
or BadgerTx but rather failures of the pre-existing pilosa infrastructure to yet
be fully adapted from files to using a transactional storage engine.
As such these are tests that RBF should not be expected to
pass yet either.
Fixing the pilosa infrastructure to allow these tests
to go green under Badger is the next and highest priority
order of business, but RBF can get much testing benefit
from the 918 green tests we do have, and hence we merge
as much as we have today.
The 14 red tests when NewIndex() is set to use
BadgerTx are as follows. Note in particular
that pilosa cluster resizing is not working yet under a
transactional store.
TestCluster_ResizeStates/Multiple_nodes,_with_data
TestImportClearRestart/0MaxOpN10000
TestImportClearRestart/1MaxOpN10000
TestImportClearRestart/2MaxOpN10000
TestImportClearRestart/3MaxOpN10000
TestExecutor_Execute_Existence/Row
TestExecutor_ForeignIndex
TestExecutor_Execute_CountDistinct/Distinct
TestExecutor_Execute_CountDistinct/Count(Distinct)
TestExecutor_Execute_CountDistinct/GroupBy(Distinct)
TestExecutor_BareDistinct
TestExecutor_Execute_TopNDistinct/TopN
TestHolderSyncer_IntField/BasicSync
TestHolderSyncer_IntField/MultiShard
1701 lines
53 KiB
Go
1701 lines
53 KiB
Go
// Copyright 2020 Pilosa Corp.
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
|
|
package pilosa
|
|
|
|
import (
|
|
"errors"
|
|
"fmt"
|
|
"io/ioutil"
|
|
"log"
|
|
"os"
|
|
"runtime"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
"unsafe"
|
|
|
|
badger "github.com/dgraph-io/badger/v2"
|
|
"github.com/pilosa/pilosa/v2/roaring"
|
|
)
|
|
|
|
// TODO: is there a more optimal time to do badger garbage collection?
|
|
// As in: do we need to be more aggressive about cleaning in
|
|
// proportion to write activity? Space monitoring available with the
|
|
// badger.DB.Size() (lsm, vlog int64) call.
|
|
//
|
|
// See: https://godoc.org/github.com/dgraph-io/badger#DB.RunValueLogGC
|
|
// and: https://github.com/dgraph-io/badger#garbage-collection
|
|
//
|
|
// For now we run GC periodically every 1 minute or as set by the
|
|
// BadgerDBWrapper.GcEveryDur duration.
|
|
//
|
|
// Background: (quoting from docs referenced above)
|
|
//
|
|
// "Badger relies on the client to perform garbage collection at a time of
|
|
// their choosing. It provides the following method, which can be invoked
|
|
// at an appropriate time:
|
|
//
|
|
// "DB.RunValueLogGC(): This method is designed to do garbage collection while
|
|
// Badger is online. Along with randomly picking a file, it uses statistics
|
|
// generated by the LSM-tree compactions to pick files that are likely to
|
|
// lead to maximum space reclamation. It is recommended to be called during
|
|
// periods of low activity in your system, or periodically. One call would
|
|
// only result in removal of at max one log file. As an optimization, you
|
|
// could also immediately re-run it whenever it returns nil error (indicating
|
|
// a successful value log GC), as shown below."
|
|
//
|
|
// ticker := time.NewTicker(5 * time.Minute)
|
|
// defer ticker.Stop()
|
|
// for range ticker.C {
|
|
// again:
|
|
// err := db.RunValueLogGC(0.5)
|
|
// if err == nil {
|
|
// goto again
|
|
// }
|
|
// }
|
|
//
|
|
|
|
// =========================================================
|
|
// A note on using a recent version of badgerdb:
|
|
//
|
|
// We require a v2 release of badger after 2020 May 13, when support for
|
|
// multiple read-write iterators within one transaction was added.
|
|
// Many executor_test.go tests do foreachRow() operations,
|
|
// which call BadgerTx.ContainerIterator(), which in turn creates
|
|
// a first read-write iterator, and then OffsetRange(), which needs a
|
|
// second iterator, while still in the same read-write transaction.
|
|
//
|
|
// The most recent v2 master was pulled in and added to go.mod
|
|
// by doing go get github.com/dgraph-io/badger/v2@master
|
|
// resulting in the go.mod line
|
|
// github.com/dgraph-io/badger/v2 v2.0.1-rc1.0.20200709123515-8e896a7af361
|
|
// as of this writing, 2020 July 09. This version contains the support
|
|
// for having multiple read-write iterators.
|
|
//
|
|
// Reference on github.com/dgraph-io/badger
|
|
//
|
|
// commit af22dfd8d51317d765f0c05dcdf1d15981cca4f3
|
|
// Author: Elliot Courant <me@elliotcourant.dev>
|
|
// Date: Wed May 13 01:07:33 2020 -0500
|
|
//
|
|
// Support multiple iterators in read-write transactions. (#1286)
|
|
//
|
|
// This adds support for multiple iterators during a read-write transaction. The
|
|
// iterators created in a read-write transaction will only be able to see writes
|
|
// that were performed before the iterator was created. Any writes that occur
|
|
// after the iterator is created will be invisible to the iterator.
|
|
//
|
|
// Fixes https://github.com/dgraph-io/badger/issues/981
|
|
//
|
|
//
|
|
// Otherwise we'll get these panics:
|
|
// 'Only one iterator can be active at one time, for a RW txn.'
|
|
// when trying to open a second iterator on the same write transaction.
|
|
// e.g. go test -v -run TestExecutor_TranslateRowsOnBool
|
|
|
|
var badgerDefaultLogger *BadgerLog
|
|
var badgerTestLogger *BadgerLog
|
|
|
|
func init() {
|
|
// badger test output clutters up the screen, dump to /dev/null for now.
|
|
// TODO(jea): figure out where badger logging should go.
|
|
null, err := os.Open(os.DevNull)
|
|
panicOn(err)
|
|
badgerTestLogger = &BadgerLog{Logger: log.New(null, "badger ", log.LstdFlags)}
|
|
badgerDefaultLogger = badgerTestLogger
|
|
|
|
// BadgerDB recommends a minimum of 128 GOMAXPROCS to make use of the IOPs
|
|
// available on the SSD. So we set that here. Details:
|
|
//
|
|
// from https://github.com/dgraph-io/badger#are-there-any-go-specific-settings-that-i-should-use
|
|
//
|
|
// "We *highly* recommend setting a high number for GOMAXPROCS,
|
|
// which allows Go to observe the full IOPS throughput provided by
|
|
// modern SSDs. In Dgraph, we have set it to 128. For more details,
|
|
// see this thread [https://groups.google.com/forum/#!topic/golang-nuts/jPb_h3TvlKE/discussion]."
|
|
//
|
|
// From that thread on golang-nuts:
|
|
//
|
|
// "Manish Rai Jain
|
|
// 8/7/17
|
|
// Hey folks,
|
|
// During Gophercon, I happened to meet Russ Cox and asked him the same question.
|
|
// If File::Read blocks goroutines, which then spawn new OS threads, in a long running job,
|
|
// there should be plenty of OS threads created already, so the random read throughput
|
|
// should increase over time and stabilize to the maximum possible value. But, that's
|
|
// not what I see in my benchmarks.
|
|
//
|
|
// And his explanation was that the GOMAXPROCS in a way acts like a multiplexer.
|
|
// From docs, "the GOMAXPROCS variable limits the number of operating system threads
|
|
// that can execute user-level Go code simultaneously." Which basically means, all
|
|
// reads must first be run only via GOMAXPROCS number of goroutines, before switching
|
|
// over to some OS thread (not really a switch, but conceptually speaking). This
|
|
// introduces a bottleneck for throughput.
|
|
// I re-ran my benchmarks with a much higher GOMAXPROCS and was able to then
|
|
// achieve the maximum throughput. The numbers are here:
|
|
// https://github.com/dgraph-io/badger-bench/blob/master/randread/maxprocs.txt
|
|
// To summarize these benchmarks, Linux fio achieves 118K IOPS, and with GOMAXPROCS=64/128,
|
|
// I'm able to achieve 105K IOPS, which is close enough. Win!
|
|
//
|
|
// Regarding the point about using io_submit etc., instead of goroutines; I managed to
|
|
// find a library which does that, but it performed worse than just using goroutines.
|
|
// https://github.com/traetox/goaio/issues/3
|
|
// From what I gather (talking to Russ and Ian), whatever work is going on in user space,
|
|
// the same work has to happen in kernel space; so there's not much benefit here.
|
|
//
|
|
// Overall, with GOMAXPROCS set to a higher value (as I've done in Dgraph), one can get
|
|
// the advertised SSD throughput using goroutines."
|
|
//
|
|
runtime.GOMAXPROCS(128)
|
|
}
|
|
|
|
// BadgerLog exists because badger requires a particular logger interface, with a
|
|
// Debugf method that is not on standard library log.Logger
|
|
type BadgerLog struct {
|
|
*log.Logger
|
|
}
|
|
|
|
// Errorf logs an error.
|
|
func (l *BadgerLog) Errorf(f string, v ...interface{}) {
|
|
l.Printf("ERROR: "+f, v...)
|
|
}
|
|
|
|
// Warningf logs a warning.
|
|
func (l *BadgerLog) Warningf(f string, v ...interface{}) {
|
|
l.Printf("WARNING: "+f, v...)
|
|
}
|
|
|
|
// Infof logs an informational statement.
|
|
func (l *BadgerLog) Infof(f string, v ...interface{}) {
|
|
l.Printf("INFO: "+f, v...)
|
|
}
|
|
|
|
// Debugf logs a debug statement.
|
|
func (l *BadgerLog) Debugf(f string, v ...interface{}) {
|
|
l.Printf("DEBUG: "+f, v...)
|
|
}
|
|
|
|
// newBadgerDBWrapper creates a new empty database, blowing away
|
|
// any prior path + "-badgerdb" directory.
|
|
func newBadgerDBWrapper(path string) (*BadgerDBWrapper, error) {
|
|
bpath := badgerPath(path)
|
|
err := os.RemoveAll(bpath)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
return openBadgerDBWrapper(bpath)
|
|
}
|
|
|
|
// badgerPath is a helper for determining the full directory
|
|
// in which the badger database will be stored.
|
|
func badgerPath(path string) string {
|
|
if !strings.HasSuffix(path, "-badgerdb") {
|
|
return path + "-badgerdb"
|
|
}
|
|
return path
|
|
}
|
|
|
|
// openBadgerDB opens the database in the bpath directoy
|
|
// without deleting any prior content. Any BadgerDB
|
|
// database directory will have the "-badgerdb" suffix.
|
|
func openBadgerDBWrapper(bpath string) (*BadgerDBWrapper, error) {
|
|
|
|
// now that newTxFactory can call us directly, we might not
|
|
// have the -badgerdb suffix.
|
|
if !strings.HasSuffix(bpath, "-badgerdb") {
|
|
bpath += "-badgerdb"
|
|
}
|
|
|
|
// regular: works on amd64, but 386 doesn't work.
|
|
opt := badger.DefaultOptions(bpath).WithLogger(badgerDefaultLogger)
|
|
|
|
// to get memory only do:
|
|
//opt := badger.DefaultOptions("").WithLogger(badgerDefaultLogger).WithInMemory(true)
|
|
|
|
db, err := badger.Open(opt)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
halt := make(chan bool)
|
|
w := &BadgerDBWrapper{
|
|
path: bpath,
|
|
db: db,
|
|
halt: halt,
|
|
hasher: NewBlake3Hasher(),
|
|
}
|
|
w.startBadgerGarbageCollectionBackgroundGoro()
|
|
return w, nil
|
|
}
|
|
|
|
// DeleteIndex deletes all the containers associated with
|
|
// the named index from the badger database.
|
|
func (w *BadgerDBWrapper) DeleteIndex(indexName string) error {
|
|
|
|
// We use the apostrophie rune `'` to locate the end of the
|
|
// index name in the key prefix, so we cannot allow indexNames
|
|
// themselves to contain apostrophies.
|
|
if strings.Contains(indexName, "'") {
|
|
return fmt.Errorf("error: bad indexName `%v` in BadgerDBWrapper.DeleteIndex() call: indexName cannot contain apostrophes/single quotes.", indexName)
|
|
}
|
|
w.muDb.Lock()
|
|
defer w.muDb.Unlock()
|
|
|
|
// a) do key-ony iteration, no value fetch;
|
|
//
|
|
// b) do deletes in large batches, to avoid alot of txn overhead;
|
|
// per recommendation https://github.com/dgraph-io/badger/issues/598
|
|
//
|
|
// c) we do not, at present, try to maintain one large
|
|
// transaction with all the keys in a index in it. Because
|
|
// there can be too many keys. Hence the index will disappear
|
|
// in chucks of 100K keys, not atomically-all-at-once.
|
|
|
|
prefix := badgerIndexOnlyPrefix(indexName)
|
|
|
|
noMoreKeysWithPrefix := false
|
|
const maxDeletesPerTxn = 100000
|
|
|
|
for !noMoreKeysWithPrefix {
|
|
err := w.db.Update(func(txn *badger.Txn) error {
|
|
o := badger.DefaultIteratorOptions
|
|
o.AllVersions = false
|
|
o.PrefetchValues = false // key-only iteration, no values.
|
|
|
|
// note: panic: Unclosed iterator at time of Txn.Discard ? panic on segfault here?
|
|
// This means we messed up and Closed() the Database already; too early.
|
|
it := txn.NewIterator(o)
|
|
|
|
defer it.Close()
|
|
n := 0
|
|
goners := make([][]byte, 0, maxDeletesPerTxn)
|
|
for it.Seek(prefix); it.ValidForPrefix(prefix); it.Next() {
|
|
|
|
// KeyCopy() is required; Key() means corruption and possible segfault.
|
|
key := it.Item().KeyCopy(nil)
|
|
goners = append(goners, key)
|
|
n++
|
|
if n >= maxDeletesPerTxn {
|
|
break
|
|
}
|
|
}
|
|
if !it.ValidForPrefix(prefix) {
|
|
noMoreKeysWithPrefix = true // done with the full delete of up to maxDeletesPerTxn
|
|
}
|
|
for _, key := range goners {
|
|
if err := txn.Delete(key); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
return nil // auto-commit happens
|
|
})
|
|
// err back from Update can be ErrConflict in case of
|
|
// a conflict. Badger docs: "Depending on the state
|
|
// of your application, you have the option to
|
|
// retry the operation if you receive this error."
|
|
panicOn(err)
|
|
|
|
} // end for: proceed to next bath of 100K keys
|
|
|
|
// Finally, run a garbage collection to delete values from the value log.
|
|
//
|
|
// "Only one GC is allowed at a time. If another value log GC
|
|
// is running, or DB has been closed, this would return an ErrRejected."
|
|
// -- https://godoc.org/github.com/dgraph-io/badger#DB.RunValueLogGC
|
|
// Still, we don't see a mutex inside the RunValueLogGC code, so
|
|
// lock muGC just to be sure.
|
|
w.muGC.Lock()
|
|
defer w.muGC.Unlock()
|
|
_ = w.db.RunValueLogGC(0.5)
|
|
|
|
return nil
|
|
}
|
|
|
|
// startBadgerGarbageCollectionBackgroundGoro handles Badger DB
|
|
// garbage colection by regularly purging the value log from
|
|
// a background goroutine. w.GcEveryDur controls how often
|
|
// it runs. The default is after every 60 seconds.
|
|
func (w *BadgerDBWrapper) startBadgerGarbageCollectionBackgroundGoro() {
|
|
go func() {
|
|
dur := w.GcEveryDur
|
|
if dur == 0 {
|
|
dur = time.Minute
|
|
}
|
|
ticker := time.NewTicker(dur)
|
|
defer ticker.Stop()
|
|
for {
|
|
select {
|
|
case <-ticker.C:
|
|
w.muGC.Lock()
|
|
again:
|
|
err := w.db.RunValueLogGC(0.5)
|
|
if err == nil {
|
|
goto again
|
|
}
|
|
w.muGC.Unlock()
|
|
case <-w.halt:
|
|
return
|
|
}
|
|
}
|
|
}()
|
|
}
|
|
|
|
// statically confirm that BadgerTx satisfies the Tx interface.
|
|
var _ Tx = (*BadgerTx)(nil)
|
|
|
|
// BadgerDBWrapper provides the NewBadgerTx() method.
|
|
// The methods on BadgerDBWrapper are thread-safe, and can be called
|
|
// from different goroutines/threads.
|
|
type BadgerDBWrapper struct {
|
|
// serialize operations on BadgerDBWrapper and thus on the .db too,
|
|
// when obtaining new txns on different goroutines.
|
|
muDb sync.Mutex
|
|
|
|
path string
|
|
db *badger.DB
|
|
|
|
// openTx and openIt are BadgerDBWrapper scoped tables of all open
|
|
// transactions and iterators. These are primarily for debugging purposes.
|
|
// openTx and openIt should only be read/written after locking the muOpenTxIt mutex.
|
|
|
|
// the bool value is the writable attribute of the key *BadgerTx
|
|
openTx map[*BadgerTx]bool
|
|
|
|
// the bool value is whether the iterator is reversed
|
|
openIt map[*BadgerIterator]bool
|
|
|
|
// protect openTx and openIt
|
|
muOpenTxIt sync.Mutex
|
|
|
|
// close(halt) to shutdown the badger gc goroutine in Close()
|
|
halt chan bool
|
|
|
|
// make BadgerDBWrapper.Close() idempotent, avoiding panic on double Close()
|
|
closed bool
|
|
|
|
// GcEveryDur controls how often the background goroutine
|
|
// runs garbage collection on the on-disk values-log.
|
|
// It defaults to running a GC every 1 minute if left as 0.
|
|
GcEveryDur time.Duration
|
|
|
|
// muGC ensures we only run one Garbage Collection at a time.
|
|
muGC sync.Mutex
|
|
|
|
hasher *Blake3Hasher
|
|
|
|
// doAllocZero sets the corresponding flag on all new BadgerTx.
|
|
// When doAllocZero is true, we zero out any data from badger
|
|
// after transcation commit and rollback. This simulates
|
|
// what would happen if we were to use the mmap-ed data
|
|
// from badger directly. Currently we copy by default for
|
|
// safety because otherwise TestAPI_ImportColumnAttrs sees
|
|
// corrupted data.
|
|
doAllocZero bool
|
|
}
|
|
|
|
// unprotectedListOpenTxAsString is a debugging helper.
|
|
// It is not thread safe, but is only used for debugging. Called internally while
|
|
// holding locks.
|
|
func (w *BadgerDBWrapper) unprotectedListOpenTxAsString() (r string) {
|
|
|
|
r = "openTx list = ["
|
|
for txn, write := range w.openTx {
|
|
r += fmt.Sprintf("txn p=%p(write:%v), ", txn, write)
|
|
}
|
|
return r + "]"
|
|
}
|
|
|
|
var _ = (*BadgerDBWrapper)(nil).unprotectedListOpenTxAsString // linter happy
|
|
|
|
// UnprotectedListOpenItAsString is exported because it is
|
|
// used for debugging in some of the pilosa_test tests.
|
|
// It is not thread safe, but only used for debugging. Called internally
|
|
// while holding locks and externally while not.
|
|
func (w *BadgerDBWrapper) UnprotectedListOpenItAsString() (r string) {
|
|
r = "openIt list = ["
|
|
for it, reverse := range w.openIt {
|
|
r += fmt.Sprintf("it p=%p(reverse:%v), ", it, reverse)
|
|
}
|
|
return r + "]"
|
|
}
|
|
|
|
// NewBadgerTx produces BadgerDB based ACID transactions. If
|
|
// the transaction will modify data, then the write flag must be true.
|
|
// Read-only queries should set write to false, to allow more concurrency.
|
|
// Methods on a BadgerTx are thread-safe, and can be called from
|
|
// different goroutines.
|
|
func (w *BadgerDBWrapper) NewBadgerTx(write bool) (tx *BadgerTx) {
|
|
w.muDb.Lock()
|
|
defer w.muDb.Unlock()
|
|
|
|
tx = &BadgerTx{
|
|
write: write,
|
|
tx: w.db.NewTransaction(write),
|
|
Db: w,
|
|
initloc: stack(),
|
|
doAllocZero: w.doAllocZero,
|
|
}
|
|
//vv("NewBadgerTx(write=%v) top, p=%p", write, tx)
|
|
//pp("NewBadgerTx(write=%v) top, p=%p, stack=\n\n'%v'", write, tx, stack())
|
|
|
|
if w.openTx == nil {
|
|
w.openTx = make(map[*BadgerTx]bool)
|
|
}
|
|
//pp("NewBadgerTx(write=%v); p=%p; (currently open txn: '%v', its: '%v'). initloc:'%v'", write, tx, w.unprotectedListOpenTxAsString(), w.UnprotectedListOpenItAsString(), tx.initloc)
|
|
w.muOpenTxIt.Lock()
|
|
w.openTx[tx] = write
|
|
w.muOpenTxIt.Unlock()
|
|
return
|
|
}
|
|
|
|
// Close shuts down the Badger database.
|
|
func (w *BadgerDBWrapper) Close() (err error) {
|
|
w.muDb.Lock()
|
|
defer w.muDb.Unlock()
|
|
if !w.closed {
|
|
close(w.halt)
|
|
w.closed = true
|
|
}
|
|
return w.db.Close()
|
|
}
|
|
|
|
// BadgerTx wraps a badger.Txn and provides the Tx interface
|
|
// method implementations.
|
|
// The methods on BadgerTx are thread-safe, and can be called
|
|
// from different goroutines.
|
|
type BadgerTx struct {
|
|
|
|
// mu serializes badger operations on this single txn instance.
|
|
//
|
|
// reference: https://godoc.org/github.com/dgraph-io/badger
|
|
// "Running [two separate -jea] transactions concurrently is OK. However, a
|
|
// transaction itself isn't thread safe, and should only
|
|
// be run serially. It doesn't matter if a transaction is
|
|
// created by one goroutine and passed down to other, as
|
|
// long as the Txn APIs are called serially."
|
|
mu sync.Mutex
|
|
|
|
write bool
|
|
Db *BadgerDBWrapper
|
|
tx *badger.Txn
|
|
|
|
opcount int
|
|
|
|
initloc string // stack trace of where we were initially created.
|
|
|
|
doAllocZero bool
|
|
|
|
// for tracking txn boundary issues, track all the memory
|
|
// that we deploy for roaring containers, and zero it on
|
|
// transaction commit/rollback.
|
|
ourAllocs [][]byte
|
|
ourContainers []*roaring.Container
|
|
}
|
|
|
|
func (tx *BadgerTx) UseRowCache() bool {
|
|
return false
|
|
}
|
|
|
|
// overWriteOurAllocs provides detection of memory
|
|
// access outside the transactional context, similar to the
|
|
// old school electric fence techniques but without setting
|
|
// memory mappings to read-only... instead we just zero
|
|
// out the memory allocated to roaring containers by a
|
|
// transaction after the commit or rollback. This,
|
|
// hopefully, will cause some downstream confusion and
|
|
// test failures, which we can use to locate who has been
|
|
// holding on to memory they should have copied prior
|
|
// to transaction commit.
|
|
func (tx *BadgerTx) overWriteOurAllocs() {
|
|
|
|
for _, s := range tx.ourAllocs {
|
|
|
|
// The Go compiler recognizes the following pattern and inserts
|
|
// an efficient memclr instruction.
|
|
// See https://github.com/golang/go/issues/5373
|
|
// and https://codereview.appspot.com/137880043
|
|
for i := range s {
|
|
s[i] = 0
|
|
}
|
|
}
|
|
// keep this around if we need to activate out-of-mmap memory access again.
|
|
//for _, v := range tx.ourContainers {
|
|
//v.Invalid = true
|
|
//v.Tx = tx
|
|
//}
|
|
}
|
|
|
|
// WholeDatabaseBlake3Hash returns the root-hash from the Merkle tree
|
|
// built by hashing all bits stored in the database backing this transaction.
|
|
func (tx *BadgerTx) WholeDatabaseBlake3Hash(index, field, view string, shard uint64) (hash string, err error) {
|
|
return
|
|
}
|
|
|
|
// Pointer gives us a memory address for the underlying transaction for debugging.
|
|
// It is public because we use it in roaring to report invalid container memory access
|
|
// outside of a transaction.
|
|
func (tx *BadgerTx) Pointer() string {
|
|
return fmt.Sprintf("%p", tx)
|
|
}
|
|
|
|
// Rollback rolls back the transaction.
|
|
func (tx *BadgerTx) Rollback() {
|
|
tx.mu.Lock()
|
|
defer tx.mu.Unlock()
|
|
|
|
//pp("BadgerTx.Rollback p=%p, its: '%v' initloc: '%v',\n rollbackloc:'%v'", tx, tx.Db.UnprotectedListOpenItAsString(), tx.initloc, stack())
|
|
tx.tx.Discard() // must hold tx.mu mutex lock
|
|
|
|
tx.Db.muOpenTxIt.Lock()
|
|
delete(tx.Db.openTx, tx)
|
|
tx.Db.muOpenTxIt.Unlock()
|
|
|
|
if tx.doAllocZero {
|
|
// and clear our allocs, to find code using them outside of a txn.
|
|
tx.overWriteOurAllocs()
|
|
}
|
|
}
|
|
|
|
// Commit commits the transaction to permanent storage.
|
|
// Commits can handle up to 100k updates to fragments
|
|
// at once, but not more. This is a BadgerDB imposed limit.
|
|
func (tx *BadgerTx) Commit() error {
|
|
tx.mu.Lock()
|
|
defer tx.mu.Unlock()
|
|
|
|
tx.Db.muOpenTxIt.Lock()
|
|
delete(tx.Db.openTx, tx)
|
|
tx.Db.muOpenTxIt.Unlock()
|
|
|
|
//pp("BadgerTx.Commit (write:%v) p=%p, stackID=%x openit: '%v' initloc: '%v', commitloc:\n%v", tx.write, tx, stackID, tx.Db.UnprotectedListOpenItAsString(), tx.initloc, stack())
|
|
|
|
err := tx.tx.Commit() // must hold tx.mu mutex lock
|
|
|
|
if tx.doAllocZero {
|
|
tx.overWriteOurAllocs()
|
|
}
|
|
return err
|
|
}
|
|
|
|
// Readonly returns true iff the BadgerTx is read-only.
|
|
func (tx *BadgerTx) Readonly() bool {
|
|
return !tx.write
|
|
}
|
|
|
|
// LeftShifted16MaxContainerKey is 0xffffffffffff0000. It is similar
|
|
// to the roaring.maxContainerKey 0x0000ffffffffffff, but
|
|
// shifted 16 bits to the left so its domain is the full [0, 2^64) bit space.
|
|
// It is used to match the semantics of the roaring.OffsetRange() API.
|
|
// This is the maximum endx value for Tx.OffsetRange(), because the lowbits,
|
|
// as in the roaring.OffsetRange(), are not allowed to be set.
|
|
// It is used in Tx.RoaringBitamp() to obtain the full contents of a fragment
|
|
// from a call from tx.OffsetRange() by requesting [0, LeftShifted16MaxContainerKey)
|
|
// with an offset of 0.
|
|
const LeftShifted16MaxContainerKey = uint64(0xffffffffffff0000) // or math.MaxUint64 - (1<<16 - 1), or 18446744073709486080
|
|
|
|
// RoaringBitmap returns the roaring.Bitmap for all bits in the fragment.
|
|
func (tx *BadgerTx) RoaringBitmap(index, field, view string, shard uint64) (*roaring.Bitmap, error) {
|
|
|
|
return tx.OffsetRange(index, field, view, shard, 0, 0, LeftShifted16MaxContainerKey)
|
|
}
|
|
|
|
// badgerKey produces the bytes that we use as a key to query badger.
|
|
// The roaringContainerKey argument is a container key into a roaring Container.
|
|
// Output examples:
|
|
//
|
|
// "idx:'i';fld:'f';vw:'standard';shd:'0';ckey@00000000000000000000" // smallest container-key
|
|
// "idx:'i';fld:'f';vw:'standard';shd:'0';ckey@18446744073709551615" // largest container-key (math.MaxUint64)
|
|
//
|
|
// NB must be kept in sync with badgerPrefix() and badgerKeyExtractContainerKey().
|
|
//
|
|
func badgerKey(index, field, view string, shard uint64, roaringContainerKey uint64) []byte {
|
|
// The %020d which adds zero padding up to 20 runes is required to
|
|
// allow the textual sort to accurately
|
|
// reflect a numeric sort order. This is because, as a string,
|
|
// math.MaxUint64 is 20 bytes long.
|
|
// Example of such a badgerKey with a container-key that is math.MaxUint64:
|
|
// ...........................................12345678901234567890
|
|
// idx:'i';fld:'f';vw:'standard';shd:'1';ckey@18446744073709551615
|
|
|
|
prefix := badgerPrefix(index, field, view, shard)
|
|
ckey := []byte(fmt.Sprintf("%020d", roaringContainerKey))
|
|
return append(prefix, ckey...)
|
|
}
|
|
|
|
// badgerKeyAndPrefix returns the equivalent of badgerKey() and badgerPrefix() calls.
|
|
func badgerKeyAndPrefix(index, field, view string, shard uint64, roaringContainerKey uint64) (key, prefix []byte) {
|
|
prefix = badgerPrefix(index, field, view, shard)
|
|
ckey := []byte(fmt.Sprintf("%020d", roaringContainerKey))
|
|
return append(prefix, ckey...), prefix
|
|
}
|
|
|
|
var _ = badgerKeyAndPrefix // keep linter happy
|
|
|
|
// badgerKeyExtractContainerKey extracts the containerKey from bkey.
|
|
func badgerKeyExtractContainerKey(bkey []byte) (containerKey uint64) {
|
|
|
|
// The zero padding means that the container-key is always the last 20 bytes of the bkey.
|
|
//
|
|
// Be sure to catch the problematic case of a user passing in only a prefix. A prefix
|
|
// ends in 'key@' rather than a full key that has 'key@00000000000000000001' (for example)
|
|
// at the end. The ParseUint call below will fail in that case.
|
|
n := len(bkey)
|
|
if n < 20 {
|
|
panic(fmt.Sprintf("badgerKeyExtractContainerKey() error: bad bkey '%v', too short!", string(bkey)))
|
|
}
|
|
last := bkey[n-20:] // badgerKey() and badgerPrefix() always return more than 20 rune []byte.
|
|
var err error
|
|
containerKey, err = strconv.ParseUint(string(last), 10, 64) // has to be the container key
|
|
if err != nil {
|
|
panic(fmt.Sprintf("badgerKeyExtractContainerKey() error: bad bkey '%v', could not convert last 20 bytes ('%v') to a unit64: '%v'", string(bkey), string(last), err))
|
|
}
|
|
return
|
|
}
|
|
|
|
// badgerPrefix returns everything from badgerKey up to and
|
|
// including the '@' fune in a badger key. The prefix excludes the roaring container key itself.
|
|
// NB must be kept in sync with badgerKey() and badgerKeyExtractContainerKey().
|
|
func badgerPrefix(index, field, view string, shard uint64) []byte {
|
|
return []byte(fmt.Sprintf("idx:'%v';fld:'%v';vw:'%v';shd:'%x';ckey@", index, field, view, shard))
|
|
}
|
|
|
|
// badgerIndexOnlyPrefix returns a prefix suitable for DeleteIndex and a key-scan to
|
|
// remove all storage associated with one index.
|
|
//
|
|
// The full name of the index must be provided, no partial index names will work.
|
|
//
|
|
// The provided key is terminated by `';` and so DeleteIndex("i") will not delete the index "i2".
|
|
//
|
|
func badgerIndexOnlyPrefix(indexName string) []byte {
|
|
return []byte(fmt.Sprintf("idx:'%v';", indexName))
|
|
}
|
|
|
|
// Container returns the requested roaring.Container, selected by fragment and ckey
|
|
func (tx *BadgerTx) Container(index, field, view string, shard uint64, ckey uint64) (c *roaring.Container, err error) {
|
|
|
|
// values returned from Get() are only valid while the transaction
|
|
// is open. If you need to use a value outside of the transaction then
|
|
// you must use copy() to copy it to another byte slice.
|
|
// BUT here we are already inside the Txn.
|
|
|
|
bkey := badgerKey(index, field, view, shard, ckey)
|
|
tx.mu.Lock()
|
|
var item *badger.Item
|
|
item, err = tx.tx.Get(bkey)
|
|
tx.mu.Unlock()
|
|
if err == badger.ErrKeyNotFound {
|
|
// Seems crazy, but we, for now at least,
|
|
// match what RoaringTx does by returning nil, nil.
|
|
return nil, nil
|
|
} else {
|
|
panicOn(err)
|
|
}
|
|
|
|
err = item.Value(func(v []byte) error {
|
|
// This func with val would only be called if item.Value encounters no error
|
|
c = tx.toContainer(item.UserMeta(), v)
|
|
return nil
|
|
})
|
|
panicOn(err)
|
|
return
|
|
}
|
|
|
|
// PutContainer stores rc under the specified fragment and container ckey.
|
|
func (tx *BadgerTx) PutContainer(index, field, view string, shard uint64, ckey uint64, rc *roaring.Container) error {
|
|
|
|
bkey := badgerKey(index, field, view, shard, ckey)
|
|
var by []byte
|
|
|
|
ct := roaring.ContainerType(rc)
|
|
|
|
switch ct {
|
|
case containerArray:
|
|
by = fromArray16(roaring.AsArray(rc))
|
|
case containerBitmap:
|
|
by = fromArray64(roaring.AsBitmap(rc))
|
|
case containerRun:
|
|
by = fromInterval16(roaring.AsRuns(rc))
|
|
case containerNil:
|
|
panic("wat? nil container is unexpected, no?!?")
|
|
default:
|
|
panic(fmt.Sprintf("unknown container type: %v", ct))
|
|
}
|
|
entry := badger.NewEntry(bkey, by).WithMeta(ct)
|
|
tx.mu.Lock()
|
|
err := tx.tx.SetEntry(entry)
|
|
tx.mu.Unlock()
|
|
|
|
// ErrTxnTooBig is returned if too many writes are fit into a single transaction.
|
|
// badger docs: "An ErrTxnTooBig will be reported in case the number of pending
|
|
// writes/deletes in the transaction exceeds a certain limit. In that case, it
|
|
// is best to commit the transaction and start a new transaction immediately."
|
|
//
|
|
if err == badger.ErrTxnTooBig {
|
|
// As now, we don't deal with this. The current strategy is to recommend setting lots
|
|
// of bits on your container and then change it in a single operation
|
|
// within the txn, rather than having too many SetEntry() calls in a transactions.
|
|
// The tests currently have these default limits:
|
|
// maxBatchCount:104857, maxBatchSize:10066329
|
|
// For now, we just panic. The user should re-write their code to do
|
|
// most of the work of setting bits outside the transaction.
|
|
panic(fmt.Sprintf("error: do not do more than 100K writes in a transaction: '%v'", err))
|
|
}
|
|
return err
|
|
}
|
|
|
|
// RemoveContainer deletes the container specified by the shard and container key ckey
|
|
func (tx *BadgerTx) RemoveContainer(index, field, view string, shard uint64, ckey uint64) error {
|
|
bkey := badgerKey(index, field, view, shard, ckey)
|
|
tx.mu.Lock()
|
|
err := tx.tx.Delete(bkey)
|
|
tx.mu.Unlock()
|
|
return err
|
|
}
|
|
|
|
// Add sets all the a bits hot in the specified fragment.
|
|
func (tx *BadgerTx) Add(index, field, view string, shard uint64, batched bool, a ...uint64) (changeCount int, err error) {
|
|
|
|
// pure hack to match RoaringTx
|
|
defer func() {
|
|
if !batched {
|
|
if changeCount > 0 {
|
|
changeCount = 1
|
|
}
|
|
}
|
|
}()
|
|
|
|
// TODO: optimization: group 'a' elements into their containers,
|
|
// and then do all the Adds on that
|
|
// container at once, so we don't retrieve a container per bit.
|
|
// (maybe, for example, using ImportRoaringBits with clear=false).
|
|
|
|
for _, v := range a {
|
|
hi, lo := highbits(v), lowbits(v)
|
|
|
|
var rct *roaring.Container
|
|
rct, err = tx.Container(index, field, view, shard, hi)
|
|
panicOn(err)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
chng := false
|
|
// TODO optimization: set all the bits in the current container at once. group by container first.
|
|
rc1, chng := rct.Add(lo)
|
|
panicOn(err)
|
|
if chng {
|
|
changeCount++
|
|
}
|
|
if err != nil {
|
|
return changeCount, err
|
|
}
|
|
err = tx.PutContainer(index, field, view, shard, hi, rc1)
|
|
panicOn(err)
|
|
}
|
|
return
|
|
}
|
|
|
|
// Remove clears all the specified a bits in the chosen fragment.
|
|
func (tx *BadgerTx) Remove(index, field, view string, shard uint64, a ...uint64) (changeCount int, err error) {
|
|
|
|
// TODO: optimization: group 'a' elements into their containers,
|
|
// and then do all the Removes on that
|
|
// container at once, so we don't retrieve a container per bit.
|
|
// (maybe, for example, using ImportRoaringBits with clear=true).
|
|
for _, v := range a {
|
|
hi, lo := highbits(v), lowbits(v)
|
|
|
|
var rct *roaring.Container
|
|
rct, err = tx.Container(index, field, view, shard, hi)
|
|
panicOn(err)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
chng := false
|
|
rc1, chng := rct.Remove(lo)
|
|
panicOn(err)
|
|
if chng {
|
|
changeCount++
|
|
}
|
|
if err != nil {
|
|
return changeCount, err
|
|
}
|
|
if rc1.N() == 0 {
|
|
err = tx.RemoveContainer(index, field, view, shard, hi)
|
|
if err != nil {
|
|
return
|
|
}
|
|
} else {
|
|
err = tx.PutContainer(index, field, view, shard, hi, rc1)
|
|
panicOn(err)
|
|
}
|
|
}
|
|
return
|
|
}
|
|
|
|
// Contains returns exists true iff the bit chosen by key is
|
|
// hot (set to 1) in specified fragment.
|
|
func (tx *BadgerTx) Contains(index, field, view string, shard uint64, key uint64) (exists bool, err error) {
|
|
|
|
lo, hi := lowbits(key), highbits(key)
|
|
bkey := badgerKey(index, field, view, shard, hi)
|
|
tx.mu.Lock()
|
|
item, err := tx.tx.Get(bkey)
|
|
tx.mu.Unlock()
|
|
if err == badger.ErrKeyNotFound {
|
|
return false, nil
|
|
}
|
|
if err != nil {
|
|
return false, err
|
|
}
|
|
err = item.Value(func(v []byte) error {
|
|
// This func with val would only be called if item.Value encounters no error
|
|
c := tx.toContainer(item.UserMeta(), v)
|
|
exists = c.Contains(lo)
|
|
return nil
|
|
})
|
|
return exists, err
|
|
}
|
|
|
|
// key is the container key for the first roaring Container
|
|
// roaring docs: Iterator returns a ContainterIterator which *after* a call to Next(), a call to Value() will
|
|
// return the first container at or after key. found will be true if a
|
|
// container is found at key.
|
|
//
|
|
// BadgerTx notes: We auto-stop at the end of this shard, not going beyond.
|
|
func (tx *BadgerTx) ContainerIterator(index, field, view string, shard uint64, firstRoaringContainerKey uint64) (citer roaring.ContainerIterator, found bool, err error) {
|
|
|
|
// needle example: "index:'i';field:'f';view:'v';shard:'0';key@00000000000000000000"
|
|
needle := badgerKey(index, field, view, shard, firstRoaringContainerKey)
|
|
|
|
// prefix example: "index:'i';field:'f';view:'v';shard:'0';key@"
|
|
prefix := badgerPrefix(index, field, view, shard)
|
|
|
|
bi := NewBadgerIterator(tx, prefix)
|
|
bi.Seek(needle)
|
|
if !bi.it.Valid() {
|
|
return bi, false, nil
|
|
}
|
|
|
|
if !bi.it.ValidForPrefix(prefix) {
|
|
return bi, false, nil
|
|
}
|
|
return bi, true, nil
|
|
}
|
|
|
|
// BadgerIterator is the iterator returned from a BadgerTx.ContainerIterator() call.
|
|
// It implements the roaring.ContainerIterator interface.
|
|
type BadgerIterator struct {
|
|
tx *BadgerTx
|
|
it *badger.Iterator
|
|
|
|
prefix []byte
|
|
seekto []byte
|
|
|
|
// seen counts how many Next() calls we have seen.
|
|
// It is used to match roaring.ContainerIterator semantics.
|
|
// Also useful for testing.
|
|
seen int
|
|
}
|
|
|
|
// NewBadgerIterator creates an iterator on tx that will
|
|
// only return badgerKeys that start with prefix.
|
|
func NewBadgerIterator(tx *BadgerTx, prefix []byte) (bi *BadgerIterator) {
|
|
|
|
tx.Db.muOpenTxIt.Lock()
|
|
defer tx.Db.muOpenTxIt.Unlock()
|
|
|
|
defer func() {
|
|
r := recover()
|
|
if r != nil {
|
|
panic(r)
|
|
}
|
|
}()
|
|
opts := badger.DefaultIteratorOptions
|
|
opts.PrefetchValues = false // else by default, pre-fetches the 1st 100 values, which would be slow.
|
|
opts.Reverse = false
|
|
|
|
tx.mu.Lock()
|
|
it := tx.tx.NewIterator(opts)
|
|
tx.mu.Unlock()
|
|
|
|
bi = &BadgerIterator{
|
|
tx: tx,
|
|
it: it,
|
|
prefix: prefix,
|
|
}
|
|
if tx.Db.openIt == nil {
|
|
tx.Db.openIt = make(map[*BadgerIterator]bool)
|
|
}
|
|
tx.Db.openIt[bi] = false // true for reverse, false for forward iteration.
|
|
bi.it.Seek(prefix)
|
|
return
|
|
}
|
|
|
|
// NewBadgerReverseIterator makes a highest-to-lowest key iterator.
|
|
// Only keys that are prefixed with prefix will be returned.
|
|
// seekto tells where to start, and should be typically shard+1
|
|
// to start at the end of shard. Really only used in Max() at the moment.
|
|
// After creating a reverse badger iterator it, we will call it.Seek(seekto).
|
|
func NewBadgerReverseIterator(tx *BadgerTx, prefix, seekto []byte) (bi *BadgerIterator) {
|
|
|
|
tx.Db.muOpenTxIt.Lock()
|
|
defer tx.Db.muOpenTxIt.Unlock()
|
|
|
|
opts := badger.DefaultIteratorOptions
|
|
opts.PrefetchValues = false // else by default, pre-fetches the 1st 100 values, which would be slow.
|
|
opts.Reverse = true
|
|
opts.Prefix = prefix // possible storage IOPs optimization by badger
|
|
tx.mu.Lock()
|
|
it := tx.tx.NewIterator(opts)
|
|
tx.mu.Unlock()
|
|
|
|
bi = &BadgerIterator{
|
|
tx: tx,
|
|
it: it,
|
|
prefix: prefix,
|
|
seekto: seekto,
|
|
}
|
|
if tx.Db.openIt == nil {
|
|
tx.Db.openIt = make(map[*BadgerIterator]bool)
|
|
}
|
|
bi.tx.Db.openIt[bi] = true // true for reverse, false for forward iteration.
|
|
bi.it.Seek(seekto)
|
|
return
|
|
}
|
|
|
|
// Close tells the database and transaction that the user is done
|
|
// with the iterator.
|
|
// From the badger docs: It is important to call this when you're done with iteration.
|
|
// else you will get an error on tx.Discard()/Commit().
|
|
func (bi *BadgerIterator) Close() {
|
|
|
|
bi.tx.Db.muOpenTxIt.Lock()
|
|
delete(bi.tx.Db.openIt, bi)
|
|
bi.it.Close()
|
|
|
|
bi.tx.Db.muOpenTxIt.Unlock()
|
|
}
|
|
|
|
// Valid returns false if there are no more values in the iterator's range.
|
|
func (bi *BadgerIterator) Valid() bool {
|
|
return bi.it.Valid()
|
|
}
|
|
|
|
// Seek allows the iterator to start at needle instead of the global begining.
|
|
func (bi *BadgerIterator) Seek(needle []byte) {
|
|
bi.it.Seek(needle)
|
|
}
|
|
|
|
// Next advances the iterator.
|
|
func (bi *BadgerIterator) Next() bool {
|
|
|
|
// have to skip the first bi.it.Next() call because badger iterators point to the
|
|
// first value immediately, but Pilosa iterators must have Next() called
|
|
// on a fresh iterator to get the first value.
|
|
if bi.seen > 0 {
|
|
bi.it.Next()
|
|
}
|
|
bi.seen++
|
|
return bi.it.ValidForPrefix(bi.prefix) // does the bi.it.Valid() inside and false if not valid always.
|
|
}
|
|
|
|
// Value retrieves what is pointed at currently by the iterator.
|
|
func (bi *BadgerIterator) Value() (containerKey uint64, c *roaring.Container) {
|
|
if !bi.it.Valid() {
|
|
panic("bi.it not valid")
|
|
}
|
|
item := bi.it.Item()
|
|
if item == nil {
|
|
panic("item was nil")
|
|
}
|
|
key := item.Key()
|
|
containerKey = badgerKeyExtractContainerKey(key)
|
|
|
|
err := item.Value(func(v []byte) error {
|
|
c = bi.tx.toContainer(item.UserMeta(), v)
|
|
return nil
|
|
})
|
|
panicOn(err)
|
|
return
|
|
}
|
|
|
|
// Closer is used by badgerFinder
|
|
type Closer interface {
|
|
Close()
|
|
}
|
|
|
|
// badgerFinder implements roaring.IteratorFinder.
|
|
// It is used by BadgerTx.ForEach()
|
|
type badgerFinder struct {
|
|
tx *BadgerTx
|
|
index string
|
|
field string
|
|
view string
|
|
shard uint64
|
|
needClose []Closer
|
|
}
|
|
|
|
// FindIterator lets badgerFinder implement the roaring.FindIterator interface.
|
|
func (bf *badgerFinder) FindIterator(seek uint64) (roaring.ContainerIterator, bool) {
|
|
a, found, err := bf.tx.ContainerIterator(bf.index, bf.field, bf.view, bf.shard, seek)
|
|
panicOn(err)
|
|
bf.needClose = append(bf.needClose, a)
|
|
return a, found
|
|
}
|
|
|
|
// Close closes all bf.needClose listed Closers.
|
|
func (bf *badgerFinder) Close() {
|
|
for _, i := range bf.needClose {
|
|
i.Close()
|
|
}
|
|
}
|
|
|
|
// NewTxIterator returns a *roaring.Iterator that MUST have Close() called on it BEFORE
|
|
// the transaction Commits or Rollsback.
|
|
func (tx *BadgerTx) NewTxIterator(index, field, view string, shard uint64) *roaring.Iterator {
|
|
bf := &badgerFinder{tx: tx, index: index, field: field, view: view, shard: shard, needClose: make([]Closer, 0)}
|
|
itr := roaring.NewIterator(bf)
|
|
return itr
|
|
}
|
|
|
|
// ForEach applies fn to each bitmap in the fragment.
|
|
func (tx *BadgerTx) ForEach(index, field, view string, shard uint64, fn func(i uint64) error) error {
|
|
itr := tx.NewTxIterator(index, field, view, shard)
|
|
defer itr.Close()
|
|
|
|
// Seek can create many container iterators, thus bf.Close() needClose list.
|
|
itr.Seek(0)
|
|
// v is the bit we are operating on.
|
|
for v, eof := itr.Next(); !eof; v, eof = itr.Next() {
|
|
if err := fn(v); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// ForEachRange applies fn on the selected range of bits on the chosen fragment.
|
|
func (tx *BadgerTx) ForEachRange(index, field, view string, shard uint64, start, end uint64, fn func(uint64) error) error {
|
|
|
|
itr := tx.NewTxIterator(index, field, view, shard)
|
|
defer itr.Close()
|
|
|
|
itr.Seek(start)
|
|
|
|
// v is the bit we are operating on.
|
|
for v, eof := itr.Next(); !eof && v < end; v, eof = itr.Next() {
|
|
if err := fn(v); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// Count operates on the full bitmap level, so it sums over all the containers
|
|
// in the bitmap.
|
|
func (tx *BadgerTx) Count(index, field, view string, shard uint64) (uint64, error) {
|
|
|
|
a, found, err := tx.ContainerIterator(index, field, view, shard, 0)
|
|
panicOn(err)
|
|
defer a.Close()
|
|
if !found {
|
|
return 0, nil
|
|
}
|
|
result := int32(0)
|
|
for a.Next() {
|
|
ckey, cont := a.Value()
|
|
_ = ckey
|
|
result += cont.N()
|
|
}
|
|
|
|
return uint64(result), nil
|
|
}
|
|
|
|
// Max is the maximum bit-value in your bitmap.
|
|
func (tx *BadgerTx) Max(index, field, view string, shard uint64) (uint64, error) {
|
|
|
|
prefix := badgerPrefix(index, field, view, shard)
|
|
seekto := badgerPrefix(index, field, view, shard+1)
|
|
|
|
it := NewBadgerReverseIterator(tx, prefix, seekto) // this iterator is still open, when we commit/discard tx.
|
|
defer it.Close()
|
|
|
|
hb, rc := it.Value()
|
|
lb := rc.Max()
|
|
|
|
return hb<<16 | uint64(lb), nil
|
|
}
|
|
|
|
// Min returns the smallest bit set in the fragment. If no bit is hot,
|
|
// the second return argument is false.
|
|
func (tx *BadgerTx) Min(index, field, view string, shard uint64) (uint64, bool, error) {
|
|
|
|
// Seek can create many container iterators, thus the bf.Close() needClose list.
|
|
bf := &badgerFinder{tx: tx, index: index, field: field, view: view, shard: shard, needClose: make([]Closer, 0)}
|
|
defer bf.Close()
|
|
itr := roaring.NewIterator(bf)
|
|
|
|
itr.Seek(0)
|
|
|
|
// v is the bit we are operating on.
|
|
v, eof := itr.Next()
|
|
if eof {
|
|
return 0, false, nil
|
|
}
|
|
return v, true, nil
|
|
}
|
|
|
|
// UnionInPlace unions all the others Bitmaps into a new Bitmap, and then writes it to the
|
|
// specified fragment.
|
|
func (tx *BadgerTx) UnionInPlace(index, field, view string, shard uint64, others ...*roaring.Bitmap) error {
|
|
|
|
rbm, err := tx.RoaringBitmap(index, field, view, shard)
|
|
panicOn(err)
|
|
|
|
rbm.UnionInPlace(others...)
|
|
// iterate over the containers that changed within rbm, and write them back to disk.
|
|
|
|
it, found := rbm.Containers.Iterator(0)
|
|
_ = found // don't care about the value of found, because first containerKey might be > 0
|
|
|
|
for it.Next() {
|
|
containerKey, rc := it.Value()
|
|
|
|
// TODO: only write the changed ones back, as optimization?
|
|
// Compare to ImportRoaringBits.
|
|
err := tx.PutContainer(index, field, view, shard, containerKey, rc)
|
|
panicOn(err)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// CountRange returns the count of hot bits in the start, end range on the fragment.
|
|
func (tx *BadgerTx) CountRange(index, field, view string, shard uint64, start, end uint64) (n uint64, err error) {
|
|
|
|
skey := highbits(start)
|
|
ekey := highbits(end)
|
|
|
|
citer, found, err := tx.ContainerIterator(index, field, view, shard, skey)
|
|
panicOn(err)
|
|
|
|
defer citer.Close() // doesn't seem to be getting called.
|
|
if !found {
|
|
return 0, nil
|
|
}
|
|
// If range is entirely in one container then just count that range.
|
|
if skey == ekey {
|
|
citer.Next()
|
|
_, c := citer.Value()
|
|
return uint64(c.CountRange(int32(lowbits(start)), int32(lowbits(end)))), nil
|
|
}
|
|
|
|
for citer.Next() {
|
|
k, c := citer.Value()
|
|
if k < skey {
|
|
citer.Close()
|
|
panic(fmt.Sprintf("should be impossible for k(%v) to be less than skey(%v). tx p=%p", k, skey, tx))
|
|
}
|
|
|
|
// k > ekey handles the case when start > end and where start and end
|
|
// are in different containers. Same container case is already handled above.
|
|
if k > ekey {
|
|
break
|
|
}
|
|
if k == skey {
|
|
n += uint64(c.CountRange(int32(lowbits(start)), roaring.MaxContainerVal+1))
|
|
continue
|
|
}
|
|
if k < ekey {
|
|
n += uint64(c.N())
|
|
continue
|
|
}
|
|
if k == ekey {
|
|
n += uint64(c.CountRange(0, int32(lowbits(end))))
|
|
break
|
|
}
|
|
}
|
|
|
|
return n, nil
|
|
}
|
|
|
|
// OffsetRange creates a new roaring.Bitmap to return in other. For all the
|
|
// hot bits in [start, endx) of the chosen fragment, it stores
|
|
// them into other but with offset added to their bit position.
|
|
// The primary client is doing this, using ShardWidth, already; see
|
|
// fragment.rowFromStorage() in fragment.go. For example:
|
|
//
|
|
// data, err := tx.OffsetRange(f.index, f.field, f.view, f.shard,
|
|
// f.shard*ShardWidth, rowID*ShardWidth, (rowID+1)*ShardWidth)
|
|
// ^ offset ^ start ^ endx
|
|
//
|
|
// The start and endx arguments are container keys that have been shifted left by 16 bits;
|
|
// their highbits() will be taken to determine the actual container keys. This
|
|
// is done to conform to the roaring.OffsetRange() argument convention.
|
|
//
|
|
func (tx *BadgerTx) OffsetRange(index, field, view string, shard, offset, start, endx uint64) (other *roaring.Bitmap, err error) {
|
|
|
|
// roaring does these three checks in its OffsetRange
|
|
if lowbits(offset) != 0 {
|
|
panic("offset must not contain low bits")
|
|
}
|
|
if lowbits(start) != 0 {
|
|
panic("range start must not contain low bits")
|
|
}
|
|
if lowbits(endx) != 0 {
|
|
panic("range end must not contain low bits")
|
|
}
|
|
|
|
other = roaring.NewSliceBitmap()
|
|
off := highbits(offset)
|
|
hi0, hi1 := highbits(start), highbits(endx)
|
|
|
|
// TODO(jea): question: do we have to account for ShardWidth here? what if the move goes
|
|
// beyond a shard?
|
|
|
|
needle := badgerKey(index, field, view, shard, hi0)
|
|
prefix := badgerPrefix(index, field, view, shard)
|
|
|
|
n2, pre2 := badgerKeyAndPrefix(index, field, view, shard, hi0)
|
|
if string(n2) != string(needle) {
|
|
panic(fmt.Sprintf("problem! n2(%v) != needle(%v), badgerKeyAndPrefix not consitent with badgerKey()", string(n2), string(needle)))
|
|
}
|
|
if string(pre2) != string(prefix) {
|
|
panic(fmt.Sprintf("problem! pre2(%v) != prefix(%v), badgerKeyAndPrefix not consitent with badgerKey()", string(pre2), string(prefix)))
|
|
}
|
|
|
|
it := NewBadgerIterator(tx, prefix) // see OffsetRange() panic 'Only one iterator can be active at one time, for a RW txn
|
|
defer it.Close()
|
|
it.Seek(needle)
|
|
for ; it.it.ValidForPrefix(prefix); it.Next() {
|
|
item := it.it.Item()
|
|
bkey := item.Key()
|
|
k := badgerKeyExtractContainerKey(bkey)
|
|
|
|
if uint64(k) >= hi1 {
|
|
break
|
|
}
|
|
destCkey := off + (k - hi0)
|
|
err := item.Value(func(v []byte) error {
|
|
|
|
c := tx.toContainer(item.UserMeta(), v)
|
|
other.Containers.Put(destCkey, c.Freeze())
|
|
|
|
return nil
|
|
})
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
return other, nil
|
|
}
|
|
|
|
// IncrementOpN increments the tx opcount by changedN
|
|
func (tx *BadgerTx) IncrementOpN(index, field, view string, shard uint64, changedN int) {
|
|
tx.opcount += changedN
|
|
}
|
|
|
|
// ImportRoaringBits handles deletes by setting clear=true.
|
|
// rowSet[rowID] returns the number of bit changed on that rowID.
|
|
func (tx *BadgerTx) ImportRoaringBits(index, field, view string, shard uint64, itr roaring.RoaringIterator, clear bool, log bool, rowSize uint64) (changed int, rowSet map[uint64]int, err error) {
|
|
n := itr.Len()
|
|
if n == 0 {
|
|
return
|
|
}
|
|
rowSet = make(map[uint64]int)
|
|
|
|
var currRow uint64
|
|
|
|
var oldC *roaring.Container
|
|
for itrKey, synthC := itr.NextContainer(); synthC != nil; itrKey, synthC = itr.NextContainer() {
|
|
if rowSize != 0 {
|
|
currRow = itrKey / rowSize
|
|
}
|
|
nsynth := int(synthC.N())
|
|
if nsynth == 0 {
|
|
continue
|
|
}
|
|
// INVAR: nsynth > 0
|
|
|
|
oldC, err = tx.Container(index, field, view, shard, itrKey)
|
|
panicOn(err)
|
|
if err != nil {
|
|
return
|
|
}
|
|
|
|
if oldC == nil || oldC.N() == 0 {
|
|
// no container at the itrKey in badger (or all zero container).
|
|
if clear {
|
|
// changed of 0 and empty rowSet is perfect, no need to change the defaults.
|
|
continue
|
|
} else {
|
|
|
|
changed += nsynth
|
|
rowSet[currRow] += nsynth
|
|
|
|
err = tx.PutContainer(index, field, view, shard, itrKey, synthC)
|
|
if err != nil {
|
|
return
|
|
}
|
|
continue
|
|
}
|
|
}
|
|
|
|
if clear {
|
|
existN := oldC.N() // number of bits set in the old container
|
|
newC := oldC.Difference(synthC)
|
|
|
|
// update rowSet and changes
|
|
if newC.N() == existN {
|
|
// INVAR: do changed need adjusting? nope. same bit count,
|
|
// so no change could have happened.
|
|
continue
|
|
} else {
|
|
changes := int(existN - newC.N())
|
|
changed += changes
|
|
rowSet[currRow] -= changes
|
|
|
|
if newC.N() == 0 {
|
|
err = tx.RemoveContainer(index, field, view, shard, itrKey)
|
|
if err != nil {
|
|
return
|
|
}
|
|
continue
|
|
}
|
|
err = tx.PutContainer(index, field, view, shard, itrKey, newC)
|
|
if err != nil {
|
|
return
|
|
}
|
|
continue
|
|
}
|
|
} else {
|
|
// setting bits
|
|
|
|
existN := oldC.N()
|
|
if existN == roaring.MaxContainerVal+1 {
|
|
// completely full container already, set will do nothing. so changed of 0 default is perfect.
|
|
continue
|
|
}
|
|
if existN == 0 {
|
|
// can nsynth be zero? No, because of the continue/invariant above where nsynth > 0
|
|
changed += nsynth
|
|
rowSet[currRow] += nsynth
|
|
err = tx.PutContainer(index, field, view, shard, itrKey, synthC)
|
|
if err != nil {
|
|
return
|
|
}
|
|
continue
|
|
}
|
|
|
|
newC := oldC.UnionInPlace(synthC)
|
|
|
|
if roaring.ContainerType(newC) == containerBitmap {
|
|
newC.Repair() // update the bit-count so .n is valid. b/c UnionInPlace doesn't update it.
|
|
}
|
|
if newC.N() != existN {
|
|
changes := int(newC.N() - existN)
|
|
changed += changes
|
|
rowSet[currRow] += changes
|
|
|
|
err = tx.PutContainer(index, field, view, shard, itrKey, newC)
|
|
if err != nil {
|
|
panicOn(err)
|
|
return
|
|
}
|
|
continue
|
|
}
|
|
}
|
|
}
|
|
return
|
|
}
|
|
|
|
//////////////////////////////////
|
|
// badger helper utility functions
|
|
|
|
func highbits(v uint64) uint64 { return v >> 16 }
|
|
func lowbits(v uint64) uint16 { return uint16(v & 0xFFFF) }
|
|
|
|
func toArray16(a []byte) []uint16 {
|
|
return (*[4096]uint16)(unsafe.Pointer(&a[0]))[: len(a)/2 : len(a)/2]
|
|
}
|
|
func toArray64(a []byte) []uint64 {
|
|
return (*[1024]uint64)(unsafe.Pointer(&a[0]))[:1024:1024]
|
|
}
|
|
func toInterval16(a []byte) []roaring.Interval16 {
|
|
return (*[2048]roaring.Interval16)(unsafe.Pointer(&a[0]))[: len(a)/4 : len(a)/4]
|
|
}
|
|
|
|
// should really be exported from the pilosa/roaring package so we don't get out of sync...
|
|
const (
|
|
containerNil byte = iota // no container
|
|
containerArray // slice of bit position values
|
|
containerBitmap // slice of 1024 uint64s
|
|
containerRun // container of run-encoded bits
|
|
)
|
|
|
|
func (tx *BadgerTx) toContainer(typ byte, v []byte) (r *roaring.Container) {
|
|
|
|
// For safety we copy v, since it lives in BadgerDB's memory-mapped vlog-file,
|
|
// and Badger will recycle it after tx ends with rollback or commit.
|
|
// We copy into Go runtime GC managed memory. Technically we don't need
|
|
// to do this if all of our use stays within the lifetime
|
|
// of the badger transaction we were started on. Hence:
|
|
//
|
|
// TODO: performance tuning might want w := v here, if we can guarantee no access to memory past the Tx lifetime.
|
|
//
|
|
// Problem is, at least some tests appear to not respect transaction boundaries...
|
|
w := make([]byte, len(v))
|
|
copy(w, v) // green go test -v -run TestAPI_ImportColumnAttrs
|
|
//w := v // if instead of append we use v directly, it causes red: go test -v -run TestAPI_ImportColumnAttrs
|
|
|
|
// register w so we can catch out-of-tx memory access
|
|
tx.ourAllocs = append(tx.ourAllocs, w)
|
|
|
|
switch typ {
|
|
case containerArray:
|
|
c := roaring.NewContainerArray(toArray16(w))
|
|
tx.ourContainers = append(tx.ourContainers, c)
|
|
return c
|
|
case containerBitmap:
|
|
c := roaring.NewContainerBitmap(-1, toArray64(w))
|
|
tx.ourContainers = append(tx.ourContainers, c)
|
|
return c
|
|
case containerRun:
|
|
c := roaring.NewContainerRun(toInterval16(w))
|
|
tx.ourContainers = append(tx.ourContainers, c)
|
|
return c
|
|
default:
|
|
panic(fmt.Sprintf("unknown container: %v", typ))
|
|
}
|
|
}
|
|
|
|
// fromArray16 converts to an 8KB page
|
|
func fromArray16(a []uint16) []byte {
|
|
return (*[8192]byte)(unsafe.Pointer(&a[0]))[: len(a)*2 : len(a)*2]
|
|
}
|
|
|
|
// fromArray64 converts to an 8KB page
|
|
func fromArray64(a []uint64) []byte {
|
|
return (*[8192]byte)(unsafe.Pointer(&a[0]))[:8192:8192]
|
|
}
|
|
|
|
// fromInterval16 converts to 8KB page
|
|
func fromInterval16(a []roaring.Interval16) []byte {
|
|
return (*[8192]byte)(unsafe.Pointer(&a[0]))[: len(a)*4 : len(a)*4]
|
|
}
|
|
|
|
// badgerKey method on fragment creates a query key in the
|
|
// standard format by invoking the top level badgerKey with
|
|
// the container key being highbits(rowID * ShardWidth).
|
|
//
|
|
// Commented out for now only to keep the golangci-lint happy,
|
|
// as it has no users at the moment.
|
|
//func (f *fragment) badgerKey(rowID uint64) []byte {
|
|
// hi0 := highbits(rowID * ShardWidth)
|
|
// return badgerKey(f.index, f.field, f.view, f.shard, hi0)
|
|
//}
|
|
|
|
// StringifiedBadgerKeys returns a string with all the container
|
|
// keys available in badger.
|
|
func (w *BadgerDBWrapper) StringifiedBadgerKeys(optionalUseThisTx Tx) (r string) {
|
|
if optionalUseThisTx == nil {
|
|
tx := w.NewBadgerTx(!writable)
|
|
defer tx.Rollback()
|
|
r = stringifiedBadgerKeysTx(tx)
|
|
return
|
|
}
|
|
|
|
btx, ok := optionalUseThisTx.(*BadgerTx)
|
|
if !ok {
|
|
return fmt.Sprintf("<not-a-BadgerTx-in-StringifiedBadgerKeys-was-%T>", optionalUseThisTx)
|
|
}
|
|
r = stringifiedBadgerKeysTx(btx)
|
|
return
|
|
}
|
|
|
|
// countBitsSet returns the number of bits set (or "hot") in
|
|
// the roaring container value found by the badgerKey()
|
|
// formatted bkey.
|
|
func (tx *BadgerTx) countBitsSet(bkey []byte) (n int) {
|
|
|
|
item, err := tx.tx.Get(bkey)
|
|
if err == badger.ErrKeyNotFound {
|
|
panic(fmt.Sprintf("badger did not have value for bkey = '%v'", string(bkey)))
|
|
}
|
|
panicOn(err)
|
|
|
|
var rc *roaring.Container
|
|
err = item.Value(func(v []byte) error {
|
|
// This func with val would only be called if item.Value encounters no error
|
|
rc = tx.toContainer(item.UserMeta(), v)
|
|
return nil
|
|
})
|
|
panicOn(err)
|
|
|
|
n = int(rc.N())
|
|
return
|
|
}
|
|
|
|
// stringifiedBadgerKeysTx reports all the badger keys and a
|
|
// corresponding blake3 hash viewable by txn within the entire
|
|
// badger database.
|
|
// It also reports how many bits are hot in the roaring container
|
|
// (how many bits are set, or 1 rather than 0).
|
|
//
|
|
// By convention, we must return the empty string if there
|
|
// are no keys present. The tests use this to confirm
|
|
// an empty database.
|
|
func stringifiedBadgerKeysTx(tx *BadgerTx) (r string) {
|
|
|
|
r = "allkeys:[\n"
|
|
it := tx.tx.NewIterator(badger.DefaultIteratorOptions)
|
|
defer it.Close()
|
|
any := false
|
|
for it.Rewind(); it.Valid(); it.Next() {
|
|
any = true
|
|
item := it.Item()
|
|
bkey := item.Key()
|
|
key := string(bkey)
|
|
ckey := badgerKeyExtractContainerKey(bkey)
|
|
hash := ""
|
|
srbm := ""
|
|
err := item.Value(func(val []byte) error {
|
|
hash = blake3sum16(val)
|
|
ct := tx.toContainer(item.UserMeta(), val)
|
|
cts := roaring.NewSliceContainers()
|
|
cts.Put(ckey, ct)
|
|
rbm := &roaring.Bitmap{Containers: cts}
|
|
srbm = bitmapAsString(rbm)
|
|
return nil
|
|
})
|
|
panicOn(err)
|
|
r += fmt.Sprintf("%v -> %v (%v hot)\n", key, hash, tx.countBitsSet(bkey))
|
|
r += " ......." + srbm + "\n"
|
|
}
|
|
r += "]\n all-in-blake3:" + blake3sum16([]byte(r))
|
|
|
|
if !any {
|
|
return ""
|
|
}
|
|
return "badger-" + r
|
|
}
|
|
|
|
func sliceToMap(slc []uint64) (m map[uint64]bool) {
|
|
m = make(map[uint64]bool)
|
|
for _, v := range slc {
|
|
m[v] = true
|
|
}
|
|
return
|
|
}
|
|
|
|
// return A - B
|
|
func mapDiff(mapA, mapB map[uint64]bool) (r []int) {
|
|
for a := range mapA {
|
|
_, ok := mapB[a]
|
|
if !ok {
|
|
r = append(r, int(a))
|
|
}
|
|
}
|
|
return
|
|
}
|
|
|
|
func asInts(a []uint64) (r []int) {
|
|
r = make([]int, len(a))
|
|
for i, v := range a {
|
|
r[i] = int(v)
|
|
}
|
|
return
|
|
}
|
|
|
|
func bitmapAsString(rbm *roaring.Bitmap) (r string) {
|
|
r = "c("
|
|
slc := rbm.Slice()
|
|
width := 0
|
|
s := ""
|
|
for _, v := range slc {
|
|
if width == 0 {
|
|
s = fmt.Sprintf("%v", v)
|
|
} else {
|
|
s = fmt.Sprintf(", %v", v)
|
|
}
|
|
width += len(s)
|
|
r += s
|
|
if width > 70 {
|
|
r += ",\n"
|
|
width = 0
|
|
}
|
|
}
|
|
if width == 0 && len(r) > 2 {
|
|
r = r[:len(r)-2]
|
|
}
|
|
return r + ")"
|
|
}
|
|
|
|
func containerAsString(ckey uint64, rc *roaring.Container) (r string) {
|
|
rbm := roaring.NewBitmap()
|
|
rbm.Containers.Put(ckey, rc)
|
|
return bitmapAsString(rbm)
|
|
}
|
|
|
|
var _ = containerAsString // happy linter
|
|
|
|
func roaringBitmapDiff(a, b *roaring.Bitmap) error {
|
|
nA := a.Count()
|
|
nB := b.Count()
|
|
|
|
slcA := a.Slice()
|
|
slcB := b.Slice()
|
|
|
|
mapA := sliceToMap(slcA)
|
|
mapB := sliceToMap(slcB)
|
|
|
|
AminusB := mapDiff(mapA, mapB)
|
|
BminusA := mapDiff(mapB, mapA)
|
|
|
|
sort.Ints(AminusB)
|
|
sort.Ints(BminusA)
|
|
|
|
res := fmt.Sprintf("nA = %v; nB = %v;\n", nA, nB)
|
|
ndiff := 0
|
|
if nA != nB {
|
|
ndiff++
|
|
}
|
|
|
|
if len(AminusB) > 0 {
|
|
res += fmt.Sprintf("==> AminusB = (len %v) '%#v'; ", len(AminusB), AminusB)
|
|
ndiff++
|
|
}
|
|
if len(BminusA) > 0 {
|
|
res += fmt.Sprintf("\n==> BminusA = (len %v) '%#v'; ", len(BminusA), BminusA)
|
|
ndiff++
|
|
}
|
|
if ndiff == 0 {
|
|
return nil
|
|
}
|
|
res += fmt.Sprintf("\n ==> A = '%#v'\n ==> B = '%#v'", asInts(slcA), asInts(slcB))
|
|
return errors.New(res)
|
|
}
|
|
|
|
func dirAsString(path string) (r string) {
|
|
r = fmt.Sprintf("dump of directory '%v':\n", path)
|
|
files, err := ioutil.ReadDir(path)
|
|
panicOn(err)
|
|
for _, f := range files {
|
|
r += f.Name() + "\n"
|
|
}
|
|
return r
|
|
}
|
|
|
|
var _ = dirAsString // happy linter
|