diff --git a/Makefile b/Makefile index e1577b62b..b8454402f 100644 --- a/Makefile +++ b/Makefile @@ -16,16 +16,13 @@ RELEASE_ENABLED = $(subst 0,,$(RELEASE)) NOCHECKPTR=$(shell go version | grep -q 'go1.1[4,5,6,7]' && echo \"-gcflags=all=-d=checkptr=0\" ) BUILD_TAGS += $(if $(RELEASE_ENABLED),release) BUILD_TAGS += shardwidth$(SHARD_WIDTH) -BUILD_TAGS += $(foreach p,$(PLUGINS),plugin$(p)) define LICENSE_HASH_CODE head -13 $1 | sed -e 's/Copyright 20[0-9][0-9]/Copyright 20XX/g' | shasum | cut -f 1 -d " " endef LICENSE_HASH=$(shell $(call LICENSE_HASH_CODE, pilosa.go)) -PLUGINS=distinct export GO111MODULE=on export GOPRIVATE=github.com/molecula -export PLUGINS # Run tests and compile Pilosa default: test build @@ -97,7 +94,7 @@ clustertests: vendor # Like clustertests, but rebuilds all images. clustertests-build: vendor - docker-compose -f $(DOCKER_COMPOSE) down + docker-compose -f $(DOCKER_COMPOSE) down -v docker-compose -f $(DOCKER_COMPOSE) up --exit-code-from=client1 --build # Create prerelease builds @@ -152,25 +149,80 @@ docker-test: # run top tests, not subdirs. print summary red/green after. # The \-\-\- FAIL avoids counting the extra two FAIL strings at then bottom of log.topt. topt: + mv log.topt.roar log.topt.roar.prev || true go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.topt.roar @echo " log.topt.roar green: \c"; cat log.topt.roar | grep PASS |wc -l @echo " log.topt.roar red: \c"; cat log.topt.roar | grep '\-\-\- FAIL' |wc -l topt-badger: + mv log.topt.badger log.topt.badger.prev || true PILOSA_TXSRC=badger go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.topt.badger @echo " log.topt.badger green: \c"; cat log.topt.badger | grep PASS |wc -l @echo " log.topt.badger red: \c"; cat log.topt.badger | grep '\-\-\- FAIL' |wc -l +topt-rb: + mv log.topt.roaring_badger log.topt.roaring_badger.prev || true + PILOSA_TXSRC=roaring_badger go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.topt.badger + @echo " log.topt.roaring_badger green: \c"; cat log.topt.roaring_badger | grep PASS |wc -l + @echo " log.topt.roaring_badger red: \c"; cat log.topt.roaring_badger | grep '\-\-\- FAIL' |wc -l + +topt-badger-race: + mv log.topt.badger-race log.topt.badger-race.prev || true + PILOSA_TXSRC=badger go test -race -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.topt.badger-race + @echo " log.topt.badger-race green: \c"; cat log.topt.badger-race | grep PASS |wc -l + @echo " log.topt.badger-race red: \c"; cat log.topt.badger-race | grep '\-\-\- FAIL' |wc -l + topt-rbf: + mv log.topt.rbf log.topt.rbf.prev || true PILOSA_TXSRC=rbf go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.topt.rbf @echo " log.topt.rbf green: \c"; cat log.topt.rbf | grep PASS |wc -l @echo " log.topt.rbf red: \c"; cat log.topt.rbf | grep '\-\-\- FAIL' |wc -l topt-race: + mv log.topt.race log.topt.race.prev || true go test -race -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.topt.race @echo " log.topt.race green: \c"; cat log.topt.race | grep PASS |wc -l @echo " log.topt.race red: \c"; cat log.topt.race | grep '\-\-\- FAIL' |wc -l +# blue-green checks. These run two different storage engines (rbf, roaring, or badger) +# and compare each transaction for a result. +bg-br: + mv log.bg.bg_roar log.bg.bg_roar.prev || true + PILOSA_TXSRC=badger_roaring go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.bg.bg_roar + @echo " log.bg.bg_roar green: \c"; cat log.bg.bg_roar | grep PASS |wc -l + @echo " log.bg.bg_roar red: \c"; cat log.bg.bg_roar | grep '\-\-\- FAIL' |wc -l + +bg-rb: + mv log.bg.roar_bg log.bg.roar_bg.prev || true + PILOSA_TXSRC=roaring_badger go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.bg.roar_bg + @echo " log.bg.roar_bg green: \c"; cat log.bg.roar_bg | grep PASS |wc -l + @echo " log.bg.roar_bg red: \c"; cat log.bg.roar_bg | grep '\-\-\- FAIL' |wc -l + +bg-fr: + mv log.bg.rbf_roar log.bg.rbf_roar.prev || true + PILOSA_TXSRC=rbf_roaring go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.bg.rbf_roar + @echo " log.bg.rbf_roar green: \c"; cat log.bg.rbf_roar | grep PASS |wc -l + @echo " log.bg.rbf_roar red: \c"; cat log.bg.rbf_roar | grep '\-\-\- FAIL' |wc -l + +bg-rf: + mv log.bg.roar_rbf log.bg.roar_rbf.prev || true + PILOSA_TXSRC=roaring_rbf go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.bg.roar_rbf + @echo " log.bg.roar_rbf green: \c"; cat log.bg.roar_rbf | grep PASS |wc -l + @echo " log.bg.roar_rbf red: \c"; cat log.bg.roar_rbf | grep '\-\-\- FAIL' |wc -l + +bg-fb: + mv log.bg.rbf_badger log.bg.rbf_badger.prev || true + PILOSA_TXSRC=rbf_badger go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.bg.rbf_badger + @echo " log.bg.rbf_badger green: \c"; cat log.bg.rbf_badger | grep PASS |wc -l + @echo " log.bg.rbf_badger red: \c"; cat log.bg.rbf_badger | grep '\-\-\- FAIL' |wc -l + +bg-bf: + mv log.bg.badger_rbf log.bg.badger_rbf.prev || true + PILOSA_TXSRC=badger_rbf go test -v -tags='$(BUILD_TAGS)' $(TESTFLAGS) $(NOCHECKPTR) 2>&1 | tee log.bg.badger_rbf + @echo " log.bg.badger_rbf green: \c"; cat log.bg.badger_rbf | grep PASS |wc -l + @echo " log.bg.badger_rbf red: \c"; cat log.bg.badger_rbf | grep '\-\-\- FAIL' |wc -l + + # Run golangci-lint golangci-lint: require-golangci-lint golangci-lint run --skip-files '.*\.peg\.go' diff --git a/api.go b/api.go index b5bc89422..3ed7358e3 100644 --- a/api.go +++ b/api.go @@ -371,16 +371,9 @@ func importWorker(importWork chan importJob) { var doClear bool switch doAction { case RequestActionOverwrite: - // TODO(jea): the question here is, why are we commiting this separately from j.tx? - // why doesn't j.tx suffice? It doesn't but why/which is correct? - tx := j.field.holder.indexes[j.field.index].Txf.NewTx(Txo{Write: true, Field: j.field}) - defer tx.Rollback() - if err := j.field.importRoaringOverwrite(j.ctx, tx, viewData, j.shard, viewName, j.req.Block); err != nil { + if err := j.field.importRoaringOverwrite(j.ctx, j.tx, viewData, j.shard, viewName, j.req.Block); err != nil { return errors.Wrap(err, "importing roaring as overwrite") } - if err := tx.Commit(); err != nil { - return errors.Wrap(err, "commit of importing roaring as overwrite") - } case RequestActionClear: doClear = true fallthrough diff --git a/badger.go b/badger.go index 65052b525..aee8d9897 100644 --- a/badger.go +++ b/badger.go @@ -15,8 +15,9 @@ package pilosa import ( - "errors" + "bytes" "fmt" + "io" "io/ioutil" "log" "os" @@ -29,7 +30,9 @@ import ( "unsafe" badger "github.com/dgraph-io/badger/v2" + badgeroptions "github.com/dgraph-io/badger/v2/options" "github.com/pilosa/pilosa/v2/roaring" + "github.com/pkg/errors" ) // TODO: is there a more optimal time to do badger garbage collection? @@ -189,15 +192,70 @@ func (l *BadgerLog) Debugf(f string, v ...interface{}) { l.Printf("DEBUG: "+f, v...) } +// badgerRegistrar facilitates shutdown +// of all the badger databases started under +// tests. Its needed because most tests don't cleanup +// the *Index(es) they create. But we still +// want to shutdown badgerDB goroutines +// after tests run. +// +// It also allows opening the same path twice to +// result in sharing the same open database handle, and +// thus the same transactional guarantees. +// +type badgerRegistrar struct { + mu sync.Mutex + mp map[*BadgerDBWrapper]bool + + path2db map[string]*BadgerDBWrapper +} + +var globalBadgerReg *badgerRegistrar = newBadgerTestRegistrar() + +func newBadgerTestRegistrar() *badgerRegistrar { + return &badgerRegistrar{ + mp: make(map[*BadgerDBWrapper]bool), + path2db: make(map[string]*BadgerDBWrapper), + } +} + +// register each badger created under tests, so we +// can clean them up. This is called by openBadgerDBWrapper() while +// holding the r.mu.Lock, since it needs to atomically +// check the registry and make a new instance only +// if one does not exist for its path, and otherwise +// return the existing instance. +func (r *badgerRegistrar) unprotectedRegister(w *BadgerDBWrapper) { + r.mp[w] = true + r.path2db[w.path] = w +} + +// unregister removes w from r +func (r *badgerRegistrar) unregister(w *BadgerDBWrapper) { + r.mu.Lock() + delete(r.mp, w) + delete(r.path2db, w.path) + r.mu.Unlock() +} + +func DumpAllBadger() { + globalBadgerReg.mu.Lock() + defer globalBadgerReg.mu.Unlock() + for w := range globalBadgerReg.mp { + _ = w + AlwaysPrintf("this badger path='%v' has: \n%v\n", w.path, w.StringifiedBadgerKeys(nil)) + } +} + // newBadgerDBWrapper creates a new empty database, blowing away // any prior path + "-badgerdb" directory. -func newBadgerDBWrapper(path string) (*BadgerDBWrapper, error) { +func (r *badgerRegistrar) newBadgerDBWrapper(path string) (*BadgerDBWrapper, error) { bpath := badgerPath(path) err := os.RemoveAll(bpath) if err != nil { return nil, err } - return openBadgerDBWrapper(bpath) + return r.openBadgerDBWrapper(bpath) } // badgerPath is a helper for determining the full directory @@ -212,7 +270,12 @@ func badgerPath(path string) string { // openBadgerDB opens the database in the bpath directoy // without deleting any prior content. Any BadgerDB // database directory will have the "-badgerdb" suffix. -func openBadgerDBWrapper(bpath string) (*BadgerDBWrapper, error) { +// +// openBadgerDB will check the registry and make a new instance only +// if one does not exist for its bpath. Otherwise it returns +// the existing instance. This insures only one badgerDB +// per bpath in this pilosa node. +func (r *badgerRegistrar) openBadgerDBWrapper(bpath string) (*BadgerDBWrapper, error) { // now that newTxFactory can call us directly, we might not // have the -badgerdb suffix. @@ -220,9 +283,33 @@ func openBadgerDBWrapper(bpath string) (*BadgerDBWrapper, error) { bpath += "-badgerdb" } + r.mu.Lock() + defer r.mu.Unlock() + w, ok := r.path2db[bpath] + if ok { + // creates the effect of having only one badger open per pilosa node. + return w, nil + } + // otherwise, make a new badger and store it in globalBadgerReg + // regular: works on amd64, but 386 doesn't work. opt := badger.DefaultOptions(bpath).WithLogger(badgerDefaultLogger) + opt.Compression = badgeroptions.None // turn off compression. + opt.ZSTDCompressionLevel = 0 // really, just in case. + + // MaxCacheSize docs: + // + // how much data cache should hold in memory. A small size of + // cache means lower memory consumption and lookups/iterations + // would take longer. It is recommended to use a cache if you're + // using compression or encryption. If compression and + // encryption both are disabled, adding a cache will lead to + // unnecessary overhead which will affect the read performance. + // Setting size to zero disables the cache altogether. + opt.MaxCacheSize = 0 + opt.LoadBloomsOnOpen = false // should speed up start-up time. + // to get memory only do: //opt := badger.DefaultOptions("").WithLogger(badgerDefaultLogger).WithInMemory(true) @@ -231,12 +318,16 @@ func openBadgerDBWrapper(bpath string) (*BadgerDBWrapper, error) { return nil, err } halt := make(chan bool) - w := &BadgerDBWrapper{ + w = &BadgerDBWrapper{ + reg: r, path: bpath, db: db, halt: halt, hasher: NewBlake3Hasher(), } + r.unprotectedRegister(w) + + w.startStack = stack() w.startBadgerGarbageCollectionBackgroundGoro() return w, nil } @@ -251,77 +342,8 @@ func (w *BadgerDBWrapper) DeleteIndex(indexName string) error { if strings.Contains(indexName, "'") { return fmt.Errorf("error: bad indexName `%v` in BadgerDBWrapper.DeleteIndex() call: indexName cannot contain apostrophes/single quotes.", indexName) } - w.muDb.Lock() - defer w.muDb.Unlock() - - // a) do key-ony iteration, no value fetch; - // - // b) do deletes in large batches, to avoid alot of txn overhead; - // per recommendation https://github.com/dgraph-io/badger/issues/598 - // - // c) we do not, at present, try to maintain one large - // transaction with all the keys in a index in it. Because - // there can be too many keys. Hence the index will disappear - // in chucks of 100K keys, not atomically-all-at-once. - prefix := badgerIndexOnlyPrefix(indexName) - - noMoreKeysWithPrefix := false - const maxDeletesPerTxn = 100000 - - for !noMoreKeysWithPrefix { - err := w.db.Update(func(txn *badger.Txn) error { - o := badger.DefaultIteratorOptions - o.AllVersions = false - o.PrefetchValues = false // key-only iteration, no values. - - // note: panic: Unclosed iterator at time of Txn.Discard ? panic on segfault here? - // This means we messed up and Closed() the Database already; too early. - it := txn.NewIterator(o) - - defer it.Close() - n := 0 - goners := make([][]byte, 0, maxDeletesPerTxn) - for it.Seek(prefix); it.ValidForPrefix(prefix); it.Next() { - - // KeyCopy() is required; Key() means corruption and possible segfault. - key := it.Item().KeyCopy(nil) - goners = append(goners, key) - n++ - if n >= maxDeletesPerTxn { - break - } - } - if !it.ValidForPrefix(prefix) { - noMoreKeysWithPrefix = true // done with the full delete of up to maxDeletesPerTxn - } - for _, key := range goners { - if err := txn.Delete(key); err != nil { - return err - } - } - return nil // auto-commit happens - }) - // err back from Update can be ErrConflict in case of - // a conflict. Badger docs: "Depending on the state - // of your application, you have the option to - // retry the operation if you receive this error." - panicOn(err) - - } // end for: proceed to next bath of 100K keys - - // Finally, run a garbage collection to delete values from the value log. - // - // "Only one GC is allowed at a time. If another value log GC - // is running, or DB has been closed, this would return an ErrRejected." - // -- https://godoc.org/github.com/dgraph-io/badger#DB.RunValueLogGC - // Still, we don't see a mutex inside the RunValueLogGC code, so - // lock muGC just to be sure. - w.muGC.Lock() - defer w.muGC.Unlock() - _ = w.db.RunValueLogGC(0.5) - - return nil + return w.DeletePrefix(prefix) } // startBadgerGarbageCollectionBackgroundGoro handles Badger DB @@ -367,6 +389,9 @@ type BadgerDBWrapper struct { path string db *badger.DB + // track our registrar for Close / goro leak reporting purposes. + reg *badgerRegistrar + // openTx and openIt are BadgerDBWrapper scoped tables of all open // transactions and iterators. These are primarily for debugging purposes. // openTx and openIt should only be read/written after locking the muOpenTxIt mutex. @@ -404,6 +429,10 @@ type BadgerDBWrapper struct { // safety because otherwise TestAPI_ImportColumnAttrs sees // corrupted data. doAllocZero bool + + // stack() from our creation point, to track tests + // that haven't closed us. + startStack string } // unprotectedListOpenTxAsString is a debugging helper. @@ -437,24 +466,30 @@ func (w *BadgerDBWrapper) UnprotectedListOpenItAsString() (r string) { // Read-only queries should set write to false, to allow more concurrency. // Methods on a BadgerTx are thread-safe, and can be called from // different goroutines. -func (w *BadgerDBWrapper) NewBadgerTx(write bool) (tx *BadgerTx) { +// +// initialIndexName is optional. It is set by the TxFactory from the Txo +// options provided at the Tx creation point. It allows us to recognize +// and isolate cross-index queries more quickly. It can always be empty "" +// but when set is highly useful for debugging. It has no impact +// on transaction behavior. +// +func (w *BadgerDBWrapper) NewBadgerTx(write bool, initialIndexName string) (tx *BadgerTx) { w.muDb.Lock() defer w.muDb.Unlock() tx = &BadgerTx{ - write: write, - tx: w.db.NewTransaction(write), - Db: w, - initloc: stack(), - doAllocZero: w.doAllocZero, + write: write, + tx: w.db.NewTransaction(write), + Db: w, + initloc: stack(), + doAllocZero: w.doAllocZero, + initialIndexName: initialIndexName, } - //vv("NewBadgerTx(write=%v) top, p=%p", write, tx) - //pp("NewBadgerTx(write=%v) top, p=%p, stack=\n\n'%v'", write, tx, stack()) if w.openTx == nil { w.openTx = make(map[*BadgerTx]bool) } - //pp("NewBadgerTx(write=%v); p=%p; (currently open txn: '%v', its: '%v'). initloc:'%v'", write, tx, w.unprotectedListOpenTxAsString(), w.UnprotectedListOpenItAsString(), tx.initloc) + w.muOpenTxIt.Lock() w.openTx[tx] = write w.muOpenTxIt.Unlock() @@ -466,6 +501,7 @@ func (w *BadgerDBWrapper) Close() (err error) { w.muDb.Lock() defer w.muDb.Unlock() if !w.closed { + w.reg.unregister(w) close(w.halt) w.closed = true } @@ -501,8 +537,15 @@ type BadgerTx struct { // for tracking txn boundary issues, track all the memory // that we deploy for roaring containers, and zero it on // transaction commit/rollback. + acMu sync.Mutex // protect ourAllocs and ourContainers ourAllocs [][]byte ourContainers []*roaring.Container + + initialIndexName string +} + +func (tx *BadgerTx) Type() string { + return BadgerTxn } func (tx *BadgerTx) UseRowCache() bool { @@ -521,6 +564,8 @@ func (tx *BadgerTx) UseRowCache() bool { // to transaction commit. func (tx *BadgerTx) overWriteOurAllocs() { + tx.acMu.Lock() + defer tx.acMu.Unlock() for _, s := range tx.ourAllocs { // The Go compiler recognizes the following pattern and inserts @@ -529,6 +574,10 @@ func (tx *BadgerTx) overWriteOurAllocs() { // and https://codereview.appspot.com/137880043 for i := range s { s[i] = 0 + // or + // Seebs suggested we might see even more crashes :) + // but since it will be slow (no memclr), we'll leave the default 0 for now. + //s[i] = -2 } } // keep this around if we need to activate out-of-mmap memory access again. @@ -632,21 +681,57 @@ func badgerKey(index, field, view string, shard uint64, roaringContainerKey uint prefix := badgerPrefix(index, field, view, shard) ckey := []byte(fmt.Sprintf("%020d", roaringContainerKey)) - return append(prefix, ckey...) + bkey := append(prefix, ckey...) + MustValidateKey(bkey) + return bkey +} + +var ckeyPartExpected = []byte(";ckey@") + +// MustValidatekey will panic on a bad badgerKey with an informative message. +func MustValidateKey(bkey []byte) { + n := len(bkey) + if n < 56 { + panic(fmt.Sprintf("bkey too short min size is 56 but we see %v in '%v'", n, string(bkey))) + } + beforeCkey := bkey[n-26 : n-20] + if !bytes.Equal(beforeCkey, ckeyPartExpected) { + panic(fmt.Sprintf(`bkey did not have expected ";ckey@" at 26 bytes from the end of the bkey '%v'; instead had '%v'`, string(bkey), string(beforeCkey))) + } +} + +func shardFromBadgerKey(bkey []byte) (shard uint64) { + MustValidateKey(bkey) + + n := len(bkey) + // idx:'i';fld:'f';vw:'standard';shd:'1';ckey@18446744073709551615 -> idx:'i';fld:'f';vw:'standard';shd:'1 + by := bkey[:n-27] + beg := bytes.LastIndex(by, []byte("'")) + if beg == -1 { + panic(fmt.Sprintf("bad bkey='%v' did not have single quote to being shard decoding", string(bkey))) + } + parseMe := string(by[beg+1:]) + shard, err := strconv.ParseUint(parseMe, 10, 64) + if err != nil { + panic(fmt.Sprintf("could not parse parseMe '%v' in strconv.ParseUint(), error: '%v'", parseMe, err)) + } + return shard } // badgerKeyAndPrefix returns the equivalent of badgerKey() and badgerPrefix() calls. func badgerKeyAndPrefix(index, field, view string, shard uint64, roaringContainerKey uint64) (key, prefix []byte) { prefix = badgerPrefix(index, field, view, shard) ckey := []byte(fmt.Sprintf("%020d", roaringContainerKey)) - return append(prefix, ckey...), prefix + bkey := append(prefix, ckey...) + MustValidateKey(bkey) + return bkey, prefix } var _ = badgerKeyAndPrefix // keep linter happy // badgerKeyExtractContainerKey extracts the containerKey from bkey. func badgerKeyExtractContainerKey(bkey []byte) (containerKey uint64) { - + MustValidateKey(bkey) // The zero padding means that the container-key is always the last 20 bytes of the bkey. // // Be sure to catch the problematic case of a user passing in only a prefix. A prefix @@ -665,11 +750,15 @@ func badgerKeyExtractContainerKey(bkey []byte) (containerKey uint64) { return } +func badgerAllShardPrefix(index, field, view string) []byte { + return []byte(fmt.Sprintf("idx:'%v';fld:'%v';vw:'%v';shd:", index, field, view)) +} + // badgerPrefix returns everything from badgerKey up to and // including the '@' fune in a badger key. The prefix excludes the roaring container key itself. // NB must be kept in sync with badgerKey() and badgerKeyExtractContainerKey(). func badgerPrefix(index, field, view string, shard uint64) []byte { - return []byte(fmt.Sprintf("idx:'%v';fld:'%v';vw:'%v';shd:'%x';ckey@", index, field, view, shard)) + return []byte(fmt.Sprintf("idx:'%v';fld:'%v';vw:'%v';shd:'%020v';ckey@", index, field, view, shard)) } // badgerIndexOnlyPrefix returns a prefix suitable for DeleteIndex and a key-scan to @@ -869,6 +958,38 @@ func (tx *BadgerTx) Contains(index, field, view string, shard uint64, key uint64 return exists, err } +func (tx *BadgerTx) SliceOfShards(index, field, view, optionalViewPath string) (sliceOfShards []uint64, err error) { + + prefix := badgerAllShardPrefix(index, field, view) + + bi := NewBadgerIterator(tx, prefix) + defer bi.Close() + bi.Seek(prefix) + if !bi.it.Valid() { + return + } + lastShard := uint64(0) + firstDone := false + for bi.Next() { + item := bi.it.Item() + key := item.Key() + shard := shardFromBadgerKey(key) + if firstDone { + if shard != lastShard { + sliceOfShards = append(sliceOfShards, shard) + } + lastShard = shard + } else { + // first time + lastShard = shard + firstDone = true + sliceOfShards = append(sliceOfShards, shard) + } + + } + return +} + // key is the container key for the first roaring Container // roaring docs: Iterator returns a ContainterIterator which *after* a call to Next(), a call to Value() will // return the first container at or after key. found will be true if a @@ -877,10 +998,10 @@ func (tx *BadgerTx) Contains(index, field, view string, shard uint64, key uint64 // BadgerTx notes: We auto-stop at the end of this shard, not going beyond. func (tx *BadgerTx) ContainerIterator(index, field, view string, shard uint64, firstRoaringContainerKey uint64) (citer roaring.ContainerIterator, found bool, err error) { - // needle example: "index:'i';field:'f';view:'v';shard:'0';key@00000000000000000000" + // needle example: "idx:'i';fld:'f';vw:'v';shd:'00000000000000000000';key@00000000000000000000" needle := badgerKey(index, field, view, shard, firstRoaringContainerKey) - // prefix example: "index:'i';field:'f';view:'v';shard:'0';key@" + // prefix example: "idx:'i';fld:'f';vw:'v';shard:'00000000000000000000';key@" prefix := badgerPrefix(index, field, view, shard) bi := NewBadgerIterator(tx, prefix) @@ -917,12 +1038,6 @@ func NewBadgerIterator(tx *BadgerTx, prefix []byte) (bi *BadgerIterator) { tx.Db.muOpenTxIt.Lock() defer tx.Db.muOpenTxIt.Unlock() - defer func() { - r := recover() - if r != nil { - panic(r) - } - }() opts := badger.DefaultIteratorOptions opts.PrefetchValues = false // else by default, pre-fetches the 1st 100 values, which would be slow. opts.Reverse = false @@ -1125,6 +1240,7 @@ func (tx *BadgerTx) Count(index, field, view string, shard uint64) (uint64, erro } // Max is the maximum bit-value in your bitmap. +// Returns zero if the bitmap is empty. Odd, but this is what roaring.Max does. func (tx *BadgerTx) Max(index, field, view string, shard uint64) (uint64, error) { prefix := badgerPrefix(index, field, view, shard) @@ -1133,7 +1249,10 @@ func (tx *BadgerTx) Max(index, field, view string, shard uint64) (uint64, error) it := NewBadgerReverseIterator(tx, prefix, seekto) // this iterator is still open, when we commit/discard tx. defer it.Close() - hb, rc := it.Value() + if !it.it.Valid() { + return 0, nil + } + hb, rc := it.Value() // getting it returns invalid, as in empty iterator lb := rc.Max() return hb<<16 | uint64(lb), nil @@ -1309,7 +1428,7 @@ func (tx *BadgerTx) IncrementOpN(index, field, view string, shard uint64, change // ImportRoaringBits handles deletes by setting clear=true. // rowSet[rowID] returns the number of bit changed on that rowID. -func (tx *BadgerTx) ImportRoaringBits(index, field, view string, shard uint64, itr roaring.RoaringIterator, clear bool, log bool, rowSize uint64) (changed int, rowSet map[uint64]int, err error) { +func (tx *BadgerTx) ImportRoaringBits(index, field, view string, shard uint64, itr roaring.RoaringIterator, clear bool, log bool, rowSize uint64, data []byte) (changed int, rowSet map[uint64]int, err error) { n := itr.Len() if n == 0 { return @@ -1456,11 +1575,19 @@ func (tx *BadgerTx) toContainer(typ byte, v []byte) (r *roaring.Container) { // TODO: performance tuning might want w := v here, if we can guarantee no access to memory past the Tx lifetime. // // Problem is, at least some tests appear to not respect transaction boundaries... + // + // Seebs suggested this nice variation: we could use individual mmaps for these + // copies, which would be unusable in production, but workable for testing, and then unmap them, + // which would get us probable segfaults on future accesses to them. + // w := make([]byte, len(v)) - copy(w, v) // green go test -v -run TestAPI_ImportColumnAttrs + copy(w, v) + // the copy above makes green: // green go test -v -run TestAPI_ImportColumnAttrs //w := v // if instead of append we use v directly, it causes red: go test -v -run TestAPI_ImportColumnAttrs // register w so we can catch out-of-tx memory access + tx.acMu.Lock() + defer tx.acMu.Unlock() tx.ourAllocs = append(tx.ourAllocs, w) switch typ { @@ -1511,7 +1638,7 @@ func fromInterval16(a []roaring.Interval16) []byte { // keys available in badger. func (w *BadgerDBWrapper) StringifiedBadgerKeys(optionalUseThisTx Tx) (r string) { if optionalUseThisTx == nil { - tx := w.NewBadgerTx(!writable) + tx := w.NewBadgerTx(!writable, "") defer tx.Rollback() r = stringifiedBadgerKeysTx(tx) return @@ -1699,3 +1826,94 @@ func dirAsString(path string) (r string) { } var _ = dirAsString // happy linter + +func (w *BadgerDBWrapper) DeleteFragment(index, field, view string, shard uint64, frag interface{}) error { + prefix := badgerPrefix(index, field, view, shard) + return w.DeletePrefix(prefix) +} + +func (w *BadgerDBWrapper) DeletePrefix(prefix []byte) error { + w.muDb.Lock() + defer w.muDb.Unlock() + + // a) do key-ony iteration, no value fetch; + // + // b) do deletes in large batches, to avoid alot of txn overhead; + // per recommendation https://github.com/dgraph-io/badger/issues/598 + // + // c) we do not, at present, try to maintain one large + // transaction with all the keys in a index in it. Because + // there can be too many keys. Hence the index will disappear + // in chucks of 100K keys, not atomically-all-at-once. + + noMoreKeysWithPrefix := false + const maxDeletesPerTxn = 100000 + + for !noMoreKeysWithPrefix { + err := w.db.Update(func(txn *badger.Txn) error { + o := badger.DefaultIteratorOptions + o.AllVersions = false + o.PrefetchValues = false // key-only iteration, no values. + + // note: panic: Unclosed iterator at time of Txn.Discard ? panic on segfault here? + // This means we messed up and Closed() the Database already; too early. + it := txn.NewIterator(o) + + defer it.Close() + n := 0 + goners := make([][]byte, 0, maxDeletesPerTxn) + for it.Seek(prefix); it.ValidForPrefix(prefix); it.Next() { + + // KeyCopy() is required; Key() means corruption and possible segfault. + key := it.Item().KeyCopy(nil) + goners = append(goners, key) + n++ + if n >= maxDeletesPerTxn { + break + } + } + if !it.ValidForPrefix(prefix) { + noMoreKeysWithPrefix = true // done with the full delete of up to maxDeletesPerTxn + } + for _, key := range goners { + if err := txn.Delete(key); err != nil { + return err + } + } + return nil // auto-commit happens + }) + // err back from Update can be ErrConflict in case of + // a conflict. Badger docs: "Depending on the state + // of your application, you have the option to + // retry the operation if you receive this error." + panicOn(err) + + } // end for: proceed to next bath of 100K keys + + // Finally, run a garbage collection to delete values from the value log. + // + // "Only one GC is allowed at a time. If another value log GC + // is running, or DB has been closed, this would return an ErrRejected." + // -- https://godoc.org/github.com/dgraph-io/badger#DB.RunValueLogGC + // Still, we don't see a mutex inside the RunValueLogGC code, so + // lock muGC just to be sure. + w.muGC.Lock() + defer w.muGC.Unlock() + _ = w.db.RunValueLogGC(0.5) + + return nil +} + +func (tx *BadgerTx) RoaringBitmapReader(index, field, view string, shard uint64, fragmentPathForRoaring string) (r io.ReadCloser, sz int64, err error) { + + rbm, err := tx.RoaringBitmap(index, field, view, shard) + if err != nil { + return nil, -1, errors.Wrap(err, "RoaringBitmapReader RoaringBitmap") + } + var buf bytes.Buffer + sz, err = rbm.WriteTo(&buf) + if err != nil { + return nil, -1, errors.Wrap(err, "RoaringBitmapReader rbm.WriteTo(buf)") + } + return ioutil.NopCloser(&buf), sz, err +} diff --git a/badger_test.go b/badger_test.go index 4f2f09f52..10fc0baa0 100644 --- a/badger_test.go +++ b/badger_test.go @@ -12,12 +12,18 @@ // See the License for the specific language governing permissions and // limitations under the License. +// explanation of build tags: +// // badgerdb builds but won't run in 32-bit 386 world, as of 2020 July 20. // See https://github.com/dgraph-io/badger/issues/1384 for any progress. // What we see is that the value-log allocations immediately run out of // memory. So we turn off 386 with a build tag to keep the .circleci happy. +// +// gendebug_test will have a TestMain if build tag generationdebug is on, +// so we avoid conflicting with that debug scenario. // +build !386 +// +build !generationdebug package pilosa @@ -40,7 +46,7 @@ var _ = &roaring.Bitmap{} func badgerDBMustHaveBitvalue(dbwrap *BadgerDBWrapper, index, field, view string, shard uint64, bitvalue uint64) { - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, index) defer tx.Rollback() exists, err := tx.Contains(index, field, view, shard, bitvalue) panicOn(err) @@ -53,7 +59,7 @@ func badgerDBMustHaveBitvalue(dbwrap *BadgerDBWrapper, index, field, view string func badgerDBMustNotHaveBitvalue(dbwrap *BadgerDBWrapper, index, field, view string, shard uint64, bitvalue uint64) { - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, index) defer tx.Rollback() exists, err := tx.Contains(index, field, view, shard, bitvalue) panicOn(err) @@ -64,7 +70,7 @@ func badgerDBMustNotHaveBitvalue(dbwrap *BadgerDBWrapper, index, field, view str } func badgerDBMustSetBitvalue(dbwrap *BadgerDBWrapper, index, field, view string, shard uint64, putme uint64) { - tx := dbwrap.NewBadgerTx(writable) + tx := dbwrap.NewBadgerTx(writable, index) // add a bit changed, err := tx.Add(index, field, view, shard, doBatched, putme) @@ -82,14 +88,14 @@ func badgerDBMustSetBitvalue(dbwrap *BadgerDBWrapper, index, field, view string, } func badgerDBMustDeleteBitvalueContainer(dbwrap *BadgerDBWrapper, index, field, view string, shard uint64, putme uint64) { - tx := dbwrap.NewBadgerTx(writable) + tx := dbwrap.NewBadgerTx(writable, index) hi := highbits(putme) panicOn(tx.RemoveContainer(index, field, view, shard, hi)) panicOn(tx.Commit()) } func badgerDBMustDeleteBitvalue(dbwrap *BadgerDBWrapper, index, field, view string, shard uint64, putme uint64) { - tx := dbwrap.NewBadgerTx(writable) + tx := dbwrap.NewBadgerTx(writable, index) _, err := tx.Remove(index, field, view, shard, putme) panicOn(err) panicOn(tx.Commit()) @@ -99,7 +105,7 @@ func mustOpenEmptyBadgerWrapper(path string) (w *BadgerDBWrapper, cleaner func() var err error fn := badgerPath(path) panicOn(os.RemoveAll(fn)) - w, err = newBadgerDBWrapper(path) + w, err = globalBadgerReg.newBadgerDBWrapper(path) panicOn(err) // verify it is empty @@ -109,6 +115,7 @@ func mustOpenEmptyBadgerWrapper(path string) (w *BadgerDBWrapper, cleaner func() } return w, func() { + w.Close() // stop any started background GC goroutine. os.RemoveAll(fn) } } @@ -118,8 +125,8 @@ func TestBadger_SetBitmap(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_SetBitmap") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) bitvalue := uint64(0) changed, err := tx.Add(index, field, view, shard, doBatched, bitvalue) if changed <= 0 { @@ -140,7 +147,7 @@ func TestBadger_SetBitmap(t *testing.T) { // commited, so should be visible outside the txn // - tx2 := dbwrap.NewBadgerTx(!writable) + tx2 := dbwrap.NewBadgerTx(!writable, index) exists, err = tx2.Contains(index, field, view, shard, bitvalue) panicOn(err) if !exists { @@ -159,9 +166,9 @@ func TestBadger_OffsetRange(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_SetBitmap") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + bitvalue := uint64(1 << 20) changed, err := tx.Add(index, field, view, shard, doBatched, bitvalue) if changed <= 0 { @@ -194,7 +201,7 @@ func TestBadger_OffsetRange(t *testing.T) { start := uint64(0 << 16) endx := bitvalue + 1<<16 - tx2 := dbwrap.NewBadgerTx(!writable) + tx2 := dbwrap.NewBadgerTx(!writable, index) rbm2, err := tx2.OffsetRange(index, field, view, shard, offset, start, endx) panicOn(err) tx2.Rollback() @@ -208,7 +215,7 @@ func TestBadger_OffsetRange(t *testing.T) { // now offset by 2M offset = uint64(2 << 20) - tx3 := dbwrap.NewBadgerTx(!writable) + tx3 := dbwrap.NewBadgerTx(!writable, index) rbm3, err := tx3.OffsetRange(index, field, view, shard, offset, start, endx) panicOn(err) tx3.Rollback() @@ -236,7 +243,7 @@ func TestBadger_Count_on_many_containers(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) } - tx := dbwrap.NewBadgerTx(writable) + tx := dbwrap.NewBadgerTx(writable, index) defer tx.Rollback() n, err := tx.Count(index, field, view, shard) @@ -252,7 +259,7 @@ func TestBadger_Count_dense_containers(t *testing.T) { defer dbwrap.Close() index, field, view, shard := "i", "f", "v", uint64(0) - tx := dbwrap.NewBadgerTx(writable) + tx := dbwrap.NewBadgerTx(writable, index) expected := 0 // can't do more than about 100k writes per badger txn by default, so @@ -280,9 +287,9 @@ func TestBadger_ContainerIterator_on_empty(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_ContainerIterator") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(!writable) - defer tx.Rollback() index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(!writable, index) + defer tx.Rollback() bitvalue := uint64(0) citer, found, err := tx.ContainerIterator(index, field, view, shard, bitvalue) panicOn(err) @@ -298,9 +305,9 @@ func TestBadger_ContainerIterator_on_one_bit(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_ContainerIterator_on_one_bit") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - defer tx.Rollback() index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + defer tx.Rollback() bitvalue := uint64(42) @@ -398,9 +405,9 @@ func TestBadger_ContainerIterator_on_one_bit_fail_to_find(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_ContainerIterator_on_one_bit") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - defer tx.Rollback() index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + defer tx.Rollback() putme := uint64(1<<16) + 3 // in the key:1 container searchme := putme + 1 @@ -453,9 +460,9 @@ func TestBadger_ContainerIterator_empty_iteration_loop(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_ContainerIterator_empty_iteration_loop") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - defer tx.Rollback() index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + defer tx.Rollback() putme := uint64(1<<16) + 3 // in the key:1 container searchme := uint64(1 << 17) // in the next container, key:2 @@ -503,9 +510,9 @@ func TestBadger_ForEach_on_one_bit(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_ContainerIterator_on_one_bit") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - defer tx.Rollback() index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + defer tx.Rollback() bitvalue := uint64(42) @@ -563,7 +570,7 @@ func TestBadger_RemoveContainer_one_bit_test(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) // delete, but rollback instead of commit - tx := dbwrap.NewBadgerTx(writable) + tx := dbwrap.NewBadgerTx(writable, index) hi := highbits(putme) panicOn(tx.RemoveContainer(index, field, view, shard, hi)) tx.Rollback() @@ -572,7 +579,7 @@ func TestBadger_RemoveContainer_one_bit_test(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) // c) within one Tx, after delete it should be gone as viewed within the txn. - tx = dbwrap.NewBadgerTx(writable) + tx = dbwrap.NewBadgerTx(writable, index) hi = highbits(putme) exists, err := tx.Contains(index, field, view, shard, putme) @@ -624,7 +631,7 @@ func TestBadger_Remove_one_bit_test(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) // delete, but rollback instead of commit - tx := dbwrap.NewBadgerTx(writable) + tx := dbwrap.NewBadgerTx(writable, index) hi, lo := highbits(putme), lowbits(putme) _, _ = hi, lo _, err := tx.Remove(index, field, view, shard, hi) @@ -635,7 +642,7 @@ func TestBadger_Remove_one_bit_test(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) // c) within one Tx, after delete it should be gone as viewed within the txn. - tx = dbwrap.NewBadgerTx(writable) + tx = dbwrap.NewBadgerTx(writable, index) exists, err := tx.Contains(index, field, view, shard, putme) panicOn(err) @@ -724,7 +731,7 @@ func TestBadger_Max_on_many_containers(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) } - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, index) defer tx.Rollback() max, err := tx.Max(index, field, view, shard) @@ -742,7 +749,7 @@ func TestBadger_Min_on_many_containers(t *testing.T) { index, field, view, shard := "i", "f", "v", uint64(0) // verify no containers flag works - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, index) min, containersExist, err := tx.Min(index, field, view, shard) _ = min panicOn(err) @@ -759,7 +766,7 @@ func TestBadger_Min_on_many_containers(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) } - tx = dbwrap.NewBadgerTx(!writable) + tx = dbwrap.NewBadgerTx(!writable, index) defer tx.Rollback() min, containersExist, err = tx.Min(index, field, view, shard) @@ -780,7 +787,7 @@ func TestBadger_CountRange_on_many_containers(t *testing.T) { index, field, view, shard := "i", "f", "v", uint64(0) // verify no containers flag works - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, index) n, err := tx.CountRange(index, field, view, shard, 0, math.MaxUint64) panicOn(err) if n != 0 { @@ -796,7 +803,7 @@ func TestBadger_CountRange_on_many_containers(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) } - tx = dbwrap.NewBadgerTx(!writable) + tx = dbwrap.NewBadgerTx(!writable, index) defer tx.Rollback() n, err = tx.CountRange(index, field, view, shard, 0, math.MaxUint64) @@ -824,7 +831,7 @@ func TestBadger_CountRange_middle_container(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) } - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, index) defer tx.Rollback() // pick out just the middle container with the 1 bit set on it. @@ -849,7 +856,7 @@ func TestBadger_CountRange_many_middle_container(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) } - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, index) defer tx.Rollback() // get them all @@ -880,7 +887,7 @@ func TestBadger_UnionInPlace(t *testing.T) { badgerDBMustHaveBitvalue(dbwrap, index, field, view, shard, putme) } - tx2 := dbwrap.NewBadgerTx(!writable) + tx2 := dbwrap.NewBadgerTx(!writable, index) n, err := tx2.Count(index, field, view, shard) panicOn(err) if n != 2 { @@ -895,7 +902,7 @@ func TestBadger_UnionInPlace(t *testing.T) { } mustAddR(others3.Add(4 << 16)) // outside the 2<<16 container - tx := dbwrap.NewBadgerTx(writable) + tx := dbwrap.NewBadgerTx(writable, index) defer tx.Rollback() err = tx.UnionInPlace(index, field, view, shard, others, others2, others3) panicOn(err) @@ -920,7 +927,7 @@ func TestBadger_RoaringBitmap(t *testing.T) { putme := expected badgerDBMustSetBitvalue(dbwrap, index, field, view, shard, putme) - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, index) defer tx.Rollback() rbm, err := tx.RoaringBitmap(index, field, view, shard) @@ -951,10 +958,8 @@ func TestBadger_reverse_badger_iterator_and_prefix_valid(t *testing.T) { return nil }) panicOn(err) - //vv("stringifiedBadgerKeys(db) = '%v'", stringifiedBadgerKeys(dbwrap.db)) - // allkeys:["a:0", "a:1", "a:2", "b:0", "b:1", "b:2", "c:0", "c:1", "c:2", ]' - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, "no-index-avail") prefix := []byte("b:") it := NewBadgerIterator(tx, prefix) @@ -1014,10 +1019,8 @@ func TestBadger_just_reverse_badger_iterator_and_prefix_valid(t *testing.T) { return nil }) panicOn(err) - //vv("stringifiedBadgerKeys(db) = '%v'", stringifiedBadgerKeys(dbwrap.db)) - // allkeys:["a:0", "a:1", "a:2", "b:0", "b:1", "b:2", "c:0", "c:1", "c:2", ]' - tx := dbwrap.NewBadgerTx(!writable) + tx := dbwrap.NewBadgerTx(!writable, "no-index-avail") seekto := []byte("c:") prefix := []byte("b:") @@ -1046,9 +1049,9 @@ func TestBadger_ImportRoaringBits(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_ImportRoaringBits") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - defer tx.Rollback() index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + defer tx.Rollback() //bitvalue := uint64(42) @@ -1062,7 +1065,7 @@ func TestBadger_ImportRoaringBits(t *testing.T) { clear := false logme := false - changed, rowSet, err := tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize) + changed, rowSet, err := tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize, nil) _ = rowSet if changed != len(bits) { panic(fmt.Sprintf("should have changed %v bits: changed='%v', rowSet='%#v', err='%v'", len(bits), changed, rowSet, err)) @@ -1079,7 +1082,7 @@ func TestBadger_ImportRoaringBits(t *testing.T) { // now test the union in place with the same set gives no change. - changed, rowSet, err = tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize) + changed, rowSet, err = tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize, nil) _ = rowSet if changed != 0 { panic(fmt.Sprintf("should have not changed any bits on the second import, but we see changed='%v', rowSet='%#v', err='%v'", changed, rowSet, err)) @@ -1103,7 +1106,7 @@ func TestBadger_ImportRoaringBits(t *testing.T) { itr, err := roaring.NewRoaringIterator(data) panicOn(err) - changed, rowSet, err := tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize) + changed, rowSet, err := tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize, nil) _ = rowSet if changed != 1 { panic(fmt.Sprintf("should have changed 1 bit: '%v', rowSet='%#v', err='%v'", changed, rowSet, err)) @@ -1128,9 +1131,9 @@ func TestBadger_ImportRoaringBits_set_nonoverlapping_bits(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_ImportRoaringBits_set_nonoverlapping_bits") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - defer tx.Rollback() index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + defer tx.Rollback() // get some roaring bits, get an itr RoaringIterator from them rowSize := uint64(0) @@ -1148,7 +1151,7 @@ func TestBadger_ImportRoaringBits_set_nonoverlapping_bits(t *testing.T) { clear := false logme := false - changed, rowSet, err := tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize) + changed, rowSet, err := tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize, nil) _ = rowSet if changed != len(bits) { panic(fmt.Sprintf("should have changed %v bits: changed='%v', rowSet='%#v', err='%v'", len(bits), changed, rowSet, err)) @@ -1165,7 +1168,7 @@ func TestBadger_ImportRoaringBits_set_nonoverlapping_bits(t *testing.T) { // now import the 2nd, overlapping set and set them. - changed, rowSet, err = tx.ImportRoaringBits(index, field, view, shard, itr2, clear, logme, rowSize) + changed, rowSet, err = tx.ImportRoaringBits(index, field, view, shard, itr2, clear, logme, rowSize, nil) _ = rowSet if changed != 4 { panic(fmt.Sprintf("should have changed 2 bits: the 1 and the 3, but we see changed='%v', rowSet='%#v', err='%v'", changed, rowSet, err)) @@ -1178,9 +1181,9 @@ func TestBadger_ImportRoaringBits_clear_nonoverlapping_bits(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_ImportRoaringBits_clear_nonoverlapping_bits") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - defer tx.Rollback() index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + defer tx.Rollback() // get some roaring bits, get an itr RoaringIterator from them rowSize := uint64(0) @@ -1198,7 +1201,7 @@ func TestBadger_ImportRoaringBits_clear_nonoverlapping_bits(t *testing.T) { clear := false logme := false - changed, rowSet, err := tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize) + changed, rowSet, err := tx.ImportRoaringBits(index, field, view, shard, itr, clear, logme, rowSize, nil) _ = rowSet if changed != len(bits) { panic(fmt.Sprintf("should have changed %v bits: changed='%v', rowSet='%#v', err='%v'", len(bits), changed, rowSet, err)) @@ -1216,7 +1219,7 @@ func TestBadger_ImportRoaringBits_clear_nonoverlapping_bits(t *testing.T) { // now import the 2nd overlapping set and clear them. clear = true - changed, rowSet, err = tx.ImportRoaringBits(index, field, view, shard, itr2, clear, logme, rowSize) + changed, rowSet, err = tx.ImportRoaringBits(index, field, view, shard, itr2, clear, logme, rowSize, nil) _ = rowSet if changed != 2 { panic(fmt.Sprintf("should have changed 1 bit: the 2, but we see changed='%v', rowSet='%#v', err='%v'", changed, rowSet, err)) @@ -1252,9 +1255,9 @@ func TestBadger_DeleteIndex(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_DeleteIndex") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - bitvalue := uint64(777) index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + bitvalue := uint64(777) bits := []uint64{0, 3, 1 << 16, 1<<16 + 3, 8 << 16} for _, v := range bits { changed, err := tx.Add(index, field, view, shard, doBatched, v) @@ -1290,7 +1293,7 @@ func TestBadger_DeleteIndex(t *testing.T) { err = dbwrap.DeleteIndex(index) panicOn(err) - tx = dbwrap.NewBadgerTx(!writable) + tx = dbwrap.NewBadgerTx(!writable, index2) defer tx.Rollback() exists, err = tx.Contains(index2, field, view, shard, bitvalue) panicOn(err) @@ -1314,9 +1317,9 @@ func TestBadger_DeleteIndex_over100k(t *testing.T) { dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_DeleteIndex_over100k") defer clean() defer dbwrap.Close() - tx := dbwrap.NewBadgerTx(writable) - bitvalue := uint64(777) index, field, view, shard := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + bitvalue := uint64(777) limit := uint64(100002) // default batch size in DeleteIndex is 100k keys per delete transaction. //limit := uint64(101) for v := uint64(1); v < limit; v++ { @@ -1328,7 +1331,7 @@ func TestBadger_DeleteIndex_over100k(t *testing.T) { panicOn(err) if v%100000 == 0 { panicOn(tx.Commit()) - tx = dbwrap.NewBadgerTx(writable) + tx = dbwrap.NewBadgerTx(writable, index) } } @@ -1345,7 +1348,7 @@ func TestBadger_DeleteIndex_over100k(t *testing.T) { err = dbwrap.DeleteIndex(index) panicOn(err) - tx = dbwrap.NewBadgerTx(!writable) + tx = dbwrap.NewBadgerTx(!writable, index2) defer tx.Rollback() exists, err := tx.Contains(index2, field, view, shard, bitvalue) panicOn(err) @@ -1445,3 +1448,135 @@ func mustAddR(changed bool, err error) { func mustRemove(changeCount int, err error) { panicOn(err) } + +func TestBadger_DeleteFragment(t *testing.T) { + + // setup + dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_DeleteFragment") + defer clean() + defer dbwrap.Close() + index, field, view, shard0 := "i", "f", "v", uint64(0) + tx := dbwrap.NewBadgerTx(writable, index) + + shard1 := uint64(1) + + bits := []uint64{0, 3, 1 << 16, 1<<16 + 3, 8 << 16} + shards := []uint64{shard0, shard1} + for _, s := range shards { + for _, v := range bits { + changed, err := tx.Add(index, field, view, s, doBatched, v) + if changed <= 0 { + panic("should have changed") + } + panicOn(err) + } + } + + for _, s := range shards { + for _, v := range bits { + exists, err := tx.Contains(index, field, view, s, v) + panicOn(err) + if !exists { + panic("ARG bitvalue was NOT SET!!!") + } + } + } + err := tx.Commit() + panicOn(err) + //vv("Dump: %v", dbwrap.StringifiedBadgerKeys(nil)) + + // end of setup + + survivor := shard0 + victim := shard1 + err = dbwrap.DeleteFragment(index, field, view, victim, nil) + panicOn(err) + + tx = dbwrap.NewBadgerTx(!writable, index) + defer tx.Rollback() + + for _, s := range shards { + for _, v := range bits { + exists, err := tx.Contains(index, field, view, s, v) + panicOn(err) + if s == survivor { + if !exists { + panic(fmt.Sprintf("ARG survivor died : bit %v", v)) + } + } else if s == victim { // victim, should have been deleted + if exists { + panic(fmt.Sprintf("ARG victim lived : bit %v", v)) + } + } + } + } +} + +func TestBadger_shardFromBadgerKey(t *testing.T) { + if shardFromBadgerKey([]byte("idx:'i';fld:'f';vw:'standard';shd:'1';ckey@18446744073709551615")) != 1 { + panic("problem") + } + if shardFromBadgerKey([]byte("idx:'i';fld:'f';vw:'standard';shd:'0';ckey@18446744073709551615")) != 0 { + panic("problem") + } + if shardFromBadgerKey([]byte("idx:'i';fld:'f';vw:'standard';shd:'18446744073709551615';ckey@18446744073709551615")) != 18446744073709551615 { + panic("problem") + } + + func() { + defer func() { + r := recover() + if r == nil { + panic("should have panic-ed") + } + }() + // called for the panic of a short ckey, only 19 bytes instead of 20 + shardFromBadgerKey([]byte("idx:'i';fld:'f';vw:'standard';shd:'18446744073709551615';ckey@1844674407370955161")) + }() + +} + +func TestBadger_SliceOfShards(t *testing.T) { + + dbwrap, clean := mustOpenEmptyBadgerWrapper("TestBadger_SliceOfShards") + defer clean() + defer dbwrap.Close() + index, field, view := "i", "f", "v" + shards := []uint64{0, 1, 2, 3, 1000001, 2000001} + putme := uint64(179) + for _, shard := range shards { + badgerDBMustSetBitvalue(dbwrap, index, field, view, shard, putme) + } + tx := dbwrap.NewBadgerTx(!writable, index) + defer tx.Rollback() + + slc, err := tx.SliceOfShards(index, field, view, "") + panicOn(err) + for i := range shards { + if shards[i] != slc[i] { + panic(fmt.Sprintf("expected at i=%v that slc[i]=%v = shards[i]=%v", i, slc[i], shards[i])) + } + } +} + +func reportTestBadgersNeedingClose() { + globalBadgerReg.mu.Lock() + defer globalBadgerReg.mu.Unlock() + n := len(globalBadgerReg.mp) + if n > 0 { + AlwaysPrintf("*** these badgers are still open (n=%v):", n) + i := 0 + for w := range globalBadgerReg.mp { + AlwaysPrintf("i=%v, w p=%p stack:\n%v\n\n", i, w, w.startStack) + i++ + } + } +} + +var _ = reportTestBadgersNeedingClose // happy linter + +func TestMain(m *testing.M) { + ret := m.Run() + //reportTestBadgersNeedingClose() + os.Exit(ret) +} diff --git a/bluegreentx.go b/bluegreentx.go index dd58c69fd..3df87577a 100644 --- a/bluegreentx.go +++ b/bluegreentx.go @@ -16,6 +16,9 @@ package pilosa import ( "fmt" + "io" + "reflect" + "sort" "github.com/pilosa/pilosa/v2/roaring" ) @@ -27,6 +30,8 @@ type blueGreenTx struct { b Tx // b's output is returned idx *Index + + checker blueGreenChecker } func newBlueGreenTx(a, b Tx, idx *Index) *blueGreenTx { @@ -37,6 +42,10 @@ var _ = newBlueGreenTx // keep linter happy var _ Tx = (*blueGreenTx)(nil) +func (c *blueGreenTx) Type() string { + return c.a.Type() + "_" + c.b.Type() +} + func (c *blueGreenTx) Readonly() bool { a := c.a.Readonly() b := c.b.Readonly() @@ -47,6 +56,7 @@ func (c *blueGreenTx) Readonly() bool { } func (c *blueGreenTx) NewTxIterator(index, field, view string, shard uint64) *roaring.Iterator { + c.checker.see(index, field, view, shard) return c.b.NewTxIterator(index, field, view, shard) } @@ -55,11 +65,71 @@ func (c *blueGreenTx) Pointer() string { } func (c *blueGreenTx) IncrementOpN(index, field, view string, shard uint64, changedN int) { + c.checker.see(index, field, view, shard) c.a.IncrementOpN(index, field, view, shard, changedN) c.b.IncrementOpN(index, field, view, shard, changedN) } +func (c *blueGreenTx) compareTxState(index, field, view string, shard uint64) { + here := fmt.Sprintf("%v/%v/%v/%v", index, field, view, shard) + aIter, aFound, aErr := c.a.ContainerIterator(index, field, view, shard, 0) + bIter, bFound, bErr := c.b.ContainerIterator(index, field, view, shard, 0) + + if aFound != bFound { + panic(fmt.Sprintf("compareTxState[%v]: A ContainerIterator had aFound=%v, but B had bFound=%v; at '%v'", here, aFound, bFound, stack())) + } + + if aErr == nil { + defer aIter.Close() + } + if bErr == nil { + defer bIter.Close() + } + if aErr != nil || bErr != nil { + if aErr != nil && bErr != nil { + panic(fmt.Sprintf("compareTxState[%v]: A reported err '%v'; B reported err '%v' at %v", here, aErr, bErr, stack())) + } + if aErr != nil { + panic(fmt.Sprintf("compareTxState[%v]: A reported err %v at %v; but B did not", here, aErr, stack())) + } + if bErr != nil { + panic(fmt.Sprintf("compareTxState[%v]: B reported err %v at %v; but A did not", here, bErr, stack())) + } + } + for aIter.Next() { + aKey, aValue := aIter.Value() + if !bIter.Next() { + panic(fmt.Sprintf("compareTxState[%v]: A found key %v, B didn't, at %v", here, aKey, stack())) + } + bKey, bValue := bIter.Value() + if bKey != aKey { + panic(fmt.Sprintf("compareTxState[%v]: A found key %v, B found %v, at %v", here, aKey, bKey, stack())) + } + if err := aValue.BitwiseCompare(bValue); err != nil { + panic(fmt.Sprintf("compareTxState[%v]: key %v differs: %v at %v", here, aKey, err, stack())) + } + } + // end checking everything in A, but does B have more? + if bIter.Next() { + bKey, _ := bIter.Value() + panic(fmt.Sprintf("compareTxState[%v]: B found key %v, A didn't, at %v", here, bKey, stack())) + } +} + +func (c *blueGreenTx) checkDatabase() { + for index, fields := range c.checker.seen() { + for field, views := range fields { + for view, shards := range views { + for shard := range shards { + c.compareTxState(index, field, view, shard) + } + } + } + } +} + func (c *blueGreenTx) Rollback() { + c.checkDatabase() defer func() { if r := recover(); r != nil { AlwaysPrintf("see Rollback() panic '%v' at '%v'", r, stack()) @@ -71,6 +141,7 @@ func (c *blueGreenTx) Rollback() { } func (c *blueGreenTx) Commit() error { + c.checkDatabase() defer func() { if r := recover(); r != nil { AlwaysPrintf("see Commit() panic '%v' at '%v'", r, stack()) @@ -86,6 +157,7 @@ func (c *blueGreenTx) Commit() error { } func (c *blueGreenTx) RoaringBitmap(index, field, view string, shard uint64) (*roaring.Bitmap, error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see RoaringBitmap() panic '%v' at '%v'", r, stack()) @@ -100,6 +172,7 @@ func (c *blueGreenTx) RoaringBitmap(index, field, view string, shard uint64) (*r } func (c *blueGreenTx) Container(index, field, view string, shard uint64, key uint64) (ct *roaring.Container, err error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see Container() panic '%v' at '%v'", r, stack()) @@ -116,6 +189,7 @@ func (c *blueGreenTx) Container(index, field, view string, shard uint64, key uin } func (c *blueGreenTx) PutContainer(index, field, view string, shard uint64, key uint64, rc *roaring.Container) error { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see PutContainer() panic '%v' at '%v'", r, stack()) @@ -126,17 +200,11 @@ func (c *blueGreenTx) PutContainer(index, field, view string, shard uint64, key errB := c.b.PutContainer(index, field, view, shard, key, rc) compareErrors(errA, errB) - /* draft idea of how to check the full databases afterwards: - hashA := c.a.RootHashString() - hashB := c.b.RootHashString() - if hashA != hashB { - panic(fmt.Sprintf("hashA = '%v' but hashB = '%v'", hashA, hashB)) - } - */ return errB } -func (c *blueGreenTx) ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64) (changed int, rowSet map[uint64]int, err error) { +func (c *blueGreenTx) ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64, data []byte) (changed int, rowSet map[uint64]int, err error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see ImportRoaringBits() panic '%v' at '%v'", r, stack()) @@ -147,49 +215,37 @@ func (c *blueGreenTx) ImportRoaringBits(index, field, view string, shard uint64, // remember where the iterator started, so we can replay it a second time. rit2 := rit.Clone() - changedA, rowSetA, errA := c.a.ImportRoaringBits(index, field, view, shard, rit, clear, log, rowSize) + changedA, rowSetA, errA := c.a.ImportRoaringBits(index, field, view, shard, rit, clear, log, rowSize, data) - changedB, rowSetB, errB := c.b.ImportRoaringBits(index, field, view, shard, rit2, clear, log, rowSize) + changedB, rowSetB, errB := c.b.ImportRoaringBits(index, field, view, shard, rit2, clear, log, rowSize, data) - if changedA != changedB { - panic(fmt.Sprintf("changedA = %v, but changedB = %v", changedA, changedB)) - } - if len(rowSetA) != len(rowSetB) { - panic(fmt.Sprintf("rowSetA = %#v, but rowSetB = %#v", rowSetA, rowSetB)) - } - for k, va := range rowSetA { - vb, ok := rowSetB[k] - if !ok { - panic(fmt.Sprintf("diff on key '%v': present in rowSetA, but not in rowSet B. rowSetA = %#v, but rowSetB = %#v", k, rowSetA, rowSetB)) + if len(data) == 0 { + // okay to check! otherwise we are in the fragment.fillFragmentFromArchive + // case where we know that RoaringTx.ImportRoaringBits changed and rowSet will + // be inaccurate. + if changedA != changedB { + panic(fmt.Sprintf("changedA = %v, but changedB = %v", changedA, changedB)) } - if va != vb { - panic(fmt.Sprintf("diff on key '%v', rowSetA has value '%v', but rowSetB has value '%v'", k, va, vb)) + if len(rowSetA) != len(rowSetB) { + panic(fmt.Sprintf("rowSetA = %#v, but rowSetB = %#v", rowSetA, rowSetB)) + } + for k, va := range rowSetA { + vb, ok := rowSetB[k] + if !ok { + panic(fmt.Sprintf("diff on key '%v': present in rowSetA, but not in rowSet B. rowSetA = %#v, but rowSetB = %#v", k, rowSetA, rowSetB)) + } + if va != vb { + panic(fmt.Sprintf("diff on key '%v', rowSetA has value '%v', but rowSetB has value '%v'", k, va, vb)) + } } } - compareErrors(errA, errB) - //compareDatabases(c.a, c.b) return changedB, rowSetB, errB } -/* // TODO: get a database-wide checksum working -func compareDatabases(a, b Tx) { - - index, field, view, shard := "i", "f", "v", uint64(0) - - ha, errA := a.WholeDatabaseBlake3Hash(index, field, view, shard) - panicOn(errA) - hb, errB := b.WholeDatabaseBlake3Hash(index, field, view, shard) - panicOn(errB) - - if ha != hb { - panic(fmt.Sprintf("a.WholeDatabaseBlake3Hash(%T) = '%v' but b.WholeDatabaseBlake3Hash(%T) = '%v'", a, ha, b, hb)) - } -} -*/ - func (c *blueGreenTx) RemoveContainer(index, field, view string, shard uint64, key uint64) error { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see RemoveContainer() panic '%v' at '%v'", r, stack()) @@ -207,6 +263,7 @@ func (c *blueGreenTx) UseRowCache() bool { } func (c *blueGreenTx) Add(index, field, view string, shard uint64, batched bool, a ...uint64) (changeCount int, err error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see Add() panic '%v' for index='%v', field='%v', view='%v', shard='%v' at '%v'", r, index, field, view, shard, stack()) @@ -249,6 +306,7 @@ func compareErrors(errA, errB error) { } func (c *blueGreenTx) Remove(index, field, view string, shard uint64, a ...uint64) (changeCount int, err error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see Remove() panic '%v' at '%v'", r, stack()) @@ -263,6 +321,7 @@ func (c *blueGreenTx) Remove(index, field, view string, shard uint64, a ...uint6 } func (c *blueGreenTx) Contains(index, field, view string, shard uint64, key uint64) (exists bool, err error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see Contains() panic '%v' at '%v'", r, stack()) @@ -278,6 +337,7 @@ func (c *blueGreenTx) Contains(index, field, view string, shard uint64, key uint } func (c *blueGreenTx) ContainerIterator(index, field, view string, shard uint64, firstRoaringContainerKey uint64) (citer roaring.ContainerIterator, found bool, err error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see ContainerIterator() panic '%v' at '%v'", r, stack()) @@ -290,10 +350,14 @@ func (c *blueGreenTx) ContainerIterator(index, field, view string, shard uint64, bit, bfound, errB := c.b.ContainerIterator(index, field, view, shard, firstRoaringContainerKey) compareErrors(errA, errB) + if errA != nil { + ait.Close() // don't leak it. + } return bit, bfound, errB } func (c *blueGreenTx) ForEach(index, field, view string, shard uint64, fn func(i uint64) error) error { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see ForEach() panic '%v' at '%v'", r, stack()) @@ -310,6 +374,7 @@ func (c *blueGreenTx) ForEach(index, field, view string, shard uint64, fn func(i } func (c *blueGreenTx) ForEachRange(index, field, view string, shard uint64, start, end uint64, fn func(uint64) error) error { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see ForEachRange() panic '%v' at '%v'", r, stack()) @@ -326,6 +391,7 @@ func (c *blueGreenTx) ForEachRange(index, field, view string, shard uint64, star } func (c *blueGreenTx) Count(index, field, view string, shard uint64) (uint64, error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see Count() panic '%v' at '%v'", r, stack()) @@ -342,6 +408,7 @@ func (c *blueGreenTx) Count(index, field, view string, shard uint64) (uint64, er } func (c *blueGreenTx) Max(index, field, view string, shard uint64) (uint64, error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see Max() panic '%v' at '%v'", r, stack()) @@ -358,6 +425,7 @@ func (c *blueGreenTx) Max(index, field, view string, shard uint64) (uint64, erro } func (c *blueGreenTx) Min(index, field, view string, shard uint64) (uint64, bool, error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see Min() panic '%v' at '%v'", r, stack()) @@ -374,6 +442,7 @@ func (c *blueGreenTx) Min(index, field, view string, shard uint64) (uint64, bool } func (c *blueGreenTx) UnionInPlace(index, field, view string, shard uint64, others ...*roaring.Bitmap) error { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see UnionInPlace() panic '%v' at '%v'", r, stack()) @@ -387,6 +456,7 @@ func (c *blueGreenTx) UnionInPlace(index, field, view string, shard uint64, othe } func (c *blueGreenTx) CountRange(index, field, view string, shard uint64, start, end uint64) (n uint64, err error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see CountRange() panic '%v' at '%v'", r, stack()) @@ -405,6 +475,7 @@ func (c *blueGreenTx) CountRange(index, field, view string, shard uint64, start, } func (c *blueGreenTx) OffsetRange(index, field, view string, shard, offset, start, end uint64) (other *roaring.Bitmap, err error) { + c.checker.see(index, field, view, shard) defer func() { if r := recover(); r != nil { AlwaysPrintf("see OffsetRange() panic '%v' at '%v'", r, stack()) @@ -419,3 +490,136 @@ func (c *blueGreenTx) OffsetRange(index, field, view string, shard, offset, star compareErrors(errA, errB) return b, errB } + +func (c *blueGreenTx) RoaringBitmapReader(index, field, view string, shard uint64, fragmentPathForRoaring string) (r io.ReadCloser, sz int64, err error) { + c.checker.see(index, field, view, shard) + defer func() { + if r := recover(); r != nil { + AlwaysPrintf("see OffsetRange() panic '%v' at '%v'", r, stack()) + panic(r) + } + }() + + rcA, szA, errA := c.a.RoaringBitmapReader(index, field, view, shard, fragmentPathForRoaring) + rcB, szB, errB := c.b.RoaringBitmapReader(index, field, view, shard, fragmentPathForRoaring) + if szA != szB { + panic(fmt.Sprintf("szA = %v, but szB = %v", szA, szB)) + } + compareErrors(errA, errB) + return &MultiReaderB{a: rcA, b: rcB}, szB, errB +} + +func (c *blueGreenTx) SliceOfShards(index, field, view, optionalViewPath string) (sliceOfShards []uint64, err error) { + // doesn't change state, so we don't really need see() call here. And we don't have a single shard for it. + //c.checker.see(index, field, view, shard) // don't have shard. + defer func() { + if r := recover(); r != nil { + AlwaysPrintf("see SliceOfShards() panic '%v' at '%v'", r, stack()) + panic(r) + } + }() + slcA, errA := c.a.SliceOfShards(index, field, view, optionalViewPath) + slcB, errB := c.b.SliceOfShards(index, field, view, optionalViewPath) + compareErrors(errA, errB) + + // sort order may be different, and that's ok. + cpa := append([]uint64{}, slcA...) + cpb := append([]uint64{}, slcB...) + sort.Slice(cpa, func(i, j int) bool { return cpa[i] < cpa[j] }) + sort.Slice(cpb, func(i, j int) bool { return cpb[i] < cpb[j] }) + + if !reflect.DeepEqual(cpa, cpb) { + // report the first difference + ma := make(map[uint64]bool) + for _, ka := range slcA { + ma[ka] = true + } + for _, kb := range slcB { + if !ma[kb] { + panic(fmt.Sprintf("blueGreenTx SliceOfShards diference! B had %v, but A did not; in the SliceOfShards returned slice.", kb)) + } + delete(ma, kb) + } + if len(ma) != 0 { + for _, firstDifference := range ma { + panic(fmt.Sprintf("blueGreenTx SliceOfShards diference! A had %v, but B did not; in the SliceOfShards returned slice.", firstDifference)) + } + } + panic(fmt.Sprintf("blueGreenTx SliceOfShards diference \n slcA='%#v';\n slcB='%#v';\n", cpa, cpb)) + } + return slcB, errB +} + +type MultiReaderB struct { + a io.ReadCloser + b io.ReadCloser +} + +// TODO(jea): test this for accuracy/correctness. +func (m *MultiReaderB) Read(p []byte) (nB int, errB error) { + nB, errB = m.b.Read(p) + p2 := make([]byte, nB) + // discard the exact same amount from A + // ReadAtLeast reads from r into buf until it has read at least + // min bytes. It returns the number of bytes copied and an error + // if fewer bytes were read. The error is EOF only if no bytes + // were read. If an EOF happens after reading fewer than min bytes, + // ReadAtLeast returns ErrUnexpectedEOF. If min is greater than + // the length of buf, ReadAtLeast returns ErrShortBuffer. On + // return, n >= min if and only if err == nil. If r returns + // an error having read at least min bytes, the error is dropped. + nA, errA := io.ReadAtLeast(m.a, p2, nB) + if errA == io.ErrUnexpectedEOF { + panic(fmt.Sprintf("MultiReaderB got ErrUnexpectedEOF: read %v bytes from B, but could only read %v bytes for A", nB, nA)) + } + if nA != nB { + panic(fmt.Sprintf("MultiReaderB read %v bytes from B, but could only read %v bytes for A", nB, nA)) + } + return +} + +func (m *MultiReaderB) Close() error { + m.a.Close() + return m.b.Close() +} + +// blueGreenChecker is used +type blueGreenChecker struct { + visited map[string]map[string]map[string]map[uint64]struct{} + done bool +} + +// see would mark a thing as seen. +func (b *blueGreenChecker) see(index, field, view string, shard uint64) { + if b.visited == nil { + b.visited = make(map[string]map[string]map[string]map[uint64]struct{}) + } + var visitedIdx map[string]map[string]map[uint64]struct{} + var visitedField map[string]map[uint64]struct{} + var visitedView map[uint64]struct{} + + if visitedIdx = b.visited[index]; visitedIdx == nil { + visitedIdx = make(map[string]map[string]map[uint64]struct{}) + b.visited[index] = visitedIdx + } + if visitedField = visitedIdx[field]; visitedField == nil { + visitedField = make(map[string]map[uint64]struct{}) + visitedIdx[field] = visitedField + } + if visitedView = visitedField[view]; visitedView == nil { + visitedView = make(map[uint64]struct{}) + visitedField[view] = visitedView + } + visitedView[shard] = struct{}{} +} + +// seen reports the things it has seen, exactly once so +// that Rollback can be called after Commit without repeating +// the check. +func (b *blueGreenChecker) seen() map[string]map[string]map[string]map[uint64]struct{} { + if b.done { + return nil + } + b.done = true + return b.visited +} diff --git a/catcher.go b/catcher.go index a4d832c20..5623733c2 100644 --- a/catcher.go +++ b/catcher.go @@ -16,6 +16,7 @@ package pilosa import ( "fmt" + "io" "github.com/pilosa/pilosa/v2/roaring" ) @@ -51,14 +52,14 @@ func (c *catcherTx) WholeDatabaseBlake3Hash(index, field, view string, shard uin return c.b.WholeDatabaseBlake3Hash(index, field, view, shard) } -func (c *catcherTx) ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64) (changed int, rowSet map[uint64]int, err error) { +func (c *catcherTx) ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64, data []byte) (changed int, rowSet map[uint64]int, err error) { defer func() { if r := recover(); r != nil { AlwaysPrintf("see ImportRoaringBits() panic '%v' at '%v'", r, stack()) panic(r) } }() - return c.b.ImportRoaringBits(index, field, view, shard, rit, clear, log, rowSize) + return c.b.ImportRoaringBits(index, field, view, shard, rit, clear, log, rowSize, data) } func (c *catcherTx) Readonly() bool { @@ -275,3 +276,26 @@ func (c *catcherTx) OffsetRange(index, field, view string, shard, offset, start, }() return c.b.OffsetRange(index, field, view, shard, offset, start, end) } + +func (c *catcherTx) RoaringBitmapReader(index, field, view string, shard uint64, fragmentPathForRoaring string) (r io.ReadCloser, sz int64, err error) { + defer func() { + if r := recover(); r != nil { + AlwaysPrintf("see RoaringBitmapReader() panic '%v' at '%v'", r, stack()) + panic(r) + } + }() + return c.b.RoaringBitmapReader(index, field, view, shard, fragmentPathForRoaring) +} + +func (c *catcherTx) Type() string { + return c.b.Type() +} +func (c *catcherTx) SliceOfShards(index, field, view, optionalViewPath string) (sliceOfShards []uint64, err error) { + defer func() { + if r := recover(); r != nil { + AlwaysPrintf("see SliceOfShards() panic '%v' at '%v'", r, stack()) + panic(r) + } + }() + return c.b.SliceOfShards(index, field, view, optionalViewPath) +} diff --git a/cluster.go b/cluster.go index ed1d92eb9..eb0cba364 100644 --- a/cluster.go +++ b/cluster.go @@ -525,7 +525,7 @@ func (c *cluster) unprotectedSetState(state string) { cleaner.Cluster = c cleaner.Closing = c.closing - // Clean holder. + // Clean holder. This is where the shard gets removed after resize. if err := cleaner.CleanHolder(); err != nil { c.logger.Printf("holder clean error: err=%s", err) } diff --git a/cluster_internal_test.go b/cluster_internal_test.go index c31b9f294..1c344da06 100644 --- a/cluster_internal_test.go +++ b/cluster_internal_test.go @@ -96,7 +96,9 @@ func newIndexWithTempPath(name string) *Index { if err != nil { panic(err) } - index, err := NewIndex(NewHolder(DefaultPartitionN), path, name) + h := NewHolder(DefaultPartitionN) + h.Path = path + index, err := h.CreateIndex(name, IndexOptions{}) if err != nil { panic(err) } @@ -160,7 +162,7 @@ func TestFragSources(t *testing.T) { defer idx.Close() // Obtain transaction. - tx := &RoaringTx{Index: idx} + tx := idx.Txf.NewTx(Txo{Write: writable, Index: idx}) defer tx.Rollback() field, err := idx.CreateFieldIfNotExists("f", OptFieldTypeDefault()) @@ -809,12 +811,12 @@ func TestCluster_ResizeStates(t *testing.T) { if idx0 == nil { t.Fatal(`idx0 was nil, could not retrieve Index("i")`) } - //idx0.Dump("node0") // addNode needs to block until the resize process has completed. if err := tc.addNode(); err != nil { t.Fatalf("adding node: %v", err) } + node1 := tc.Clusters[1] // Ensure that nodes come up in state NORMAL. @@ -846,8 +848,6 @@ func TestCluster_ResizeStates(t *testing.T) { if idx1 == nil { t.Fatal(`idx1 was nil, could not retrieve Index("i")`) } - //idx0.Dump("after rebalance, node0") - //idx1.Dump("after rebalance, node1") // Ensure checksums are the same. if chksum, err := node1Fragment.Checksum(); err != nil { @@ -872,6 +872,7 @@ func TestAE(t *testing.T) { c.abortAntiEntropy() close(ch) }() + defer c.abortAntiEntropyQ() // avoid leaking a goroutine. select { case <-ch: return @@ -883,11 +884,13 @@ func TestAE(t *testing.T) { t.Run("AbortBlocksInitialized", func(t *testing.T) { c := newCluster() c.initializeAntiEntropy() + ch := make(chan struct{}) go func() { c.abortAntiEntropy() close(ch) }() + defer c.abortAntiEntropyQ() // avoid leak of goroutine. select { case <-ch: t.Fatalf("aborting anti entropy on an initialized cluster didn't block") diff --git a/executor.go b/executor.go index 62b98b5be..6a072372a 100644 --- a/executor.go +++ b/executor.go @@ -19,12 +19,12 @@ import ( "encoding/json" "fmt" "math" + "math/bits" "sort" "strings" "sync" "time" - "github.com/molecula/ext" "github.com/pilosa/pilosa/v2/pql" pb "github.com/pilosa/pilosa/v2/proto" "github.com/pilosa/pilosa/v2/roaring" @@ -64,12 +64,6 @@ type executor struct { workersWG sync.WaitGroup workerPoolSize int work chan job - // global registry to check for name clashes - additionalOps map[string]*ext.BitmapOp - // typed registries we can use in lookups - additionalBitmapOps map[string]ext.BitmapOpBitmap - additionalCountOps map[string]ext.BitmapOpUnaryCount - additionalFieldOps map[string]ext.BitmapOpBSIBitmap } // executorOption is a functional option type for pilosa.Executor @@ -125,38 +119,6 @@ func (e *executor) Close() error { return nil } -func (e *executor) registerOps(ops []ext.BitmapOp) error { - if e.additionalOps == nil { - e.additionalOps = make(map[string]*ext.BitmapOp) - e.additionalBitmapOps = make(map[string]ext.BitmapOpBitmap) - e.additionalCountOps = make(map[string]ext.BitmapOpUnaryCount) - e.additionalFieldOps = make(map[string]ext.BitmapOpBSIBitmap) - } - for i, op := range ops { - name := op.Name - if _, exists := e.additionalOps[name]; exists { - return fmt.Errorf("op name '%s' already defined", name) - } - e.additionalOps[name] = &ops[i] - typ := ops[i].Func.BitmapOpType() - switch { - case typ.Input == ext.OpInputBitmap && typ.Output == ext.OpOutputCount: - e.additionalCountOps[name] = ops[i].Func.(ext.BitmapOpUnaryCount) - case typ.Input == ext.OpInputBitmap && typ.Output == ext.OpOutputBitmap: - e.additionalBitmapOps[name] = ops[i].Func.(ext.BitmapOpBitmap) - case typ.Input == ext.OpInputNaryBSI && typ.Output == ext.OpOutputSignedBitmap: - if fn, ok := ops[i].Func.(ext.BitmapOpBSIBitmapPrecall); ok { - e.additionalFieldOps[name] = ext.BitmapOpBSIBitmap(fn) - } else { - e.additionalFieldOps[name] = ops[i].Func.(ext.BitmapOpBSIBitmap) - } - default: - return fmt.Errorf("unsupported types for '%s': input type %d, output type %d", name, typ.Input, typ.Output) - } - } - return nil -} - // Execute executes a PQL query. func (e *executor) Execute(ctx context.Context, index string, q *pql.Query, shards []uint64, opt *execOptions) (QueryResponse, error) { span, ctx := tracing.StartSpanFromContext(ctx, "Executor.Execute") @@ -228,7 +190,6 @@ func (e *executor) Execute(ctx context.Context, index string, q *pql.Query, shar } else if err := validateQueryContext(ctx); err != nil { return resp, err } - resp.Results = results // Fill column attributes if requested. @@ -369,7 +330,6 @@ func (e *executor) readColumnAttrSets(index *Index, ids []uint64) ([]*ColumnAttr // handlePreCalls traverses the call tree looking for calls that need // precomputed values. Right now, that's just Distinct. func (e *executor) handlePreCalls(ctx context.Context, tx Tx, index string, c *pql.Call, shards []uint64, opt *execOptions) error { - if c.Name == "Precomputed" { idx := c.Args["valueidx"].(int64) if idx >= 0 && idx < int64(len(opt.EmbeddedData)) { @@ -404,18 +364,17 @@ func (e *executor) handlePreCalls(ctx context.Context, tx Tx, index string, c *p // like Distinct, where you can't predict output shard for a result // from the shard being queried. if newIndex != "" && newIndex != index { - if err := e.handlePreCallChildren(ctx, tx, index, c, shards, opt); err != nil { - return err - } - c.Type = pql.PrecallGlobal index = newIndex // we need to recompute shards, then shards = nil } + if err := e.handlePreCallChildren(ctx, tx, index, c, shards, opt); err != nil { + return err + } + // child calls already handled, no precall for this, so we're done if c.Type == pql.PrecallNone { - // otherwise, handle the children - return e.handlePreCallChildren(ctx, tx, index, c, shards, opt) + return nil } // We don't try to handle sub-calls from here. I'm not 100% // sure that's right, but I think the fact that they're happening @@ -548,6 +507,7 @@ func (e *executor) execute(ctx context.Context, tx Tx, index string, q *pql.Quer if err != nil { return nil, err } + results = append(results, v) // Some Calls can have significant data associated with them // that gets generated during processing, such as Precomputed @@ -708,15 +668,6 @@ func (e *executor) executeCall(ctx context.Context, tx Tx, index string, c *pql. return nil, err } - // Special handling for mutation and top-n calls. - if op, ok := e.additionalCountOps[c.Name]; ok { - statFn() - return e.executeGenericCount(ctx, tx, index, c, op, shards, opt) - } - if op, ok := e.additionalFieldOps[c.Name]; ok { - statFn() - return e.executeGenericField(ctx, tx, index, c, op, shards, opt) - } switch c.Name { case "Sum": statFn() @@ -739,6 +690,9 @@ func (e *executor) executeCall(ctx context.Context, tx Tx, index string, c *pql. case "ClearRow": statFn() return e.executeClearRow(ctx, tx, index, c, shards, opt) + case "Distinct": + statFn() + return e.executeDistinct(ctx, tx, index, c, shards, opt) case "Store": statFn() return e.executeSetRow(ctx, tx, index, c, shards, opt) @@ -1151,11 +1105,9 @@ func (e *executor) executeSum(ctx context.Context, tx Tx, index string, c *pql.C return other, nil } -// executeGenericField executes a generic call on a field. Note that in this -// implementation, the operation is always a BSI op. -func (e *executor) executeGenericField(ctx context.Context, tx Tx, index string, c *pql.Call, op ext.BitmapOpBSIBitmap, shards []uint64, opt *execOptions) (SignedRow, error) { - span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeGenericField") - span.LogKV("name", c.Name) +// executeDistinct executes a Distinct call on a field. +func (e *executor) executeDistinct(ctx context.Context, tx Tx, index string, c *pql.Call, shards []uint64, opt *execOptions) (SignedRow, error) { + span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeDistinct") defer span.Finish() field := c.Args["field"] @@ -1165,7 +1117,7 @@ func (e *executor) executeGenericField(ctx context.Context, tx Tx, index string, // Execute calls in bulk on each remote node and merge. mapFn := func(ctx context.Context, shard uint64) (interface{}, error) { - return e.executeGenericFieldShard(ctx, tx, index, c, op, shard) + return e.executeDistinctShard(ctx, tx, index, c, shard) } // Merge returned results at coordinating node. @@ -1440,13 +1392,6 @@ func (e *executor) executeBitmapCallShard(ctx context.Context, tx Tx, index stri span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeBitmapCallShard") defer span.Finish() - if _, ok := e.additionalCountOps[c.Name]; ok { - return nil, fmt.Errorf("count op %s used as bitmap call", c.Name) - } - if op, ok := e.additionalBitmapOps[c.Name]; ok { - return e.executeGenericBitmapShard(ctx, tx, index, c, op, shard) - } - switch c.Name { case "Row", "Range": return e.executeRowShard(ctx, tx, index, c, shard) @@ -1464,6 +1409,8 @@ func (e *executor) executeBitmapCallShard(ctx context.Context, tx Tx, index stri return e.executeShiftShard(ctx, tx, index, c, shard) case "All": // Allow a shard computation to use All() (note, limit/offset not applied) return e.executeAllCallShard(ctx, tx, index, c, shard) + case "Distinct": + return nil, errors.New("Distinct shouldn't be hit as a bitmap call") case "Precomputed": return e.executePrecomputedCallShard(ctx, tx, index, c, shard) default: @@ -1471,11 +1418,10 @@ func (e *executor) executeBitmapCallShard(ctx context.Context, tx Tx, index stri } } -// executeGenericFieldShard executes a generic/extension command on a -// single shard. Note that in this implementation, the op is always -// a BSI op. -func (e *executor) executeGenericFieldShard(ctx context.Context, tx Tx, index string, c *pql.Call, op ext.BitmapOpBSIBitmap, shard uint64) (SignedRow, error) { - span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeGenericShard") +// executeDistinctShard executes a Distinct call on a single shard, yielding +// a SignedRow of the values found. +func (e *executor) executeDistinctShard(ctx context.Context, tx Tx, index string, c *pql.Call, shard uint64) (result SignedRow, err error) { + span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeDistinctShard") defer span.Finish() var filter *Row @@ -1483,7 +1429,7 @@ func (e *executor) executeGenericFieldShard(ctx context.Context, tx Tx, index st if len(c.Children) == 1 { row, err := e.executeBitmapCallShard(ctx, tx, index, c.Children[0], shard) if err != nil { - return SignedRow{}, errors.Wrap(err, "executing bitmap call") + return result, errors.Wrap(err, "executing bitmap call") } filter = row if filter != nil && len(filter.segments) > 0 { @@ -1497,29 +1443,115 @@ func (e *executor) executeGenericFieldShard(ctx context.Context, tx Tx, index st field := e.Holder.Field(index, fieldName) if field == nil { - return SignedRow{}, nil + return result, nil } bsig := field.bsiGroup(fieldName) if bsig == nil { - return SignedRow{}, nil + return result, nil + } + view := viewBSIGroupPrefix + fieldName + + depth := uint64(bsig.BitDepth) + offset := bsig.Base + + existsBitmap, err := tx.OffsetRange(index, fieldName, view, shard, 0, ShardWidth*0, ShardWidth*1) + if err != nil { + return result, err + } + if filter != nil { + existsBitmap = existsBitmap.Intersect(filterBitmap) + } + if !existsBitmap.Any() { + return result, nil } - fragment := e.Holder.fragment(index, fieldName, viewBSIGroupPrefix+fieldName, shard) - if fragment == nil { - return SignedRow{}, nil + signBitmap, err := tx.OffsetRange(index, fieldName, view, shard, 0, ShardWidth*1, ShardWidth*2) + if err != nil { + return result, nil } - var out ext.SignedBitmap - if filterBitmap != nil { - out = op(ext.BitmapBSI{FieldData: WrapBitmap(fragment.storage), ShardWidth: ShardWidth, Offset: bsig.Base, Depth: bsig.BitDepth}, []ext.Bitmap{WrapBitmap(filterBitmap)}, c.Args) - } else { - out = op(ext.BitmapBSI{FieldData: WrapBitmap(fragment.storage), ShardWidth: ShardWidth, Offset: bsig.Base, Depth: bsig.BitDepth}, []ext.Bitmap{}, c.Args) + dataBitmaps := make([]*roaring.Bitmap, depth) + + for i := uint64(0); i < depth; i++ { + dataBitmaps[i], err = tx.OffsetRange(index, fieldName, view, shard, 0, ShardWidth*(i+2), ShardWidth*(i+3)) + if err != nil { + return result, err + } } + // we need spaces for sign bit, existence/filter bit, and data + // row bits, which we'll be grabbing 65k bits at a time + stashWords := make([]uint64, 1024*(depth+2)) + bitStashes := make([][]uint64, depth) + for i := uint64(0); i < depth; i++ { + start := i * 1024 + last := start + 1024 + bitStashes[i] = stashWords[start:last] + i++ + } + stashOffset := depth * 1024 + existStash := stashWords[stashOffset : stashOffset+1024] + signStash := stashWords[stashOffset+1024 : stashOffset+2048] + dataBits := make([][]uint64, depth) + + posValues := make([]uint64, 0, 64) + negValues := make([]uint64, 0, 64) + + posBitmap := roaring.NewFileBitmap() + negBitmap := roaring.NewFileBitmap() + + existIterator, _ := existsBitmap.Containers.Iterator(0) + for existIterator.Next() { + key, value := existIterator.Value() + if value.N() == 0 { + continue + } + exists := value.AsBitmap(existStash) + sign := signBitmap.Containers.Get(key).AsBitmap(signStash) + for i := uint64(0); i < depth; i++ { + dataBits[i] = dataBitmaps[i].Containers.Get(key).AsBitmap(bitStashes[i]) + } + for idx, word := range exists { + // mask holds a mask we can test the other words against. + mask := uint64(1) + for word != 0 { + shift := uint(bits.TrailingZeros64(word)) + // we shift one *more* than that, to move the + // actual one bit off. + word >>= shift + 1 + mask <<= shift + value := int64(0) + for b := uint64(0); b < depth; b++ { + if dataBits[b][idx]&mask != 0 { + value += (1 << b) + } + } + if sign[idx]&mask != 0 { + value *= -1 + } + value += int64(offset) + if value < 0 { + negValues = append(negValues, uint64(-value)) + } else { + posValues = append(posValues, uint64(value)) + } + // and now we processed that bit, so we move the mask over one. + mask <<= 1 + } + if len(negValues) > 0 { + _, _ = negBitmap.AddN(negValues...) + negValues = negValues[:0] + } + if len(posValues) > 0 { + _, _ = posBitmap.AddN(posValues...) + posValues = posValues[:0] + } + } + } return SignedRow{ - Neg: NewRowFromBitmap(UnwrapBitmap(out.Neg)), - Pos: NewRowFromBitmap(UnwrapBitmap(out.Pos)), + Neg: NewRowFromBitmap(negBitmap), + Pos: NewRowFromBitmap(posBitmap), }, nil } @@ -2681,11 +2713,13 @@ func (e *executor) executeRowsShard(ctx context.Context, tx Tx, index string, fi } func (e *executor) executeRowShard(ctx context.Context, tx Tx, index string, c *pql.Call, shard uint64) (*Row, error) { + span, _ := tracing.StartSpanFromContext(ctx, "Executor.executeRowShard") defer span.Finish() // Handle bsiGroup ranges differently. if c.HasConditionArg() { + // looks the same on badger/roaring. we think. return e.executeRowBSIGroupShard(ctx, tx, index, c, shard) } @@ -2795,6 +2829,7 @@ func (e *executor) executeRowShard(ctx context.Context, tx Tx, index string, c * // executeRowBSIGroupShard executes a range(bsiGroup) call for a local shard. func (e *executor) executeRowBSIGroupShard(ctx context.Context, tx Tx, index string, c *pql.Call, shard uint64) (_ *Row, err error) { + span, _ := tracing.StartSpanFromContext(ctx, "Executor.executeRowBSIGroupShard") defer span.Finish() @@ -2975,44 +3010,6 @@ func (e *executor) executeIntersectShard(ctx context.Context, tx Tx, index strin return other, nil } -// executeGenericBitmapShard executes a generic bitmap call for a local shard. -func (e *executor) executeGenericBitmapShard(ctx context.Context, tx Tx, index string, c *pql.Call, op ext.BitmapOpBitmap, shard uint64) (*Row, error) { - span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeGenericBitmapShard") - defer span.Finish() - - if op.BitmapOpArity() == ext.OpArityUnary { - if len(c.Children) != 1 { - return nil, fmt.Errorf("%s needs exactly one row parameter", c.Name) - } - row, err := e.executeBitmapCallShard(ctx, tx, index, c.Children[0], shard) - if err != nil { - return nil, err - } - return row.GenericUnaryOp(op.BitmapOpFunc(), c.Args), nil - } - - var err error - rows := make([]*Row, len(c.Children)) - for i, input := range c.Children { - rows[i], err = e.executeBitmapCallShard(ctx, tx, index, input, shard) - if err != nil { - return nil, err - } - } - var other *Row - switch op.BitmapOpArity() { - case ext.OpArityBinary: - other = rows[0] - for _, row := range rows[1:] { - other = other.GenericBinaryOp(op.BitmapOpFunc(), row, c.Args) - } - case ext.OpArityNary: - other = rows[0].GenericNaryOp(op.BitmapOpFunc(), rows[1:], c.Args) - } - other.invalidateCount() - return other, nil -} - // executeUnionShard executes a union() call for a local shard. func (e *executor) executeUnionShard(ctx context.Context, tx Tx, index string, c *pql.Call, shard uint64) (*Row, error) { span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeUnionShard") @@ -3164,41 +3161,6 @@ func (e *executor) executeShiftShard(ctx context.Context, tx Tx, index string, c return row.Shift(n) } -// executeGeneric executes a provided count-like call. -func (e *executor) executeGenericCount(ctx context.Context, tx Tx, index string, c *pql.Call, op ext.BitmapOpUnaryCount, shards []uint64, opt *execOptions) (uint64, error) { - span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeGenericCount") - defer span.Finish() - - if len(c.Children) == 0 { - return 0, fmt.Errorf("%s() requires an input bitmap", c.Name) - } else if len(c.Children) > 1 { - return 0, fmt.Errorf("%s() only accepts a single bitmap input", c.Name) - } - - // Execute calls in bulk on each remote node and merge. - mapFn := func(ctx context.Context, shard uint64) (interface{}, error) { - row, err := e.executeBitmapCallShard(ctx, tx, index, c.Children[0], shard) - if err != nil { - return 0, err - } - return row.GenericCount(op, c.Args), nil - } - - // Merge returned results at coordinating node. - reduceFn := func(ctx context.Context, prev, v interface{}) interface{} { - other, _ := prev.(uint64) - return other + v.(uint64) - } - - result, err := e.mapReduce(ctx, index, shards, c, opt, mapFn, reduceFn) - if err != nil { - return 0, err - } - n, _ := result.(uint64) - - return n, nil -} - // executeCount executes a count() call. func (e *executor) executeCount(ctx context.Context, tx Tx, index string, c *pql.Call, shards []uint64, opt *execOptions) (uint64, error) { span, ctx := tracing.StartSpanFromContext(ctx, "Executor.executeCount") diff --git a/executor_test.go b/executor_test.go index 9a78b96c0..73f1d44ce 100644 --- a/executor_test.go +++ b/executor_test.go @@ -3413,6 +3413,7 @@ func TestExecutor_Execute_Existence(t *testing.T) { defer c.Close() hldr := test.Holder{Holder: c[0].Server.Holder()} index := hldr.MustCreateIndexIfNotExists("i", pilosa.IndexOptions{TrackExistence: true}) + _, err := index.CreateField("f", pilosa.OptFieldTypeDefault()) if err != nil { t.Fatal(err) @@ -3438,6 +3439,7 @@ func TestExecutor_Execute_Existence(t *testing.T) { } else if bits := res.Results[0].(*pilosa.Row).Columns(); !reflect.DeepEqual(bits, []uint64{ShardWidth + 2}) { t.Fatalf("unexpected columns after Not: %+v", bits) } + // Reopen cluster to ensure existence field is reloaded. if err := c[0].Reopen(); err != nil { t.Fatal(err) @@ -4174,25 +4176,6 @@ func TestExecutor_Execute_SetRow(t *testing.T) { t.Fatalf("unexpected columns: %+v", bits) } }) - t.Run("Err_Store(Distinct)", func(t *testing.T) { - c := test.MustRunCluster(t, 1) - defer c.Close() - hldr := test.Holder{Holder: c[0].Server.Holder()} - index := hldr.MustCreateIndexIfNotExists("i", pilosa.IndexOptions{TrackExistence: true}) - f1, err := index.CreateField("f1", pilosa.OptFieldTypeDefault()) - if err != nil { - t.Fatal(err) - } - f2, err := index.CreateField("f2", pilosa.OptFieldTypeDefault()) - if err != nil { - t.Fatal(err) - } - - q := fmt.Sprintf(`Store(Distinct(field=%s), %s=2)`, f1.Name(), f2.Name()) - if res, err := c[0].API.Query(context.Background(), &pilosa.QueryRequest{Index: index.Name(), Query: q}); err == nil { - t.Fatalf("expected 'unsupported result type' error, got: %+v", res) - } - }) } func benchmarkExistence(nn bool, b *testing.B) { @@ -5312,6 +5295,7 @@ func runCallTest(t *testing.T, writeQuery string, readQueries []string, indexOpt defer c.Close() hldr := test.Holder{Holder: c[0].Server.Holder()} index := hldr.MustCreateIndexIfNotExists("i", *indexOptions) + defer index.Close() _, err := index.CreateField("f", fieldOption...) if err != nil { t.Fatal(err) @@ -5723,12 +5707,15 @@ func TestExecutor_Execute_CountDistinct(t *testing.T) { if err := json.NewDecoder(bytes.NewReader(data)).Decode(schema); err != nil { t.Fatal(err) } + if err := api.ApplySchema(context.TODO(), schema, false); err != nil { t.Fatal(err) } + // AntitodePoint == row 1 b/c keys field. writeQuery := `Set(100, type=AntidotePoint)Set(100, equip_id=100)Set(100, site_id=100)Set(100, id=100)` - for _, i := range schema.Indexes { + for k, i := range schema.Indexes { + _ = k if _, err := api.Query(context.TODO(), &pilosa.QueryRequest{Index: i.Name, Query: writeQuery}); err != nil { t.Fatal(err) } @@ -5739,11 +5726,11 @@ func TestExecutor_Execute_CountDistinct(t *testing.T) { Intersect( Distinct( Intersect(Row(type=AntidotePoint)), - index=power_ts, field=equip_id), + index=equipment, field=equip_id), Distinct( Intersect(Row(type=AntidotePoint)), - index=power_ts, field=equip_id) - ), index=equipment, field=site_id)` + index=sites, field=equip_id) + ), index=power_ts, field=site_id)` // Check if test query gives correct results (one column 100) t.Run("Distinct", func(t *testing.T) { @@ -5760,6 +5747,7 @@ func TestExecutor_Execute_CountDistinct(t *testing.T) { } if r.Pos.Count() != 1 { t.Fatalf("invalid pilosa.SignedRow.Pos.Count, expected: 1, got: %v", r.Pos.Count()) + } if r.Pos.Columns()[0] != 100 { t.Fatalf("invalid pilosa.SignedRow.Pos.Columns, expected: [100], got: %v", r.Pos.Columns()) diff --git a/extension.go b/extension.go deleted file mode 100644 index 1a492cce3..000000000 --- a/extension.go +++ /dev/null @@ -1,96 +0,0 @@ -// Copyright 2019 Pilosa Corp. -// -// Licensed under the Apache License, Version 2.0 (the "License"); -// you may not use this file except in compliance with the License. -// You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, software -// distributed under the License is distributed on an "AS IS" BASIS, -// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -// See the License for the specific language governing permissions and -// limitations under the License. - -package pilosa - -import ( - "fmt" - - "github.com/molecula/ext" - "github.com/pilosa/pilosa/v2/roaring" -) - -// WrapBitmap yields an extension-Bitmap from a roaring Bitmap. -func WrapBitmap(bm *roaring.Bitmap) ext.Bitmap { - return wrappedBitmap{bm} -} - -// wrappedBitmap is a very shallow glue shim to convert a roaring Bitmap to -// an extension Bitmap. -type wrappedBitmap struct{ *roaring.Bitmap } - -// UnwrapBitmap converts an extension-bitmap to its underlying roaring Bitmap. -func UnwrapBitmap(bm ext.Bitmap) *roaring.Bitmap { - if inner, ok := bm.(wrappedBitmap); ok { - if inner.Bitmap != nil { - return inner.Bitmap - } - return roaring.NewFileBitmap() - } - return roaring.NewFileBitmap() -} - -func (b wrappedBitmap) Intersect(other ext.Bitmap) ext.Bitmap { - return wrappedBitmap{b.Bitmap.Intersect(other.(wrappedBitmap).Bitmap)} -} - -func (b wrappedBitmap) Union(other ext.Bitmap) ext.Bitmap { - return wrappedBitmap{b.Bitmap.Union(other.(wrappedBitmap).Bitmap)} -} - -func (b wrappedBitmap) IntersectionCount(other ext.Bitmap) uint64 { - return b.Bitmap.IntersectionCount(other.(wrappedBitmap).Bitmap) -} - -func (b wrappedBitmap) Difference(other ext.Bitmap) ext.Bitmap { - return wrappedBitmap{b.Bitmap.Difference(other.(wrappedBitmap).Bitmap)} -} - -func (b wrappedBitmap) Xor(other ext.Bitmap) ext.Bitmap { - return wrappedBitmap{b.Bitmap.Xor(other.(wrappedBitmap).Bitmap)} -} - -func (b wrappedBitmap) Shift(n int) (ext.Bitmap, error) { - shifted, err := b.Bitmap.Shift(n) - return wrappedBitmap{shifted}, err -} - -func (b wrappedBitmap) Flip(start, last uint64) ext.Bitmap { - return wrappedBitmap{b.Bitmap.Flip(start, last)} -} - -func (b wrappedBitmap) New() ext.Bitmap { - return WrapBitmap(roaring.NewFileBitmap()) -} - -// ContainerBits tries to get one container's worth of bits. -func (b wrappedBitmap) ContainerBits(offset uint64, target []uint64) (out []uint64) { - // it's an error to call this with a non-container-aligned offset - if offset&0xFFFF != 0 { - return nil - } - if b.Bitmap == nil { - fmt.Printf("ContainerBits on bitmap with no contents\n") - return nil - } - if b.Bitmap.Containers == nil { - fmt.Printf("ContainerBits on bitmap with nil Containers\n") - return nil - } - c := b.Bitmap.Containers.Get(offset >> 16) - if c == nil { - return nil - } - return c.AsBitmap(target) -} diff --git a/field.go b/field.go index a9a80e2fe..0802c3848 100644 --- a/field.go +++ b/field.go @@ -92,6 +92,8 @@ type Field struct { name string qualifiedName string + idx *Index + viewMap map[string]*view // Row attribute storage and cache @@ -1181,6 +1183,7 @@ func (f *Field) createViewIfNotExistsBase(name string) (*view, bool, error) { func (f *Field) newView(path, name string) *view { view := newView(f.holder, path, f.index, f.name, name, f.options) + view.idx = f.idx view.rowAttrStore = f.rowAttrStore view.stats = f.Stats view.broadcaster = f.broadcaster @@ -1432,6 +1435,14 @@ func (f *Field) SetValue(tx Tx, columnID uint64, value int64) (changed bool, err if err != nil { return false, errors.Wrap(err, "creating view") } + if view.holder == nil { + panic("view.holder should not be nil") + } + if view.idx == nil { + panic("view.idx should not be nil") + } + view.holder.addIndexFromField(view.idx) + return view.setValue(tx, columnID, bsig.BitDepth, baseValue) } @@ -1761,6 +1772,10 @@ func (f *Field) importRoaring(ctx context.Context, tx Tx, data []byte, shard uin return nil } +func (f *Field) GetIndex() *Index { + return f.idx +} + func (f *Field) importRoaringOverwrite(ctx context.Context, tx Tx, data []byte, shard uint64, viewName string, block int) error { span, ctx := tracing.StartSpanFromContext(ctx, "Field.importRoaringOverwrite") defer span.Finish() @@ -1787,7 +1802,7 @@ func (f *Field) importRoaringOverwrite(ctx context.Context, tx Tx, data []byte, switch f.Options().Type { case FieldTypeInt, FieldTypeDecimal: frag.mu.Lock() - if err := frag.calculateMaxRowID(); err != nil { + if err := frag.calculateMaxRowID(tx); err != nil { return err } maxRowID, _, err := frag.maxRow(tx, nil) @@ -2132,6 +2147,9 @@ func isValidCacheType(v string) bool { } } +// TODO(jea): why isn't this bits.Len64(x) using import "math/bits" +// That would be much (80x or more) faster and correct if the high bit is set. +// // bitDepth returns the number of bits required to store a value. func bitDepth(v uint64) uint { for i := uint(0); i < 63; i++ { diff --git a/field_internal_test.go b/field_internal_test.go index 8a5aaffe4..529cabee8 100644 --- a/field_internal_test.go +++ b/field_internal_test.go @@ -205,10 +205,17 @@ func NewTestField(t *testing.T, opts FieldOption) *TestField { if err != nil { t.Fatal(err) } - field, err := NewField(NewHolder(DefaultPartitionN), path, "i", "f", opts) + h := NewHolder(DefaultPartitionN) + h.Path = path + idx, err := h.CreateIndex("i", IndexOptions{}) + if err != nil { + panic(err) + } + field, err := NewField(h, path, "i", "f", opts) if err != nil { t.Fatal(err) } + field.idx = idx return &TestField{Field: field} } @@ -223,6 +230,9 @@ func OpenField(t *testing.T, opts FieldOption) *TestField { // Close closes the field and removes the underlying data. func (f *TestField) Close() error { + if f.idx != nil { + panicOn(f.idx.Txf.CloseIndex(f.idx)) + } defer os.RemoveAll(f.Path()) return f.Field.Close() } @@ -235,10 +245,17 @@ func (f *TestField) Reopen() error { } path, index, name := f.Path(), f.Index(), f.Name() - f.Field, err = NewField(NewHolder(DefaultPartitionN), path, index, name, OptFieldTypeDefault()) + h := NewHolder(DefaultPartitionN) + h.Path = path + idx, err := h.CreateIndex(index, IndexOptions{}) if err != nil { return err } + f.Field, err = NewField(h, path, index, name, OptFieldTypeDefault()) + if err != nil { + return err + } + f.Field.idx = idx if err := f.Open(); err != nil { return err @@ -311,7 +328,8 @@ func TestField_RowTime(t *testing.T) { defer f.Close() // Obtain transaction. - tx := &RoaringTx{Field: f.Field} + tx := f.idx.Txf.NewTx(Txo{Write: writable, Index: f.idx, Field: f.Field}) + defer tx.Rollback() if err := f.setTimeQuantum(TimeQuantum("YMDH")); err != nil { t.Fatal(err) @@ -323,6 +341,12 @@ func TestField_RowTime(t *testing.T) { f.MustSetBit(tx, 1, 4, time.Date(2010, time.January, 6, 12, 0, 0, 0, time.UTC)) f.MustSetBit(tx, 1, 5, time.Date(2010, time.January, 5, 13, 0, 0, 0, time.UTC)) + panicOn(tx.Commit()) + + // obtain 2nd transaction to read it back. + tx = f.idx.Txf.NewTx(Txo{Write: !writable, Index: f.idx, Field: f.Field}) + defer tx.Rollback() + if r, err := f.RowTime(tx, 1, time.Date(2010, time.November, 5, 12, 0, 0, 0, time.UTC), "Y"); err != nil { t.Fatal(err) } else if !reflect.DeepEqual(r.Columns(), []uint64{1, 3, 4, 5}) { @@ -358,6 +382,7 @@ func TestField_RowTime(t *testing.T) { func TestField_PersistAvailableShards(t *testing.T) { availableShardFileFlushDuration.Set(200 * time.Millisecond) //shorten the default time to force a file write f := OpenField(t, OptFieldTypeDefault()) + defer f.Close() // bm represents remote available shards. bm := roaring.NewBitmap(1, 2, 3) @@ -379,6 +404,7 @@ func TestField_PersistAvailableShards(t *testing.T) { func TestField_CorruptAvailableShards(t *testing.T) { availableShardFileFlushDuration.Set(200 * time.Millisecond) //shorten the default time to force a file write f := OpenField(t, OptFieldTypeDefault()) + defer f.Close() // bm represents remote available shards. bm := roaring.NewBitmap(1, 2, 3) @@ -411,6 +437,7 @@ func TestField_CorruptAvailableShards(t *testing.T) { func TestField_TruncatedAvailableShards(t *testing.T) { availableShardFileFlushDuration.Set(200 * time.Millisecond) //shorten the default time to force a file write f := OpenField(t, OptFieldTypeDefault()) + defer f.Close() // bm represents remote available shards. bm := roaring.NewBitmap(1, 2, 3) @@ -441,6 +468,7 @@ func TestField_TruncatedAvailableShards(t *testing.T) { func TestField_PersistAvailableShardsFootprint(t *testing.T) { availableShardFileFlushDuration.Set(200 * time.Millisecond) //shorten the default time to force a file write f := OpenField(t, OptFieldTypeDefault()) + defer f.Close() // bm represents remote available shards. bm := roaring.NewBitmap() @@ -554,6 +582,7 @@ func TestField_ApplyOptions(t *testing.T) { // to result in a value of 9 instead of 1. func TestBSIGroup_importValue(t *testing.T) { f := OpenField(t, OptFieldTypeInt(-100, 200)) + defer f.Close() options := &ImportOptions{} for i, tt := range []struct { @@ -581,12 +610,17 @@ func TestBSIGroup_importValue(t *testing.T) { []uint64{100}, }, } { - tx := &RoaringTx{Field: f.Field} + tx := f.idx.Txf.NewTx(Txo{Write: writable, Index: f.idx, Field: f.Field}) + defer tx.Rollback() if err := f.importValue(tx, tt.columnIDs, tt.values, options); err != nil { t.Fatalf("test %d, importing values: %s", i, err.Error()) } + panicOn(tx.Commit()) + tx = f.idx.Txf.NewTx(Txo{Write: !writable, Index: f.idx, Field: f.Field}) + defer tx.Rollback() + if row, err := f.Range(tx, f.name, pql.EQ, tt.checkVal); err != nil { t.Fatalf("test %d, getting range: %s", i, err.Error()) } else if !reflect.DeepEqual(row.Columns(), tt.expCols) { @@ -597,6 +631,7 @@ func TestBSIGroup_importValue(t *testing.T) { func TestIntField_MinMaxForShard(t *testing.T) { f := OpenField(t, OptFieldTypeInt(-100, 200)) + defer f.Close() options := &ImportOptions{} for i, test := range []struct { @@ -648,12 +683,17 @@ func TestIntField_MinMaxForShard(t *testing.T) { }, } { t.Run(test.name+strconv.Itoa(i), func(t *testing.T) { - tx := &RoaringTx{Field: f.Field} + tx := f.idx.Txf.NewTx(Txo{Write: writable, Index: f.idx, Field: f.Field}) + defer tx.Rollback() if err := f.importValue(tx, test.columnIDs, test.values, options); err != nil { t.Fatalf("test %d, importing values: %s", i, err.Error()) } + panicOn(tx.Commit()) + tx = f.idx.Txf.NewTx(Txo{Write: !writable, Index: f.idx, Field: f.Field}) + defer tx.Rollback() + maxvc, err := f.MaxForShard(tx, 0, nil) if err != nil { t.Fatalf("getting max for shard: %v", err) @@ -753,6 +793,7 @@ func TestDecimalField_MinMaxBoundaries(t *testing.T) { func TestDecimalField_MinMaxForShard(t *testing.T) { f := OpenField(t, OptFieldTypeDecimal(3)) + defer f.Close() options := &ImportOptions{} for i, test := range []struct { @@ -804,12 +845,17 @@ func TestDecimalField_MinMaxForShard(t *testing.T) { }, } { t.Run(test.name+strconv.Itoa(i), func(t *testing.T) { - tx := &RoaringTx{Field: f.Field} + tx := f.idx.Txf.NewTx(Txo{Write: writable, Index: f.idx, Field: f.Field}) + defer tx.Rollback() if err := f.importFloatValue(tx, test.columnIDs, test.values, options); err != nil { t.Fatalf("test %d, importing values: %s", i, err.Error()) } + panicOn(tx.Commit()) + tx = f.idx.Txf.NewTx(Txo{Write: !writable, Index: f.idx, Field: f.Field}) + defer tx.Rollback() + maxvc, err := f.MaxForShard(tx, 0, nil) if err != nil { t.Fatalf("getting max for shard: %v", err) diff --git a/field_test.go b/field_test.go index aeade9fcd..2fc804b7c 100644 --- a/field_test.go +++ b/field_test.go @@ -25,6 +25,8 @@ import ( "github.com/pilosa/pilosa/v2/test" ) +var panicOn = pilosa.PanicOn + // Ensure a field can set & read a bsiGroup value. func TestField_SetValue(t *testing.T) { t.Run("OK", func(t *testing.T) { @@ -35,7 +37,10 @@ func TestField_SetValue(t *testing.T) { if err != nil { t.Fatal(err) } - tx := &pilosa.RoaringTx{Field: f.Field} + + idxPilosa := f.Field.GetIndex() + tx := idxPilosa.NewTx(pilosa.Txo{Write: writable, Index: idxPilosa, Field: f.Field}) + defer tx.Rollback() // Set value on field. if changed, err := f.SetValue(tx, 100, 21); err != nil { @@ -69,7 +74,9 @@ func TestField_SetValue(t *testing.T) { if err != nil { t.Fatal(err) } - tx := &pilosa.RoaringTx{Field: f.Field} + idxP := f.Field.GetIndex() + tx := idxP.Txf.NewTx(pilosa.Txo{Write: writable, Index: idxP, Field: f.Field}) + defer tx.Rollback() // Set value. if changed, err := f.SetValue(tx, 100, 21); err != nil { @@ -103,7 +110,9 @@ func TestField_SetValue(t *testing.T) { if err != nil { t.Fatal(err) } - tx := &pilosa.RoaringTx{Field: f.Field} + idxP := f.Field.GetIndex() + tx := idxP.Txf.NewTx(pilosa.Txo{Write: writable, Index: idxP, Field: f.Field}) + defer tx.Rollback() // Set value. if _, err := f.SetValue(tx, 100, 21); err != pilosa.ErrBSIGroupNotFound { @@ -119,7 +128,9 @@ func TestField_SetValue(t *testing.T) { if err != nil { t.Fatal(err) } - tx := &pilosa.RoaringTx{Field: f.Field} + idxP := f.Field.GetIndex() + tx := idxP.Txf.NewTx(pilosa.Txo{Write: writable, Index: idxP, Field: f.Field}) + defer tx.Rollback() // Set value. if _, err := f.SetValue(tx, 100, 15); err != pilosa.ErrBSIGroupValueTooLow { @@ -135,7 +146,9 @@ func TestField_SetValue(t *testing.T) { if err != nil { t.Fatal(err) } - tx := &pilosa.RoaringTx{Field: f.Field} + idxP := f.Field.GetIndex() + tx := idxP.Txf.NewTx(pilosa.Txo{Write: writable, Index: idxP, Field: f.Field}) + defer tx.Rollback() // Set value. if _, err := f.SetValue(tx, 100, 31); err != pilosa.ErrBSIGroupValueTooHigh { @@ -204,7 +217,9 @@ func TestField_AvailableShards(t *testing.T) { if err != nil { t.Fatal(err) } - tx := &pilosa.RoaringTx{Field: f.Field} + idxP := f.Field.GetIndex() + tx := idxP.Txf.NewTx(pilosa.Txo{Write: writable, Index: idxP, Field: f.Field}) + defer tx.Rollback() // Set values on shards 0 & 2, and verify. if _, err := f.SetBit(tx, 0, 100, nil); err != nil { @@ -214,6 +229,7 @@ func TestField_AvailableShards(t *testing.T) { } else if diff := cmp.Diff(f.AvailableShards().Slice(), []uint64{0, 2}); diff != "" { t.Fatal(diff) } + panicOn(tx.Commit()) // Set remote shards and verify. if err := f.AddRemoteAvailableShards(roaring.NewBitmap(1, 2, 4)); err != nil { @@ -244,7 +260,9 @@ func TestField_ClearValue(t *testing.T) { if err != nil { t.Fatal(err) } - tx := &pilosa.RoaringTx{Field: f.Field} + idxP := f.Field.GetIndex() + tx := idxP.Txf.NewTx(pilosa.Txo{Write: writable, Index: idxP, Field: f.Field}) + defer tx.Rollback() // Set value on field. if changed, err := f.SetValue(tx, 100, 21); err != nil { @@ -252,6 +270,9 @@ func TestField_ClearValue(t *testing.T) { } else if !changed { t.Fatal("expected change") } + panicOn(tx.Commit()) + + tx = idxP.Txf.NewTx(pilosa.Txo{Write: !writable, Index: idxP, Field: f.Field}) // Read value. if value, exists, err := f.Value(tx, 100); err != nil { @@ -261,12 +282,17 @@ func TestField_ClearValue(t *testing.T) { } else if !exists { t.Fatal("expected value to exist") } + tx.Rollback() + tx = idxP.Txf.NewTx(pilosa.Txo{Write: writable, Index: idxP, Field: f.Field}) if changed, err := f.ClearValue(tx, 100); err != nil { t.Fatal(err) } else if !changed { t.Fatal(err) } + panicOn(tx.Commit()) + tx = idxP.Txf.NewTx(pilosa.Txo{Write: !writable, Index: idxP, Field: f.Field}) + defer tx.Rollback() // Read value. if _, exists, err := f.Value(tx, 100); err != nil { diff --git a/fragment.go b/fragment.go index 9734429ed..9dc5b205b 100644 --- a/fragment.go +++ b/fragment.go @@ -108,6 +108,9 @@ type fragment struct { view string shard uint64 + // idx cached to avoid repeatedly looking it up everywhere. + idx *Index + // parent holder, used to find snapshot queue, etc. holder *Holder @@ -180,6 +183,10 @@ type fragment struct { // newFragment returns a new instance of Fragment. func newFragment(holder *Holder, path, index, field, view string, shard uint64, flags byte) *fragment { + idx := holder.Index(index) + if idx == nil { + panic(fmt.Sprintf("holder=%#v but got nil idx back from holder!", holder)) + } f := &fragment{ path: path, index: index, @@ -187,6 +194,7 @@ func newFragment(holder *Holder, path, index, field, view string, shard uint64, view: view, shard: shard, flags: flags, + idx: idx, CacheType: DefaultCacheType, CacheSize: DefaultCacheSize, @@ -250,7 +258,9 @@ func (f *fragment) Open() error { f.checksums = make(map[int][]byte) // Read last bit to determine max row. - return f.calculateMaxRowID() + tx := f.idx.Txf.NewTx(Txo{Write: !writable, Index: f.idx, Fragment: f}) + defer tx.Rollback() + return f.calculateMaxRowID(tx) }(); err != nil { f.close() return err @@ -398,6 +408,15 @@ func (f *fragment) inspectStorage(data []byte, file *os.File, newGen generation, // logic is now mostly in importStorage (reading in a bitmap) and applyStorage // (remapping an existing bitmap to match a new backing store). func (f *fragment) openStorage(unmarshalData bool) error { + + if !f.idx.NeedsSnapshot() { + f.gen = &NopGeneration{} + f.rowCache = &simpleCache{make(map[uint64]*Row)} + f.currdata = struct{ from, to uintptr }{} + f.prevdata = f.currdata + return nil // openStorage becomes a noop under RBF, Badger, etc. + } + // Create a roaring bitmap to serve as storage for the shard. if f.storage == nil { f.storage = roaring.NewFileBitmap() @@ -473,10 +492,16 @@ func (f *fragment) openCache() error { return nil } + tx := f.idx.Txf.NewTx(Txo{Write: !writable, Index: f.idx, Fragment: f}) + defer tx.Rollback() + // Read in all rows by ID. // This will cause them to be added to the cache. for _, id := range pb.IDs { - n := f.storage.CountRange(id*ShardWidth, (id+1)*ShardWidth) + n, err := tx.CountRange(f.index, f.field, f.view, f.shard, id*ShardWidth, (id+1)*ShardWidth) + if err != nil { + return errors.Wrap(err, "CountRange") + } f.cache.BulkAdd(id, n) } f.cache.Invalidate() @@ -583,7 +608,7 @@ func (f *fragment) rowFromStorage(tx Tx, rowID uint64) (*Row, error) { row := &Row{ segments: []rowSegment{{ - data: data, // this data contains BadgerTx data, which should not survive Txn commit. + data: data, shard: f.shard, writable: true, }}, @@ -598,7 +623,11 @@ func (f *fragment) rowFromStorage(tx Tx, rowID uint64) (*Row, error) { func (f *fragment) setBit(tx Tx, rowID, columnID uint64) (changed bool, err error) { f.mu.Lock() defer f.mu.Unlock() - err = f.gen.Transaction(&f.storage.OpWriter, func() error { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err = f.gen.Transaction(wp, func() error { // handle mutux field type if f.mutexVector != nil { if err := f.handleMutex(tx, rowID, columnID); err != nil { @@ -680,7 +709,11 @@ func (f *fragment) unprotectedSetBit(tx Tx, rowID, columnID uint64) (changed boo func (f *fragment) clearBit(tx Tx, rowID, columnID uint64) (changed bool, err error) { f.mu.Lock() defer f.mu.Unlock() - err = f.gen.Transaction(&f.storage.OpWriter, func() error { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err = f.gen.Transaction(wp, func() error { changed, err = f.unprotectedClearBit(tx, rowID, columnID) return err }) @@ -739,7 +772,11 @@ func (f *fragment) unprotectedClearBit(tx Tx, rowID, columnID uint64) (changed b func (f *fragment) setRow(tx Tx, row *Row, rowID uint64) (changed bool, err error) { f.mu.Lock() defer f.mu.Unlock() - err = f.gen.Transaction(&f.storage.OpWriter, func() error { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err = f.gen.Transaction(wp, func() error { changed, err = f.unprotectedSetRow(tx, row, rowID) return err }) @@ -802,7 +839,11 @@ func (f *fragment) unprotectedSetRow(tx Tx, row *Row, rowID uint64) (changed boo func (f *fragment) clearRow(tx Tx, rowID uint64) (changed bool, err error) { f.mu.Lock() defer f.mu.Unlock() - err = f.gen.Transaction(&f.storage.OpWriter, func() error { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err = f.gen.Transaction(wp, func() error { changed, err = f.unprotectedClearRow(tx, rowID) return err }) @@ -845,7 +886,11 @@ func (f *fragment) unprotectedClearRow(tx Tx, rowID uint64) (changed bool, err e // This updates both the on-disk storage and the in-cache bitmap. func (f *fragment) unprotectedClearBlock(tx Tx, block int) (changed bool, err error) { firstRow := uint64(block * HashBlockSize) - err = f.gen.Transaction(&f.storage.OpWriter, func() error { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err = f.gen.Transaction(wp, func() error { var rowChanged bool for rowID := uint64(firstRow); rowID < firstRow+HashBlockSize; rowID++ { if changed, err := f.unprotectedClearRow(tx, rowID); err != nil { @@ -953,8 +998,11 @@ func (f *fragment) positionsForValue(columnID uint64, bitDepth uint, value int64 func (f *fragment) setValueBase(tx Tx, columnID uint64, bitDepth uint, value int64, clear bool) (changed bool, err error) { f.mu.Lock() defer f.mu.Unlock() - - err = f.gen.Transaction(&f.storage.OpWriter, func() error { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err = f.gen.Transaction(wp, func() error { // Convert value to an unsigned representation. uvalue := uint64(value) if value < 0 { @@ -1291,8 +1339,12 @@ func (f *fragment) maxRow(tx Tx, filter *Row) (uint64, uint64, error) { // calculateMaxRowID determines the field's maxRowID value based // on the contents of its storage, and sets the struct argument. -func (f *fragment) calculateMaxRowID() (err error) { - f.maxRowID = f.storage.Max() / ShardWidth +func (f *fragment) calculateMaxRowID(tx Tx) (err error) { + max, err := tx.Max(f.index, f.field, f.view, f.shard) + if err != nil { + return err + } + f.maxRowID = max / ShardWidth return nil } @@ -1484,7 +1536,6 @@ func (f *fragment) rangeGT(tx Tx, bitDepth uint, predicate int64, allowEquality if err != nil { return nil, err } - // Create predicate without sign bit. upredicate := absInt64(predicate) @@ -1492,7 +1543,6 @@ func (f *fragment) rangeGT(tx Tx, bitDepth uint, predicate int64, allowEquality if err != nil { return nil, err } - switch { case predicate == 0 && !allowEquality: // Match all positive numbers except zero. @@ -2217,7 +2267,11 @@ func (f *fragment) bulkImportStandard(tx Tx, rowIDs, columnIDs []uint64, options // operations to the op log. func (f *fragment) importPositions(tx Tx, set, clear []uint64, rowSet map[uint64]struct{}) error { //tx.AddN() - err := f.gen.Transaction(&f.storage.OpWriter, func() error { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err := f.gen.Transaction(wp, func() error { // segfault if len(set) > 0 { f.stats.Count(MetricImportingN, int64(len(set)), 1) @@ -2453,14 +2507,18 @@ func (f *fragment) unprotectedImportRoaring(ctx context.Context, tx Tx, data []b span, ctx := tracing.StartSpanFromContext(ctx, "importRoaring.ImportRoaringBits") var changed int var rowSet map[uint64]int - err := f.gen.Transaction(&f.storage.OpWriter, func() (err error) { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err := f.gen.Transaction(wp, func() (err error) { var rit roaring.RoaringIterator rit, err = roaring.NewRoaringIterator(data) if err != nil { return err } - changed, rowSet, err = tx.ImportRoaringBits(f.index, f.field, f.view, f.shard, rit, clear, true, rowSize) + changed, rowSet, err = tx.ImportRoaringBits(f.index, f.field, f.view, f.shard, rit, clear, true, rowSize, nil) return err }) @@ -2550,6 +2608,9 @@ func track(start time.Time, message string, stats stats.StatsClient, logger logg // snapshot does the actual snapshot operation. it does not check or care // about f.snapshotPending. func (f *fragment) snapshot() (err error) { + if !f.idx.NeedsSnapshot() { + return nil + } if !f.open { return errors.New("snapshot request on closed fragment") } @@ -2688,31 +2749,16 @@ func (f *fragment) WriteTo(w io.Writer) (n int64, err error) { return 0, nil } +// used in shipping the slices across the network for a resize. func (f *fragment) writeStorageToArchive(tw *tar.Writer) error { - // Open separate file descriptor to read from. - file, err := os.Open(f.path) + + tx := f.idx.Txf.NewTx(Txo{Write: !writable, Index: f.idx}) + defer tx.Rollback() + file, sz, err := tx.RoaringBitmapReader(f.index, f.field, f.view, f.shard, f.path) if err != nil { - return errors.Wrap(err, "opening file") - } - defer file.Close() - - // Retrieve the current file size under lock so we don't read - // while an operation is appending to the end. - var sz int64 - if err := func() error { - f.mu.Lock() - defer f.mu.Unlock() - - fi, err := file.Stat() - if err != nil { - return errors.Wrap(err, "statting") - } - sz = fi.Size() - - return nil - }(); err != nil { return err } + defer file.Close() // Write archive header. if err := tw.WriteHeader(&tar.Header{ @@ -2779,9 +2825,15 @@ func (f *fragment) ReadFrom(r io.Reader) (n int64, err error) { // Process file based on file name. switch hdr.Name { case "data": - if err := f.readStorageFromArchive(tr); err != nil { + idx := f.holder.Index(f.index) + tx := idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: f}) + defer tx.Rollback() + if err := f.fillFragmentFromArchive(tx, tr); err != nil { return 0, errors.Wrap(err, "reading storage") } + if err := tx.Commit(); err != nil { + return 0, errors.Wrap(err, "Commit after tx.ReadFragmentFromArchive") + } case "cache": if err := f.readCacheFromArchive(tr); err != nil { return 0, errors.Wrap(err, "reading cache") @@ -2794,7 +2846,40 @@ func (f *fragment) ReadFrom(r io.Reader) (n int64, err error) { return 0, nil } +// should be morally equivalent to fragment.readStorageFromArchive() +// below for RoaringTx, but also work on any Tx because it uses +// tx.ImportRoaringBits(). +func (f *fragment) fillFragmentFromArchive(tx Tx, r io.Reader) error { + + // this is reading from inside a tarball, so definitely no need + // to close it here. + data, err := ioutil.ReadAll(r) + if err != nil { + return errors.Wrap(err, "fillFragmentFromArchive ioutil.ReadAll(r)") + } + if len(data) == 0 { + return nil + } + + // For reference, compare to what fragment.go:313 fragment.importStorage() does. + + clear := false + log := false + rowSize := uint64(0) + itr, err := roaring.NewRoaringIterator(data) + if err != nil { + return errors.Wrap(err, "fillFragmentFromArchive NewRoaringIterator") + } + changed, rowSet, err := tx.ImportRoaringBits(f.index, f.field, f.view, f.shard, itr, clear, log, rowSize, data) + _, _ = changed, rowSet + if err != nil { + return errors.Wrap(err, "fillFragmentFromArchive ImportRoaringBits") + } + return nil +} + func (f *fragment) readStorageFromArchive(r io.Reader) error { + // Create a temporary file to copy into. path := f.path + copyExt file, err := os.Create(path) @@ -2808,6 +2893,12 @@ func (f *fragment) readStorageFromArchive(r io.Reader) error { return errors.Wrap(err, "copying") } + // TODO(jea): isn't this next Rename a file handle leak? + // try closing first + if err := f.closeStorage(); err != nil { + return errors.Wrap(err, "closeStorage-prior-to-Rename-and-openStorage") + } + // Move snapshot to data file location. if err := os.Rename(path, f.path); err != nil { return errors.Wrap(err, "renaming") diff --git a/fragment_internal_test.go b/fragment_internal_test.go index dca084b5a..ff0de833f 100644 --- a/fragment_internal_test.go +++ b/fragment_internal_test.go @@ -24,6 +24,7 @@ import ( "math" "math/rand" "os" + "path/filepath" "reflect" "runtime" "runtime/debug" @@ -445,7 +446,7 @@ func TestFragment_SetValue(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set value. if changed, err := f.setValue(tx, 100, 10, 20); err != nil { @@ -481,7 +482,7 @@ func TestFragment_SetValue(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set values. m := make(map[uint64]int64) @@ -538,12 +539,11 @@ func TestFragment_Sum(t *testing.T) { const bitDepth = 16 f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") - _ = idx defer f.Clean(t) // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set values. vals := []struct { @@ -562,6 +562,10 @@ func TestFragment_Sum(t *testing.T) { } } + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + t.Run("NoFilter", func(t *testing.T) { if sum, n, err := f.sum(tx, nil, bitDepth); err != nil { t.Fatal(err) @@ -582,10 +586,19 @@ func TestFragment_Sum(t *testing.T) { } }) + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: f}) + defer tx.Rollback() + // verify that clearValue clears values if _, err := f.clearValue(tx, 1000, bitDepth, 23); err != nil { t.Fatal(err) } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + t.Run("ClearValue", func(t *testing.T) { if sum, n, err := f.sum(tx, nil, bitDepth); err != nil { t.Fatal(err) @@ -602,12 +615,11 @@ func TestFragment_MinMax(t *testing.T) { const bitDepth = 16 f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") - _ = idx defer f.Clean(t) // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set values. if _, err := f.setValue(tx, 1000, bitDepth, 382); err != nil { @@ -626,6 +638,12 @@ func TestFragment_MinMax(t *testing.T) { t.Fatal(err) } + panicOn(tx.Commit()) + + // the new tx is shared by Min/Max below. + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + t.Run("Min", func(t *testing.T) { tests := []struct { filter *Row @@ -689,7 +707,7 @@ func TestFragment_Range(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set values. if _, err := f.setValue(tx, 1000, bitDepth, 382); err != nil { @@ -716,7 +734,8 @@ func TestFragment_Range(t *testing.T) { defer f.Clean(t) // Obtain transaction. - tx := &RoaringTx{fragment: f} + tx := f.txTestingOnly + defer tx.Rollback() // Set values. if _, err := f.setValue(tx, 1000, 1, 0); err != nil { @@ -745,7 +764,7 @@ func TestFragment_Range(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set values. if _, err := f.setValue(tx, 1000, bitDepth, 382); err != nil { @@ -773,7 +792,7 @@ func TestFragment_Range(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set values. if _, err := f.setValue(tx, 1000, bitDepth, 382); err != nil { @@ -826,7 +845,7 @@ func TestFragment_Range(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() if _, err := f.setValue(tx, 1, 1, 1); err != nil { t.Fatal(err) @@ -846,7 +865,7 @@ func TestFragment_Range(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() if _, err := f.setValue(tx, 1, 2, 3); err != nil { t.Fatal(err) @@ -868,7 +887,7 @@ func TestFragment_Range(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set values. if _, err := f.setValue(tx, 1000, bitDepth, 382); err != nil { @@ -921,7 +940,7 @@ func TestFragment_Range(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() if _, err := f.setValue(tx, 1, 2, 0); err != nil { t.Fatal(err) @@ -942,7 +961,8 @@ func TestFragment_Range(t *testing.T) { defer f.Clean(t) // Obtain transaction. - tx := &RoaringTx{fragment: f} + tx := f.txTestingOnly + defer tx.Rollback() if _, err := f.setValue(tx, 1, 2, 0); err != nil { t.Fatal(err) @@ -964,7 +984,7 @@ func TestFragment_Range(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set values. if _, err := f.setValue(tx, 1000, bitDepth, 382); err != nil { @@ -1016,7 +1036,8 @@ func TestFragment_Range(t *testing.T) { defer f.Clean(t) // Obtain transaction. - tx := &RoaringTx{fragment: f} + tx := f.txTestingOnly + defer tx.Rollback() if _, err := f.setValue(tx, 1, 64, 0xf0); err != nil { t.Fatal(err) @@ -1037,7 +1058,7 @@ func TestFragment_Range(t *testing.T) { func benchmarkSetValues(b *testing.B, bitDepth uint, f *fragment, cfunc func(uint64) uint64) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() column := uint64(0) for i := 0; i < b.N; i++ { @@ -1074,7 +1095,7 @@ func BenchmarkFragment_SetValue(b *testing.B) { func benchmarkImportValues(b *testing.B, bitDepth uint, f *fragment, cfunc func(uint64) uint64) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() column := uint64(0) b.StopTimer() @@ -1145,7 +1166,7 @@ func BenchmarkFragment_RepeatedSmallImports(b *testing.B) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() err := f.importRoaringT(tx, getZipfRowsSliceRoaring(uint64(numRows), 1, 0, ShardWidth), false) if err != nil { @@ -1162,6 +1183,7 @@ func BenchmarkFragment_RepeatedSmallImports(b *testing.B) { b.Fatalf("doing small bulk import: %v", err) } } + tx.Rollback() // don't exhaust the Tx space under b.N iterations. } }) } @@ -1239,7 +1261,7 @@ func BenchmarkFragment_RepeatedSmallValueImports(b *testing.B) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() err := f.importValue(tx, initialCols, initialVals, 21, false) if err != nil { @@ -1257,6 +1279,7 @@ func BenchmarkFragment_RepeatedSmallValueImports(b *testing.B) { b.Fatalf("importing values: %v", err) } } + tx.Rollback() // don't exhaust the Tx over the b.N iterations. } }) } @@ -1268,12 +1291,11 @@ func BenchmarkFragment_RepeatedSmallValueImports(b *testing.B) { // Ensure a fragment can snapshot correctly. func TestFragment_Snapshot(t *testing.T) { f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") - _ = idx defer f.Clean(t) // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set and then clear bits on the fragment. if _, err := f.setBit(tx, 1000, 1); err != nil { @@ -1283,6 +1305,9 @@ func TestFragment_Snapshot(t *testing.T) { } else if _, err := f.clearBit(tx, 1000, 1); err != nil { t.Fatal(err) } + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() // Snapshot bitmap and verify data. if err := f.Snapshot(); err != nil { @@ -1307,7 +1332,7 @@ func TestFragment_ForEachBit(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set bits on the fragment. if _, err := f.setBit(tx, 100, 20); err != nil { @@ -1341,7 +1366,7 @@ func TestFragment_Top(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set bits on the rows 100, 101, & 102. f.mustSetBits(tx, 100, 1, 3, 200) @@ -1364,12 +1389,11 @@ func TestFragment_Top(t *testing.T) { // Ensure a fragment can filter rows when retrieving the top n rows. func TestFragment_Top_Filter(t *testing.T) { f, idx := mustOpenFragment("i", "f", viewStandard, 0, CacheTypeRanked) - _ = idx defer f.Clean(t) // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set bits on the rows 100, 101, & 102. f.mustSetBits(tx, 100, 1, 3, 200) @@ -1386,6 +1410,10 @@ func TestFragment_Top_Filter(t *testing.T) { t.Fatalf("setAttrs: %v", err) } + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + // Retrieve top rows. if pairs, err := f.top(tx, topOptions{ N: 2, @@ -1410,7 +1438,7 @@ func TestFragment_TopN_Intersect(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Create an intersecting input row. src := NewRow(1, 2, 3) @@ -1446,7 +1474,7 @@ func TestFragment_TopN_Intersect_Large(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Create an intersecting input row. src := NewRow( @@ -1499,7 +1527,7 @@ func TestFragment_TopN_IDs(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set bits on various rows. f.mustSetBits(tx, 100, 1, 2, 3) @@ -1525,7 +1553,7 @@ func TestFragment_TopN_NopCache(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set bits on various rows. f.mustSetBits(tx, 100, 1, 2, 3) @@ -1700,7 +1728,7 @@ func TestFragment_Blocks_Empty(t *testing.T) { defer f.Clean(t) // Obtain transaction. - tx := f.txTestingOnly //&RoaringTx{fragment: f} + tx := f.txTestingOnly defer tx.Rollback() // Set bits on a different block. @@ -1727,7 +1755,7 @@ func TestFragment_LRUCache_Persistence(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set bits on the fragment. for i := uint64(0); i < 1000; i++ { @@ -1781,6 +1809,7 @@ func TestFragment_RankCache_Persistence(t *testing.T) { // Obtain transaction. tx := index.Txf.NewTx(Txo{Write: writable, Index: index, Fragment: f}) + defer tx.Rollback() // Set bits on the fragment. for i := uint64(0); i < 1000; i++ { @@ -1789,6 +1818,10 @@ func TestFragment_RankCache_Persistence(t *testing.T) { } } + panicOn(tx.Commit()) + tx = index.Txf.NewTx(Txo{Write: !writable, Index: index, Fragment: f}) + defer tx.Rollback() + // Verify correct cache type and size. if cache, ok := f.cache.(*rankCache); !ok { t.Fatalf("unexpected cache: %T", f.cache) @@ -1819,7 +1852,8 @@ func TestFragment_WriteTo_ReadFrom(t *testing.T) { defer f0.Clean(t) // Obtain transaction. - tx := &RoaringTx{fragment: f0} + tx := idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: f0}) + defer tx.Rollback() // Set and then clear bits on the fragment. if _, err := f0.setBit(tx, 1000, 1); err != nil { @@ -1847,7 +1881,7 @@ func TestFragment_WriteTo_ReadFrom(t *testing.T) { _ = idx defer f1.Clean(t) - if rn, err := f1.ReadFrom(&buf); err != nil { + if rn, err := f1.ReadFrom(&buf); err != nil { // eventually calls fragment.fillFragmentFromArchive t.Fatal(err) } else if wn != rn { t.Fatalf("read/write byte count mismatch: wn=%d, rn=%d", wn, rn) @@ -1898,13 +1932,12 @@ func BenchmarkFragment_Blocks(b *testing.B) { func BenchmarkFragment_IntersectionCount(b *testing.B) { f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") - _ = idx defer f.Clean(b) f.MaxOpN = math.MaxInt32 // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Generate some intersecting data. for i := 0; i < 10000; i += 2 { @@ -1918,6 +1951,10 @@ func BenchmarkFragment_IntersectionCount(b *testing.B) { } } + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + // Snapshot to disk before benchmarking. if err := f.Snapshot(); err != nil { b.Fatal(err) @@ -1939,7 +1976,7 @@ func TestFragment_Tanimoto(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() src := NewRow(1, 2, 3) @@ -1967,7 +2004,7 @@ func TestFragment_Zero_Tanimoto(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() src := NewRow(1, 2, 3) @@ -1997,7 +2034,7 @@ func TestFragment_Snapshot_Run(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set bits on the fragment. for i := uint64(1); i < 3; i++ { @@ -2029,7 +2066,7 @@ func TestFragment_SetMutex(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() var cols []uint64 @@ -2148,7 +2185,7 @@ func TestFragment_ImportSet(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set import. err := f.bulkImport(tx, test.setRowIDs, test.setColIDs, &ImportOptions{}) @@ -2181,6 +2218,140 @@ func TestFragment_ImportSet(t *testing.T) { } } +func TestFragment_ImportSet_WithTxCommit(t *testing.T) { + tests := []struct { + setRowIDs []uint64 + setColIDs []uint64 + setExp map[uint64][]uint64 + clearRowIDs []uint64 + clearColIDs []uint64 + clearExp map[uint64][]uint64 + }{ + { + []uint64{1, 1, 1, 1}, + []uint64{0, 1, 2, 3}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + }, + []uint64{}, + []uint64{}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + }, + }, + { + []uint64{1, 1, 1, 1, 2, 2, 2, 2}, + []uint64{0, 1, 2, 3, 0, 1, 2, 3}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + 2: {0, 1, 2, 3}, + }, + []uint64{1, 1, 2}, + []uint64{1, 2, 3}, + map[uint64][]uint64{ + 1: {0, 3}, + 2: {0, 1, 2}, + }, + }, + { + []uint64{1, 1, 1, 1, 2}, + []uint64{0, 1, 2, 3, 1}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + 2: {1}, + }, + []uint64{1, 1, 1, 1}, + []uint64{0, 1, 2, 3}, + map[uint64][]uint64{ + 1: {}, + 2: {1}, + }, + }, + { + []uint64{1, 1, 1, 1, 2, 2, 1}, + []uint64{0, 1, 2, 3, 1, 8, 1}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + 2: {1, 8}, + }, + []uint64{1, 1}, + []uint64{0, 0}, + map[uint64][]uint64{ + 1: {1, 2, 3}, + 2: {1, 8}, + }, + }, + { + []uint64{1, 2, 3}, + []uint64{8, 8, 8}, + map[uint64][]uint64{ + 1: {8}, + 2: {8}, + 3: {8}, + }, + []uint64{1, 2, 3}, + []uint64{9, 9, 9}, + map[uint64][]uint64{ + 1: {8}, + 2: {8}, + 3: {8}, + }, + }, + } + + for i, test := range tests { + t.Run(fmt.Sprintf("importset%d", i), func(t *testing.T) { + f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") + _ = idx + defer f.Clean(t) + + // Obtain transaction. + tx := f.txTestingOnly + defer tx.Rollback() + + // Set import. + err := f.bulkImport(tx, test.setRowIDs, test.setColIDs, &ImportOptions{}) + if err != nil { + t.Fatalf("bulk importing ids: %v", err) + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Check for expected results. + for k, v := range test.setExp { + cols := f.mustRow(tx, k).Columns() + if !reflect.DeepEqual(cols, v) { + t.Fatalf("expected: %v, but got: %v", v, cols) + } + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Clear import. + err = f.bulkImport(tx, test.clearRowIDs, test.clearColIDs, &ImportOptions{Clear: true}) + if err != nil { + t.Fatalf("bulk clearing ids: %v", err) + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Check for expected results. + for k, v := range test.clearExp { + cols := f.mustRow(tx, k).Columns() + if !reflect.DeepEqual(cols, v) { + t.Fatalf("expected: %v, but got: %v", v, cols) + } + } + }) + } +} + func TestFragment_ConcurrentImport(t *testing.T) { t.Run("bulkImportStandard", func(t *testing.T) { f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") @@ -2189,7 +2360,7 @@ func TestFragment_ConcurrentImport(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() eg := errgroup.Group{} eg.Go(func() error { return f.bulkImportStandard(tx, []uint64{1, 2}, []uint64{1, 2}, &ImportOptions{}) }) @@ -2291,7 +2462,7 @@ func TestFragment_ImportMutex(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set import. err := f.bulkImport(tx, test.setRowIDs, test.setColIDs, &ImportOptions{}) @@ -2324,6 +2495,141 @@ func TestFragment_ImportMutex(t *testing.T) { } } +// Ensure a fragment can import mutually exclusive values. +// Now with Commits in the middle. +func TestFragment_ImportMutex_WithTxCommit(t *testing.T) { + tests := []struct { + setRowIDs []uint64 + setColIDs []uint64 + setExp map[uint64][]uint64 + clearRowIDs []uint64 + clearColIDs []uint64 + clearExp map[uint64][]uint64 + }{ + { + []uint64{1, 1, 1, 1}, + []uint64{0, 1, 2, 3}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + }, + []uint64{}, + []uint64{}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + }, + }, + { + []uint64{1, 1, 1, 1, 2, 2, 2, 2}, + []uint64{0, 1, 2, 3, 0, 1, 2, 3}, + map[uint64][]uint64{ + 1: {}, + 2: {0, 1, 2, 3}, + }, + []uint64{1, 1, 2}, + []uint64{1, 2, 3}, + map[uint64][]uint64{ + 1: {}, + 2: {0, 1, 2}, + }, + }, + { + []uint64{1, 1, 1, 1, 2}, + []uint64{0, 1, 2, 3, 1}, + map[uint64][]uint64{ + 1: {0, 2, 3}, + 2: {1}, + }, + []uint64{1, 1, 1, 1}, + []uint64{0, 1, 2, 3}, + map[uint64][]uint64{ + 1: {}, + 2: {1}, + }, + }, + { + []uint64{1, 1, 1, 1, 2, 2, 1}, + []uint64{0, 1, 2, 3, 1, 8, 1}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + 2: {8}, + }, + []uint64{1, 1}, + []uint64{0, 0}, + map[uint64][]uint64{ + 1: {1, 2, 3}, + 2: {8}, + }, + }, + { + []uint64{1, 2, 3}, + []uint64{8, 8, 8}, + map[uint64][]uint64{ + 1: {}, + 2: {}, + 3: {8}, + }, + []uint64{1, 2, 3}, + []uint64{9, 9, 9}, + map[uint64][]uint64{ + 1: {}, + 2: {}, + 3: {8}, + }, + }, + } + + for i, test := range tests { + t.Run(fmt.Sprintf("importmutex%d", i), func(t *testing.T) { + f, idx := mustOpenMutexFragment("i", "f", viewStandard, 0, "") + defer f.Clean(t) + + // Obtain transaction. + tx := f.txTestingOnly + defer tx.Rollback() + + // Set import. + err := f.bulkImport(tx, test.setRowIDs, test.setColIDs, &ImportOptions{}) + if err != nil { + t.Fatalf("bulk importing ids: %v", err) + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Check for expected results. + for k, v := range test.setExp { + cols := f.mustRow(tx, k).Columns() + if !reflect.DeepEqual(cols, v) { + t.Fatalf("row: %d, expected: %v, but got: %v", k, v, cols) + } + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Clear import. + err = f.bulkImport(tx, test.clearRowIDs, test.clearColIDs, &ImportOptions{Clear: true}) + if err != nil { + t.Fatalf("bulk clearing ids: %v", err) + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Check for expected results. + for k, v := range test.clearExp { + cols := f.mustRow(tx, k).Columns() + if !reflect.DeepEqual(cols, v) { + t.Fatalf("row: %d expected: %v, but got: %v", k, v, cols) + } + } + }) + } +} + // Ensure a fragment can import bool values. func TestFragment_ImportBool(t *testing.T) { tests := []struct { @@ -2415,7 +2721,7 @@ func TestFragment_ImportBool(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // Set import. err := f.bulkImport(tx, test.setRowIDs, test.setColIDs, &ImportOptions{}) @@ -2448,6 +2754,142 @@ func TestFragment_ImportBool(t *testing.T) { } } +// Ensure a fragment can import bool values, with Commits in between writes and reads. +func TestFragment_ImportBool_WithTxCommit(t *testing.T) { + tests := []struct { + setRowIDs []uint64 + setColIDs []uint64 + setExp map[uint64][]uint64 + clearRowIDs []uint64 + clearColIDs []uint64 + clearExp map[uint64][]uint64 + }{ + { + []uint64{1, 1, 1, 1}, + []uint64{0, 1, 2, 3}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + }, + []uint64{}, + []uint64{}, + map[uint64][]uint64{ + 1: {0, 1, 2, 3}, + }, + }, + { + []uint64{0, 0, 0, 0, 1, 1, 1, 1}, + []uint64{0, 1, 2, 3, 0, 1, 2, 3}, + map[uint64][]uint64{ + 0: {}, + 1: {0, 1, 2, 3}, + }, + []uint64{1, 1, 2}, + []uint64{1, 2, 3}, + map[uint64][]uint64{ + 0: {}, + 1: {0, 3}, + 2: {}, + }, + }, + { + []uint64{0, 0, 0, 0, 1}, + []uint64{0, 1, 2, 3, 1}, + map[uint64][]uint64{ + 0: {0, 2, 3}, + 1: {1}, + }, + []uint64{1, 1, 1, 1}, + []uint64{0, 1, 2, 3}, + map[uint64][]uint64{ + 0: {0, 2, 3}, + 1: {}, + }, + }, + { + []uint64{1, 1, 1, 1, 0, 0, 1}, + []uint64{0, 1, 2, 3, 1, 8, 1}, + map[uint64][]uint64{ + 0: {8}, + 1: {0, 1, 2, 3}, + }, + []uint64{1, 1}, + []uint64{0, 0}, + map[uint64][]uint64{ + 0: {8}, + 1: {1, 2, 3}, + }, + }, + { + []uint64{0, 1, 2}, + []uint64{8, 8, 8}, + map[uint64][]uint64{ + 0: {}, + 1: {}, // This isn't {8} because fragment doesn't validate bool values. + 2: {8}, + }, + []uint64{1, 2, 3}, + []uint64{9, 9, 9}, + map[uint64][]uint64{ + 0: {}, + 1: {}, + 2: {8}, + }, + }, + } + + for i, test := range tests { + t.Run(fmt.Sprintf("importmutex%d", i), func(t *testing.T) { + f, idx := mustOpenBoolFragment("i", "f", viewStandard, 0, "") + _ = idx + defer f.Clean(t) + + // Obtain transaction. + tx := f.txTestingOnly + defer tx.Rollback() + + // Set import. + err := f.bulkImport(tx, test.setRowIDs, test.setColIDs, &ImportOptions{}) + if err != nil { + t.Fatalf("bulk importing ids: %v", err) + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Check for expected results. + for k, v := range test.setExp { + cols := f.mustRow(tx, k).Columns() + if !reflect.DeepEqual(cols, v) { + t.Fatalf("expected: %v, but got: %v", v, cols) + } + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Clear import. + err = f.bulkImport(tx, test.clearRowIDs, test.clearColIDs, &ImportOptions{Clear: true}) + if err != nil { + t.Fatalf("bulk importing ids: %v", err) + } + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + // Check for expected results. + for k, v := range test.clearExp { + cols := f.mustRow(tx, k).Columns() + if !reflect.DeepEqual(cols, v) { + t.Fatalf("expected: %v, but got: %v", v, cols) + } + } + }) + } +} + func BenchmarkFragment_Snapshot(b *testing.B) { if *FragmentPath == "" { b.Skip("no fragment specified") @@ -2478,10 +2920,6 @@ func BenchmarkFragment_FullSnapshot(b *testing.B) { _ = idx defer f.Clean(b) - // Obtain transaction. - tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME - // Generate some intersecting data. maxX := ShardWidth / 2 sz := maxX @@ -2499,9 +2937,14 @@ func BenchmarkFragment_FullSnapshot(b *testing.B) { val += 2 i++ } + + tx := idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + if err := f.bulkImport(tx, rows, cols, options); err != nil { b.Fatalf("Error Building Sample: %s", err) } + panicOn(tx.Commit()) if row > max { max = row } @@ -2544,14 +2987,16 @@ func BenchmarkFragment_Import(b *testing.B) { copy(rowsUse, rows) copy(colsUse, cols) f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") - _ = idx // Obtain transaction. + _ = idx + // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() b.StartTimer() if err := f.bulkImport(tx, rowsUse, colsUse, options); err != nil { b.Errorf("Error Building Sample: %s", err) } b.StopTimer() + tx.Rollback() f.Clean(b) } } @@ -2645,6 +3090,10 @@ func BenchmarkImportRoaringConcurrent(b *testing.B) { } } func BenchmarkImportRoaringUpdateConcurrent(b *testing.B) { + skipForBadger := os.Getenv("PILOSA_TXSRC") == "badger" + if skipForBadger { + b.Skip("skip for badger") + } if testing.Short() { b.SkipNow() } @@ -2664,6 +3113,7 @@ func BenchmarkImportRoaringUpdateConcurrent(b *testing.B) { // the cost of actually doing the op log for the large initial data set // is excessive. force storage into snapshotted state, then use import // to generate an op log and/or snapshot. + // note: skipped for badger, above. _, _, err := frags[j].storage.ImportRoaringBits(data, false, false, 0) if err != nil { b.Fatalf("importing roaring: %v", err) @@ -2719,7 +3169,7 @@ func BenchmarkImportStandard(b *testing.B) { _ = idx // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() b.StartTimer() err := f.bulkImport(tx, rowIDs, columnIDs, &ImportOptions{}) @@ -2727,6 +3177,7 @@ func BenchmarkImportStandard(b *testing.B) { b.Errorf("import error: %v", err) } b.StopTimer() + tx.Rollback() f.Clean(b) } }) @@ -2757,7 +3208,7 @@ func BenchmarkImportRoaringUpdate(b *testing.B) { // to generate an op log and/or snapshot. itr, err := roaring.NewRoaringIterator(data) panicOn(err) - _, _, err = tx.ImportRoaringBits(f.index, f.field, f.view, f.shard, itr, false, false, 0) + _, _, err = tx.ImportRoaringBits(f.index, f.field, f.view, f.shard, itr, false, false, 0, nil) if err != nil { b.Errorf("import error: %v", err) } @@ -2877,14 +3328,21 @@ func BenchmarkImportIntoLargeFragment(b *testing.B) { } origF.Close() fi.Close() - nf := newFragment(NewHolder(DefaultPartitionN), fi.Name(), "i", "f", viewStandard, 0, 0) + + h := NewHolder(DefaultPartitionN) + h.Path = fi.Name() + idx, err := h.CreateIndex("i", IndexOptions{}) + panicOn(err) + + nf := newFragment(h, fi.Name(), "i", "f", viewStandard, 0, 0) err = nf.Open() if err != nil { b.Fatalf("opening fragment: %v", err) } // Obtain transaction. - tx := &RoaringTx{fragment: nf} + tx := idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: nf}) + defer tx.Rollback() copy(rows, rowsOrig) copy(cols, colsOrig) @@ -2894,7 +3352,7 @@ func BenchmarkImportIntoLargeFragment(b *testing.B) { if err != nil { b.Fatalf("bulkImport: %v", err) } - + panicOn(tx.Commit()) nf.Clean(b) } } @@ -2924,12 +3382,14 @@ func BenchmarkImportRoaringIntoLargeFragment(b *testing.B) { th := newTestHolder() idx := fragTestMustOpenIndex("i", th, IndexOptions{}) + if th.NeedsSnapshot() { + th.SnapshotQueue = newSnapshotQueue(1, 1, nil) + } nf := newFragment(th, fi.Name(), "i", "f", viewStandard, 0, 0) tx := idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: nf}) defer tx.Rollback() - //nf := newFragment(NewHolder(DefaultPartitionN), fi.Name(), "i", "f", viewStandard, 0, 0) err = nf.Open() if err != nil { b.Fatalf("opening fragment: %v", err) @@ -3094,6 +3554,7 @@ func BenchmarkFileWrite(b *testing.B) { ///////////////////////////////////////////////////////////////////// +// not called under badger b/c f.idx.NeedsSnapshot() in Clean() avoids it. func (f *fragment) sanityCheck(t testing.TB) { newBM := roaring.NewFileBitmap() file, err := os.Open(f.path) @@ -3110,6 +3571,7 @@ func (f *fragment) sanityCheck(t testing.TB) { t.Fatalf("sanityCheck couldn't unmarshal fragment %s: %v", f.path, err) } // Refactor fragment.storage + // note: not called for badger, see above. if equal, reason := newBM.BitwiseEqual(f.storage); !equal { t.Fatalf("fragment %s: unmarshalled bitmap different: %v", f.path, reason) } @@ -3122,25 +3584,35 @@ func (f *fragment) Clean(t testing.TB) { // check or else, in some cases, the background snapshot queue // can decide to pick it up. func() { + // should we skip snapshot queue stuff under badger/rbf? defer f.mu.Unlock() - err := defaultSnapshotQueue.Await(f) - if err != nil { - t.Fatalf("snapshot failed before sanity check: %v", err) - } - f.sanityCheck(t) - if f.storage != nil && f.storage.Source != nil { - if f.storage.Source.Dead() { - t.Fatalf("cleaning up fragment %s, source %s, source already dead", f.path, f.storage.Source.ID()) + + // badger doesn't need snapshot, so this stuff is skipped. + // The snapshot queue stuff doesn't work under badger. + if f.idx.NeedsSnapshot() { + err := defaultSnapshotQueue.Await(f) + if err != nil { + t.Fatalf("snapshot failed before sanity check: %v", err) + } + f.sanityCheck(t) + if f.storage != nil && f.storage.Source != nil { + if f.storage.Source.Dead() { + t.Fatalf("cleaning up fragment %s, source %s, source already dead", f.path, f.storage.Source.ID()) + } } } }() if f.txTestingOnly != nil { f.txTestingOnly.Rollback() + panicOn(f.idx.Txf.CloseIndex(f.idx)) } errc := f.Close() // prevent double-closes of generation during testing. f.gen = nil - errf := os.Remove(f.path) + var errf error + if FileExists(f.path) { + errf = os.Remove(f.path) // remove /var/folders/2x/hm9gp5ys3k9gmm5f_vzm_6wc0000gn/T/pilosa-index-768377904/i/f/views/standard/fragments/0: no such file or directory + } errp := os.Remove(f.cachePath()) if errc != nil || errf != nil { t.Fatal("cleaning up fragment: ", errc, errf, errp) @@ -3180,15 +3652,9 @@ func mustOpenBSIFragment(index, field, view string, shard uint64) (*fragment, *I return mustOpenFragmentFlags(index, field, view, shard, "", 1) } -var testHolder = NewHolder(DefaultPartitionN) - -func init() { - testHolder.SnapshotQueue = newSnapshotQueue(1, 1, nil) -} - func newTestHolder() *Holder { h := NewHolder(DefaultPartitionN) - h.SnapshotQueue = newSnapshotQueue(1, 1, nil) + //h.SnapshotQueue = newSnapshotQueue(1, 1, nil) return h } @@ -3199,13 +3665,15 @@ func fragTestMustOpenIndex(index string, holder *Holder, opt IndexOptions) *Inde panic(err) } holder.Path = path + holder.mu.Lock() idx, err := holder.createIndex(index, opt) + holder.mu.Unlock() panicOn(err) idx.keys = opt.Keys idx.trackExistence = opt.TrackExistence - if err := idx.Open(); err != nil { + if err := idx.Open(false); err != nil { panic(err) } return idx @@ -3226,13 +3694,14 @@ func mustOpenFragmentFlags(index, field, view string, shard uint64, cacheType st // new: th := newTestHolder() idx := fragTestMustOpenIndex(index, th, IndexOptions{}) + if th.NeedsSnapshot() { + th.SnapshotQueue = newSnapshotQueue(1, 1, nil) + } f := newFragment(th, file.Name(), index, field, view, shard, flags) tx := idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: f}) f.txTestingOnly = tx - //old: f := newFragment(testHolder, file.Name(), index, field, view, shard, flags) - f.CacheType = cacheType f.RowAttrStore = &memAttrStore{ store: make(map[uint64]map[string]interface{}), @@ -3292,9 +3761,9 @@ func TestFragment_RowsIteration(t *testing.T) { t.Run("firstContainer", func(t *testing.T) { f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") _ = idx - defer f.Clean(t) tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() + defer f.Clean(t) expectedAll := make([]uint64, 0) expectedOdd := make([]uint64, 0) @@ -3326,9 +3795,9 @@ func TestFragment_RowsIteration(t *testing.T) { t.Run("secondRow", func(t *testing.T) { f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") _ = idx - defer f.Clean(t) tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() + defer f.Clean(t) expected := []uint64{1, 2} if _, err := f.setBit(tx, 1, 66000); err != nil { @@ -3357,9 +3826,9 @@ func TestFragment_RowsIteration(t *testing.T) { t.Run("combinations", func(t *testing.T) { f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") _ = idx - defer f.Clean(t) tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() + defer f.Clean(t) expectedRows := make([]uint64, 0) for r := uint64(1); r < uint64(10000); r += 250 { @@ -3602,9 +4071,9 @@ func TestFragmentRowIterator(t *testing.T) { t.Run("basic", func(t *testing.T) { f, idx := mustOpenFragment("i", "f", "v", 0, CacheTypeRanked) _ = idx - defer f.Clean(t) tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() + defer f.Clean(t) f.mustSetBits(tx, 0, 0) f.mustSetBits(tx, 1, 0) @@ -3650,7 +4119,7 @@ func TestFragmentRowIterator(t *testing.T) { _ = idx defer f.Clean(t) tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() f.mustSetBits(tx, 1, 0) f.mustSetBits(tx, 3, 0) @@ -3696,7 +4165,7 @@ func TestFragmentRowIterator(t *testing.T) { _ = idx defer f.Clean(t) tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() f.mustSetBits(tx, 0, 0) f.mustSetBits(tx, 1, 0) @@ -3731,7 +4200,7 @@ func TestFragmentRowIterator(t *testing.T) { _ = idx defer f.Clean(t) tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() f.mustSetBits(tx, 1, 0) f.mustSetBits(tx, 3, 0) @@ -3762,7 +4231,194 @@ func TestFragmentRowIterator(t *testing.T) { }) } +// same, with commits +func TestFragmentRowIterator_WithTxCommit(t *testing.T) { + t.Run("basic", func(t *testing.T) { + f, idx := mustOpenFragment("i", "f", "v", 0, CacheTypeRanked) + _ = idx + tx := f.txTestingOnly + defer tx.Rollback() + defer f.Clean(t) + + f.mustSetBits(tx, 0, 0) + f.mustSetBits(tx, 1, 0) + f.mustSetBits(tx, 2, 0) + f.mustSetBits(tx, 3, 0) + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + iter, err := f.rowIterator(tx, false) + if err != nil { + t.Fatal(err) + } + for i := uint64(0); i < 4; i++ { + row, id, _, wrapped, err := iter.Next() + if err != nil { + t.Fatal(err) + } + if id != i { + t.Fatalf("expected row %d but got %d", i, id) + } + if wrapped { + t.Fatalf("shouldn't have wrapped") + } + if !reflect.DeepEqual(row.Columns(), []uint64{0}) { + t.Fatalf("got wrong columns back on iteration %d - should just be 0 but %v", i, row.Columns()) + } + } + row, id, _, wrapped, err := iter.Next() + if err != nil { + t.Fatal(err) + } + if row != nil { + t.Fatalf("row should be nil after iterator is exhausted, got %v", row.Columns()) + } + if id != 0 { + t.Fatalf("id should be 0 after iterator is exhausted, got %d", id) + } + if !wrapped { + t.Fatalf("wrapped should be true after iterator is exhausted") + } + }) + + t.Run("skipped rows", func(t *testing.T) { + f, idx := mustOpenFragment("i", "f", "v", 0, CacheTypeRanked) + _ = idx + defer f.Clean(t) + tx := f.txTestingOnly + defer tx.Rollback() + + f.mustSetBits(tx, 1, 0) + f.mustSetBits(tx, 3, 0) + f.mustSetBits(tx, 5, 0) + f.mustSetBits(tx, 7, 0) + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + iter, err := f.rowIterator(tx, false) + if err != nil { + t.Fatal(err) + } + for i := uint64(1); i < 8; i += 2 { + row, id, _, wrapped, err := iter.Next() + if err != nil { + t.Fatal(err) + } + if id != i { + t.Fatalf("expected row %d but got %d", i, id) + } + if wrapped { + t.Fatalf("shouldn't have wrapped") + } + if !reflect.DeepEqual(row.Columns(), []uint64{0}) { + t.Fatalf("got wrong columns back on iteration %d - should just be 0 but %v", i, row.Columns()) + } + } + row, id, _, wrapped, err := iter.Next() + if err != nil { + t.Fatal(err) + } + if row != nil { + t.Fatalf("row should be nil after iterator is exhausted, got %v", row.Columns()) + } + if id != 0 { + t.Fatalf("id should be 0 after iterator is exhausted, got %d", id) + } + if !wrapped { + t.Fatalf("wrapped should be true after iterator is exhausted") + } + }) + + t.Run("basic wrapped", func(t *testing.T) { + f, idx := mustOpenFragment("i", "f", "v", 0, CacheTypeRanked) + _ = idx + defer f.Clean(t) + tx := f.txTestingOnly + defer tx.Rollback() + + f.mustSetBits(tx, 0, 0) + f.mustSetBits(tx, 1, 0) + f.mustSetBits(tx, 2, 0) + f.mustSetBits(tx, 3, 0) + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + iter, err := f.rowIterator(tx, true) + if err != nil { + t.Fatal(err) + } + for i := uint64(0); i < 5; i++ { + row, id, _, wrapped, err := iter.Next() + if err != nil { + t.Fatal(err) + } + if id != i%4 { + t.Fatalf("expected row %d but got %d", i%4, id) + } + if wrapped && i < 4 { + t.Fatalf("shouldn't have wrapped") + } else if !wrapped && i >= 4 { + t.Fatalf("should have wrapped") + } + if !reflect.DeepEqual(row.Columns(), []uint64{0}) { + t.Fatalf("got wrong columns back on iteration %d - should just be 0 but %v", i, row.Columns()) + } + } + }) + + t.Run("skipped rows wrapped", func(t *testing.T) { + f, idx := mustOpenFragment("i", "f", "v", 0, CacheTypeRanked) + _ = idx + defer f.Clean(t) + tx := f.txTestingOnly + defer tx.Rollback() + + f.mustSetBits(tx, 1, 0) + f.mustSetBits(tx, 3, 0) + f.mustSetBits(tx, 5, 0) + f.mustSetBits(tx, 7, 0) + + panicOn(tx.Commit()) + tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + defer tx.Rollback() + + iter, err := f.rowIterator(tx, true) + if err != nil { + t.Fatal(err) + } + for i := uint64(1); i < 10; i += 2 { + row, id, _, wrapped, err := iter.Next() + if err != nil { + t.Fatal(err) + } + if id != i%8 { + t.Errorf("expected row %d but got %d", i%8, id) + } + if wrapped && i < 8 { + t.Errorf("shouldn't have wrapped") + } else if !wrapped && i >= 8 { + t.Errorf("should have wrapped") + } + if !reflect.DeepEqual(row.Columns(), []uint64{0}) { + t.Fatalf("got wrong columns back on iteration %d - should just be 0 but %v", i, row.Columns()) + } + } + }) +} + func TestUnionInPlaceMapped(t *testing.T) { + + skipForBadger := os.Getenv("PILOSA_TXSRC") == "badger" + if skipForBadger { + t.Skip("skip for badger") + } + f, idx := mustOpenFragment("i", "f", "v", 0, CacheTypeNone) // note: clean has to be deferred first, because it has to run with // the lock *not* held, because it is sometimes so it has to grab the @@ -3928,7 +4584,7 @@ func TestIntLTRegression(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() _, err := f.setValue(tx, 1, 6, 33) if err != nil { @@ -3961,7 +4617,8 @@ func TestFragmentBSIUnsigned(t *testing.T) { defer f.Clean(t) // Obtain transaction. - tx := &RoaringTx{fragment: f} + tx := f.txTestingOnly + defer tx.Rollback() // Number of bits to test. const k = 6 @@ -4116,13 +4773,181 @@ func TestFragmentBSIUnsigned(t *testing.T) { }) } +// same, WithTxCommit version +func TestFragmentBSIUnsigned_WithTxCommit(t *testing.T) { + f, idx := mustOpenFragment("i", "f", "v", 0, CacheTypeNone) + _ = idx + defer f.Clean(t) + + // Obtain transaction. + tx := f.txTestingOnly + defer tx.Rollback() + + // Number of bits to test. + const k = 6 + + // Load all numbers into an effectively diagonal matrix. + for i := 0; i < 1<", func(t *testing.T) { + for i := minCheck; i < maxCheck; i++ { + row, err := f.rangeGT(tx, k, int64(i), false) + if err != nil { + t.Fatalf("failed to query fragment: %v", err) + } + var expect []uint64 + switch { + case i < 0: + expect = cols + case i < len(cols)-1: + expect = cols[i+1:] + default: + } + got := row.Columns() + if !sliceEq(expect, got) { + t.Errorf("expected %v but got %v for x > %d", expect, got, i) + } + } + }) + t.Run(">=", func(t *testing.T) { + for i := minCheck; i < maxCheck; i++ { + row, err := f.rangeGT(tx, k, int64(i), true) + if err != nil { + t.Fatalf("failed to query fragment: %v", err) + } + var expect []uint64 + switch { + case i < 0: + expect = cols + case i < len(cols): + expect = cols[i:] + default: + } + got := row.Columns() + if !sliceEq(expect, got) { + t.Errorf("expected %v but got %v for x >= %d", expect, got, i) + } + } + }) + t.Run("Range", func(t *testing.T) { + for i := minCheck; i < maxCheck; i++ { + for j := i; j < maxCheck; j++ { + row, err := f.rangeBetween(tx, k, int64(i), int64(j)) + if err != nil { + t.Fatalf("failed to query fragment: %v", err) + } + var lower, upper int + switch { + case i < 0: + lower = 0 + case i > len(cols): + lower = len(cols) + default: + lower = i + } + switch { + case j < 0: + upper = 0 + case j >= len(cols): + upper = len(cols) + default: + upper = j + 1 + } + expect := cols[lower:upper] + got := row.Columns() + if !sliceEq(expect, got) { + t.Errorf("expected %v but got %v for %d <= x <= %d", expect, got, i, j) + } + } + } + }) + t.Run("==", func(t *testing.T) { + for i := minCheck; i < maxCheck; i++ { + row, err := f.rangeEQ(tx, k, int64(i)) + if err != nil { + t.Fatalf("failed to query fragment: %v", err) + } + var expect []uint64 + if i >= 0 && i < len(cols) { + expect = cols[i : i+1] + } + got := row.Columns() + if !sliceEq(expect, got) { + t.Errorf("expected %v but got %v for x == %d", expect, got, i) + } + } + }) +} + func TestFragmentBSISigned(t *testing.T) { f, idx := mustOpenFragment("i", "f", "v", 0, CacheTypeNone) _ = idx defer f.Clean(t) // Obtain transaction. - tx := &RoaringTx{fragment: f} + tx := f.txTestingOnly + defer tx.Rollback() // Number of bits to test. const k = 6 @@ -4365,7 +5190,18 @@ func TestImportClearRestart(t *testing.T) { check(t, tx, f, exp) - f2 := newFragment(NewHolder(DefaultPartitionN), f.path, "i", "f", viewStandard, 0, 0) + h := NewHolder(DefaultPartitionN) + h.Path = filepath.Dir(f.path) + idx2, err := h.CreateIndex("i", IndexOptions{}) + _ = idx2 + panicOn(err) + + // OVERWRITING the f.path with a new fragment + f2 := newFragment(h, f.path, "i", "f", viewStandard, 0, 0) + + // f2, idx2 := mustOpenFragment("i", "f", viewStandard, 0, "") + // _ = idx2 + f2.MaxOpN = maxOpN f2.CacheType = f.CacheType @@ -4408,7 +5244,13 @@ func TestImportClearRestart(t *testing.T) { panicOn(tx2.Commit()) - f3 := newFragment(NewHolder(DefaultPartitionN), f2.path, "i", "f", viewStandard, 0, 0) + h3 := NewHolder(DefaultPartitionN) + h3.Path = filepath.Dir(f2.path) + idx3, err := h3.CreateIndex("i", IndexOptions{}) + _ = idx3 + panicOn(err) + + f3 := newFragment(h3, f2.path, "i", "f", viewStandard, 0, 0) f3.MaxOpN = maxOpN f3.CacheType = f.CacheType @@ -4529,13 +5371,18 @@ func TestImportMultipleValues(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() err := f.importValue(tx, test.cols, test.vals, test.depth, false) if err != nil { t.Fatalf("importing values: %v", err) } + // probably too slow, would hit disk alot: + //panicOn(tx.Commit()) + //tx = idx.Txf.NewTx(Txo{Write: !writable, Index: idx, Fragment: f}) + //defer tx.Rollback() + for i := range test.checkCols { cc, cv := test.checkCols[i], test.checkVals[i] n, exists, err := f.value(tx, cc, test.depth) @@ -4592,7 +5439,7 @@ func TestImportValueRowCache(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // First import (tc1) if err := f.importValue(tx, test.tc1.cols, test.tc1.vals, test.tc1.depth, false); err != nil { @@ -4625,12 +5472,13 @@ func TestFragmentConcurrentReadWrite(t *testing.T) { _ = idx defer f.Clean(t) + // Obtain transaction, but don't start another b/c the + // two goroutines below need the same view. + tx := f.txTestingOnly + defer tx.Rollback() + eg := &errgroup.Group{} eg.Go(func() error { - // Obtain transaction. - tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME - for i := uint64(0); i < 1000; i++ { _, err := f.setBit(tx, i%4, i) if err != nil { @@ -4640,10 +5488,6 @@ func TestFragmentConcurrentReadWrite(t *testing.T) { return nil }) - // Obtain transaction. - tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME - acc := uint64(0) for i := uint64(0); i < 100; i++ { r := f.mustRow(tx, i%4) @@ -4659,6 +5503,7 @@ func TestFragmentConcurrentReadWrite(t *testing.T) { func TestRemapCache(t *testing.T) { f, idx := mustOpenFragment("i", "f", viewStandard, 0, "") _ = idx + index, field, view, shard := f.index, f.field, f.view, f.shard // request a panic that doesn't kill the program on fault wouldFault := debug.SetPanicOnFault(true) @@ -4678,10 +5523,10 @@ func TestRemapCache(t *testing.T) { // Obtain transaction. tx := f.txTestingOnly - defer tx.Rollback() //LOOKATME + defer tx.Rollback() // create a container - _, err := f.storage.Add(65537) + _, err := tx.Add(index, field, view, shard, !doBatched, 65537) if err != nil { t.Fatalf("storage add: %v", err) } @@ -4694,7 +5539,7 @@ func TestRemapCache(t *testing.T) { _ = f.mustRow(tx, 0) // add a bit that isn't in that container, so that container doesn't // change - _, err = f.storage.Add(2) + _, err = tx.Add(index, field, view, shard, !doBatched, 2) if err != nil { t.Fatalf("storage add: %v", err) } diff --git a/generation.go b/generation.go index e9c4e93c5..268585b1d 100644 --- a/generation.go +++ b/generation.go @@ -261,6 +261,7 @@ func (m *mmapGeneration) openFile() (shouldClose bool, err error) { if err != nil { return false, err } + // do we actually want this in every openFile? I don't know. if err := syscall.Flock(int(m.file.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil { _ = syswrap.CloseFile(m.file) @@ -438,3 +439,25 @@ func newGeneration(existing generation, path string, readData bool, setup func([ // does get cleaned up. return &m, nil } + +// NopGeneration is used in fragment.openStorage() to short-circuit +// generation stuff that only applies to RoaringTx; doesn't apply to RBFTx/BadgerTx/etc. +type NopGeneration struct { +} + +func (g *NopGeneration) Transaction(w *io.Writer, f func() error) error { + return f() +} +func (g *NopGeneration) Done() {} +func (g *NopGeneration) Generation() int64 { + return 0 +} +func (g *NopGeneration) ID() string { + return "NOP" +} +func (g *NopGeneration) Dead() bool { + return true +} +func (g *NopGeneration) Bytes() (ret []byte) { + return +} diff --git a/generation_test.go b/generation_test.go index 7df71fc98..bc0a1854f 100644 --- a/generation_test.go +++ b/generation_test.go @@ -53,8 +53,11 @@ func TestGenerationPanic(t *testing.T) { if unsafe.Pointer(&prevData[0]) == unsafe.Pointer(&newData[0]) { t.Fatalf("test can't run usefully, didn't get new data pointer") } - - err := f.gen.Transaction(&f.storage.OpWriter, func() error { + var wp *io.Writer + if f.storage != nil { + wp = &f.storage.OpWriter + } + err := f.gen.Transaction(wp, func() error { prevData[0] = 0x3c return nil }) diff --git a/holder.go b/holder.go index 58dca7f2c..e74672d31 100644 --- a/holder.go +++ b/holder.go @@ -528,9 +528,9 @@ func (h *Holder) Open() error { if h.isCoordinator() { index.createdAt = timestamp() - err = index.OpenWithTimestamp() + err = index.OpenWithTimestamp(false) } else { - err = index.Open() + err = index.Open(false) } if err != nil { if err == ErrName { @@ -626,6 +626,17 @@ func (h *Holder) BeginTx(writable bool, index *Index) (Tx, error) { return index.Txf.NewTx(Txo{Write: writable, Index: index}), nil } +func (h *Holder) NeedsSnapshot() bool { + h.mu.RLock() + defer h.mu.RUnlock() + for _, idx := range h.indexes { + if idx.NeedsSnapshot() { + return true + } + } + return false +} + // HasData returns true if Holder contains at least one index. // This is used to determine if the rebalancing of data is necessary // when a node joins the cluster. @@ -802,7 +813,20 @@ func (h *Holder) applyCreatedAt(indexes []*IndexInfo) { } // IndexPath returns the path where a given index is stored. -func (h *Holder) IndexPath(name string) string { return filepath.Join(h.Path, name) } +func (h *Holder) IndexPath(name string) string { + return filepath.Join(h.Path, name) +} + +// HolderPathFromIndexPath is +// used by test/index.go:71 in test.Index.Reopen() to get the right +// path into a test Holder that doesn't know its own proper path. +// If the Holder changes index paths to being something other than +// holderPath + "/" + indexName, this will need adjusting too. +func (h *Holder) HolderPathFromIndexPath(indexPath, indexName string) string { + n := len(indexPath) + hpath2 := indexPath[:n-(len(indexName)+1)] + return hpath2 +} // Index returns the index by name. func (h *Holder) Index(name string) *Index { @@ -811,7 +835,9 @@ func (h *Holder) Index(name string) *Index { return h.index(name) } -func (h *Holder) index(name string) *Index { return h.indexes[name] } +func (h *Holder) index(name string) *Index { + return h.indexes[name] +} // Indexes returns a list of all indexes in the holder. func (h *Holder) Indexes() []*Index { @@ -867,7 +893,7 @@ func (h *Holder) createIndex(name string, opt IndexOptions) (*Index, error) { index.keys = opt.Keys index.trackExistence = opt.TrackExistence - if err = index.Open(); err != nil { + if err = index.Open(true); err != nil { return nil, errors.Wrap(err, "opening") } if err = index.saveMeta(); err != nil { @@ -1799,3 +1825,15 @@ func (h *Holder) Process(ctx context.Context, op HolderOperator) (err error) { } return nil } + +// used by Index.openFields(), enabling Tx / Txf by telling +// the holder about its own indexes. +func (h *Holder) addIndexFromField(idx *Index) { + h.mu.Lock() + h.indexes[idx.Name()] = idx + h.mu.Unlock() +} + +func (h *Holder) unprotectedAddIndexFromField(idx *Index) { + h.indexes[idx.Name()] = idx +} diff --git a/holder_test.go b/holder_test.go index a527307ba..4280dadbd 100644 --- a/holder_test.go +++ b/holder_test.go @@ -32,6 +32,8 @@ import ( ) func TestHolder_Open(t *testing.T) { + skipForBadger := os.Getenv("PILOSA_TXSRC") == "badger" + t.Run("ErrIndexName", func(t *testing.T) { h := test.MustOpenHolder() @@ -165,6 +167,9 @@ func TestHolder_Open(t *testing.T) { }) t.Run("ErrFragmentStoragePermission", func(t *testing.T) { + if skipForBadger { + t.Skip("skipping for badger") + } if os.Geteuid() == 0 { t.Skip("Skipping permissions test since user is root.") } @@ -201,6 +206,10 @@ func TestHolder_Open(t *testing.T) { } }) t.Run("ErrFragmentStorageCorrupt", func(t *testing.T) { + if skipForBadger { + t.Skip("skipping for badger") + } + h := test.MustOpenHolder() defer h.Close() @@ -233,6 +242,10 @@ func TestHolder_Open(t *testing.T) { } }) t.Run("ErrFragmentStorageRecoverable", func(t *testing.T) { + if skipForBadger { + t.Skip("skipping for badger") + } + h := test.MustOpenHolder() defer h.Close() @@ -723,24 +736,17 @@ func TestHolderSyncer_IntField(t *testing.T) { hldr0.SetValue("i", "f", 1, 1) // in c0 expect the 1 bit - //idx0.Dump("in c0, before SyncData") // Set data on node1. columnID=2, value=2 idx1 := hldr1.SetValue("i", "f", 2, 2) _ = idx1 - //idx1.Dump("in c1, before SyncData") - - //vv("before c[0] SyncData") err = c[0].Server.SyncData() if err != nil { t.Fatalf("syncing node 0: %v", err) } - //vv("after c[0] SyncData") // expect 3 rows, the 1 bit + 2 rows for the 2 value as BSI. But, we only see that c0 overwrote c1. - //idx0.Dump("in c0, after syncData") - //idx1.Dump("in c1, after syncData") // Problem is: data at c1 was replaced by c0, instead of being merged with existing c1. // Problem is: data at c0 did not receive and merge the c1 data. @@ -810,8 +816,6 @@ func TestHolderSyncer_IntField(t *testing.T) { } // dump the badger keys for both c0 and c1 - //vv("in c0, allkeys = '%v'", idx0.StringifiedBadgerKeys(nil)) - //vv("in c1, allkeys = '%v'", c[1].index.StringifiedBadgerKeys()) // Verify data is the same on both nodes. for i, hldr := range []*test.Holder{hldr0, hldr1} { diff --git a/index.go b/index.go index 1474616bc..f9c6d2099 100644 --- a/index.go +++ b/index.go @@ -99,16 +99,16 @@ func NewIndex(holder *Holder, path, name string) (*Index, error) { } } - txf, err := newTxFactory(txsrc, path) - if err != nil { - return nil, errors.Wrap(err, "creating newTxFactory") - } - - err = validateName(name) + err := validateName(name) if err != nil { return nil, errors.Wrap(err, "validating name") } + txf, err := NewTxFactory(txsrc, holder.Path, name) + if err != nil { + return nil, errors.Wrap(err, "creating newTxFactory") + } + idx := &Index{ path: path, name: name, @@ -134,6 +134,14 @@ func NewIndex(holder *Holder, path, name string) (*Index, error) { return idx, nil } +func (i *Index) NewTx(txo Txo) Tx { + return i.Txf.NewTx(txo) +} + +func (i *Index) NeedsSnapshot() bool { + return i.Txf.NeedsSnapshot() +} + // CreatedAt is an timestamp for a specific version of an index. func (i *Index) CreatedAt() int64 { i.mu.RLock() @@ -148,7 +156,9 @@ func (i *Index) Name() string { return i.name } func (i *Index) QualifiedName() string { return i.qualifiedName } // Path returns the path the index was initialized with. -func (i *Index) Path() string { return i.path } +func (i *Index) Path() string { + return i.path +} // TranslateStorePath returns the translation database path for a partition. func (i *Index) TranslateStorePath(partitionID int) string { @@ -181,12 +191,12 @@ func (i *Index) options() IndexOptions { } // Open opens and initializes the index. -func (i *Index) Open() error { return i.open(false) } +func (i *Index) Open(haveHolderLock bool) error { return i.open(false, haveHolderLock) } // OpenWithTimestamp opens and initializes the index and set a new CreatedAt timestamp for fields. -func (i *Index) OpenWithTimestamp() error { return i.open(true) } +func (i *Index) OpenWithTimestamp(haveHolderLock bool) error { return i.open(true, haveHolderLock) } -func (i *Index) open(withTimestamp bool) (err error) { +func (i *Index) open(withTimestamp, haveHolderLock bool) (err error) { // Ensure the path exists. i.holder.Logger.Debugf("ensure index path exists: %s", i.path) if err := os.MkdirAll(i.path, 0777); err != nil { @@ -200,7 +210,7 @@ func (i *Index) open(withTimestamp bool) (err error) { } i.holder.Logger.Debugf("open fields for index: %s", i.name) - if err := i.openFields(withTimestamp); err != nil { + if err := i.openFields(withTimestamp, haveHolderLock); err != nil { return errors.Wrap(err, "opening fields") } @@ -243,7 +253,7 @@ func (i *Index) open(withTimestamp bool) (err error) { var indexQueue = make(chan struct{}, 8) // openFields opens and initializes the fields inside the index. -func (i *Index) openFields(withTimestamp bool) error { +func (i *Index) openFields(withTimestamp, haveHolderLock bool) error { f, err := os.Open(i.path) if err != nil { return errors.Wrap(err, "opening directory") @@ -273,7 +283,31 @@ fileLoop: <-indexQueue }() i.holder.Logger.Debugf("open field: %s", fi.Name()) + mu.Lock() + + // i.holder needs to know about its index i for the Txf to work. + // + // We face either a deadlock or a race here. + // + // We get a deadlock in TestIndex_CreateField/"BSIFields"/"OK" + // if we call addIndexFromField, because in that test + // we get here while already holding i.holder.mu. + // + // On the other had, we get races on other tests + // such as TestExecutor_Execute_Existence/Row + // if we call unprotectedAddIndexFromField which does + // not lock i.holder.mu. + // + // The resolution was to have the goroutines that are holding + // the lock already tell us. That is the haveHolderLock + // argument. + if haveHolderLock { + i.holder.unprotectedAddIndexFromField(i) + } else { + i.holder.addIndexFromField(i) + } + fld, err := i.newField(i.fieldPath(filepath.Base(fi.Name())), filepath.Base(fi.Name())) if withTimestamp { fld.createdAt = timestamp() @@ -287,6 +321,7 @@ fileLoop: // up a foreign index. fld.holder = i.holder + // open all the views if err := fld.Open(); err != nil { return fmt.Errorf("open field: name=%s, err=%s", fld.Name(), err) } @@ -546,6 +581,9 @@ func (i *Index) createField(name string, opt *FieldOptions) (*Field, error) { // Add to index's field lookup. i.fields[name] = f + // enable Txf to find the index in field_test.go TestField_SetValue + f.idx = i + // Kick off the field's translation sync process. if err := i.translationSyncer.Reset(); err != nil { return nil, errors.Wrap(err, "resetting translation syncer") @@ -559,6 +597,7 @@ func (i *Index) newField(path, name string) (*Field, error) { if err != nil { return nil, err } + f.idx = i f.Stats = i.Stats f.broadcaster = i.broadcaster f.rowAttrStore = i.newAttrStore(filepath.Join(f.path, ".data")) diff --git a/index_internal_test.go b/index_internal_test.go index 5b5a5b5a3..4f569c745 100644 --- a/index_internal_test.go +++ b/index_internal_test.go @@ -25,7 +25,10 @@ func mustOpenIndex(opt IndexOptions) *Index { if err != nil { panic(err) } - index, err := NewIndex(NewHolder(1), path, "i") + h := NewHolder(1) + h.Path = path + index, err := h.CreateIndex("i", opt) + if err != nil { panic(err) } @@ -33,7 +36,7 @@ func mustOpenIndex(opt IndexOptions) *Index { index.keys = opt.Keys index.trackExistence = opt.TrackExistence - if err := index.Open(); err != nil { + if err := index.Open(false); err != nil { panic(err) } return index @@ -44,7 +47,7 @@ func (i *Index) reopen() error { if err := i.Close(); err != nil { return err } - if err := i.Open(); err != nil { + if err := i.Open(false); err != nil { return err } return nil diff --git a/mmap_test.go b/mmap_test.go index 20fbd7685..43321cea7 100644 --- a/mmap_test.go +++ b/mmap_test.go @@ -32,11 +32,11 @@ type cv struct { func forceSnapshotsCheckMapping(t *testing.T) { depth := uint(6) f, idx := mustOpenBSIFragment("i", "f", viewStandard, 0) - _ = idx f.Logger = logger.NewLogfLogger(t) defer f.Clean(t) - tx := &RoaringTx{fragment: f} + tx := idx.Txf.NewTx(Txo{Write: writable, Index: idx, Fragment: f}) + defer tx.Rollback() for i := 0; i < f.MaxOpN; i++ { _, _ = f.setBit(tx, 0, uint64(32*i)) diff --git a/pql/ast.go b/pql/ast.go index 8ebfaf122..b150cd3c5 100644 --- a/pql/ast.go +++ b/pql/ast.go @@ -22,8 +22,6 @@ import ( "strconv" "strings" "time" - - "github.com/molecula/ext" ) // Query represents a PQL query. @@ -345,7 +343,8 @@ var callInfoByFunc = map[string]callInfo{ "Row": {allowUnknown: true}, "Range": {allowUnknown: true}, - "Distinct": {allowUnknown: true}, + "Distinct": {allowUnknown: true, callType: PrecallGlobal}, + "Condition": {allowUnknown: true}, // allow only "field=X" cases with string field names "Max": allowField, @@ -462,30 +461,6 @@ var callInfoByFunc = map[string]callInfo{ }, } -// RegisterPluginFuncs adds arg validation for plugin funcs. Not very good -// arg validation. -func RegisterPluginFuncs(ops []ext.BitmapOp) { - for _, op := range ops { - // ignore overlap for now. This should change. - if _, ok := callInfoByFunc[op.Name]; ok { - continue - } - ci := callInfo{allowUnknown: true} - if len(op.Reserved) > 0 { - // mark these as valid/known reserved words - ci.prototypes = make(map[string]interface{}) - for _, res := range op.Reserved { - ci.prototypes[res] = nil - } - } - t := op.Func.BitmapOpType() - if t.Precall == ext.OpPrecallGlobal { - ci.callType = PrecallGlobal - } - callInfoByFunc[op.Name] = ci - } -} - // CheckCallInfo tries to validate that arguments are correct and valid for the // given call. It does not guarantee checking all possible errors; for instance, // if an argument is a field name, CheckCallInfo can't validate that the field @@ -505,6 +480,11 @@ func (c *Call) CheckCallInfo() error { if !ok && strings.HasPrefix(k, "_") { return fmt.Errorf("'%s': unknown reserved arg '%s'", c.String(), k) } + if call, ok := v.(*Call); ok { + if err := call.CheckCallInfo(); err != nil { + return err + } + } if acceptable == nil { continue } diff --git a/rbf/db.go b/rbf/db.go index 09f4b5aa9..3bb960f71 100644 --- a/rbf/db.go +++ b/rbf/db.go @@ -71,6 +71,10 @@ func NewDBWithShard(path string, shard int) *DB { } } +func (db *DB) DeleteFragment(index, field, view string, shard uint64, frag interface{}) error { + panic("TODO: implement rbf.DB.DeleteFragment") +} + // DataPath returns the path to the data file for the DB. func (db *DB) DataPath() string { return filepath.Join(db.Path, "data") diff --git a/roaring/container_stash.go b/roaring/container_stash.go index 3531a0317..095f462e5 100644 --- a/roaring/container_stash.go +++ b/roaring/container_stash.go @@ -387,7 +387,7 @@ func (c *Container) bitmap() []uint64 { // is provided. The target should be zeroed, or this becomes an implicit // union. func (c *Container) AsBitmap(target []uint64) (out []uint64) { - if c.typeID == containerBitmap { + if c != nil && c.typeID == containerBitmap { return c.bitmap() } // Reminder: len(nil) == 0. @@ -399,6 +399,10 @@ func (c *Container) AsBitmap(target []uint64) (out []uint64) { out[i] = 0 } } + // A nil *Container is a valid empty container. + if c == nil { + return out + } if c.typeID == containerArray { a := c.array() for _, v := range a { diff --git a/row.go b/row.go index 30c11d2fe..ac0ef56e3 100644 --- a/row.go +++ b/row.go @@ -18,7 +18,6 @@ import ( "encoding/json" "sort" - "github.com/molecula/ext" pb "github.com/pilosa/pilosa/v2/proto" "github.com/pilosa/pilosa/v2/roaring" "github.com/pkg/errors" @@ -326,65 +325,6 @@ func (r *Row) Union(others ...*Row) *Row { return &Row{segments: output} } -// GenericBinaryOp returns the output of a generic op on r and other. -func (r *Row) GenericBinaryOp(op ext.GenericBitmapOpBitmap, other *Row, args map[string]interface{}) *Row { - var segments []rowSegment - itr := newMergeSegmentIterator(r.segments, other.segments) - for s0, s1 := itr.next(); s0 != nil || s1 != nil; s0, s1 = itr.next() { - if s1 == nil { - segments = append(segments, *s0) - continue - } else if s0 == nil { - segments = append(segments, *s1) - continue - } - segments = append(segments, *s0.GenericBinaryOp(op, s1, args)) - } - - return &Row{segments: segments} -} - -// GenericNaryOp returns the output of an nary op on r and others. -func (r *Row) GenericNaryOp(op ext.GenericBitmapOpBitmap, others []*Row, args map[string]interface{}) *Row { - segments := make([][]rowSegment, 0, len(others)+1) - if len(r.segments) > 0 { - segments = append(segments, r.segments) - } - nextSegs := make([][]rowSegment, 0, len(others)+1) - toProcess := make([]*rowSegment, 0, len(others)+1) - var output []rowSegment - for _, other := range others { - if len(other.segments) > 0 { - segments = append(segments, other.segments) - } - } - for len(segments) > 0 { - shard := segments[0][0].shard - for _, segs := range segments { - if segs[0].shard < shard { - shard = segs[0].shard - } - } - nextSegs = nextSegs[:0] - toProcess := toProcess[:0] - for _, segs := range segments { - if segs[0].shard == shard { - toProcess = append(toProcess, &segs[0]) - segs = segs[1:] - } - if len(segs) > 0 { - nextSegs = append(nextSegs, segs) - } - } - // at this point, "toProcess" is a list of all the segments - // sharing the lowest ID, and nextSegs is a list of all the others. - // Swap the segment lists (so we don't have to reallocate it) - segments, nextSegs = nextSegs, segments - output = append(output, *toProcess[0].GenericNaryOp(op, toProcess[1:], args)) - } - return &Row{segments: output} -} - // Difference returns the diff of r and other. func (r *Row) Difference(others ...*Row) *Row { var output []rowSegment @@ -408,17 +348,6 @@ func (r *Row) Difference(others ...*Row) *Row { return &Row{segments: output} } -// GenericUnaryOp returns the results of a generic op on r. -func (r *Row) GenericUnaryOp(op ext.GenericBitmapOpBitmap, args map[string]interface{}) *Row { - work := r - var segments []rowSegment - for _, segment := range work.segments { - opped := segment.GenericUnaryOp(op, args) - segments = append(segments, *opped) - } - return &Row{segments: segments} -} - // Shift returns the bitwise shift of r by n bits. // Currently only positive shift values are supported. // @@ -523,15 +452,6 @@ func (r *Row) Count() uint64 { return n } -// GenericCount applies an op to lots of things. -func (r *Row) GenericCount(op ext.BitmapOpUnaryCount, args map[string]interface{}) uint64 { - var n int64 - for i := range r.segments { - n += op([]ext.Bitmap{WrapBitmap(r.segments[i].data)}, args) - } - return uint64(n) -} - // MarshalJSON returns a JSON-encoded byte slice of r. func (r *Row) MarshalJSON() ([]byte, error) { var o struct { @@ -644,33 +564,6 @@ func (s *rowSegment) Union(others ...*rowSegment) *rowSegment { } } -// GenericOp performs a generic op on s and other -func (s *rowSegment) GenericBinaryOp(op ext.GenericBitmapOpBitmap, other *rowSegment, args map[string]interface{}) *rowSegment { - data := op([]ext.Bitmap{WrapBitmap(s.data), WrapBitmap(other.data)}, args) - - return &rowSegment{ - data: UnwrapBitmap(data), - shard: s.shard, - n: data.Count(), - } -} - -// GenericOp performs a generic op on s and others -func (s *rowSegment) GenericNaryOp(op ext.GenericBitmapOpBitmap, others []*rowSegment, args map[string]interface{}) *rowSegment { - bitmaps := make([]ext.Bitmap, len(others)+1) - bitmaps[0] = WrapBitmap(s.data) - for i, seg := range others { - bitmaps[i+1] = WrapBitmap(seg.data) - } - data := op(bitmaps, args) - - return &rowSegment{ - data: UnwrapBitmap(data), - shard: s.shard, - n: data.Count(), - } -} - // Difference returns the diff of s and other. func (s *rowSegment) Difference(others ...*rowSegment) *rowSegment { datas := make([]*roaring.Bitmap, len(others)) @@ -713,17 +606,6 @@ func (s *rowSegment) Shift() (*rowSegment, error) { }, nil } -// GenericUnaryOp returns s subject to op. -func (s *rowSegment) GenericUnaryOp(op ext.GenericBitmapOpBitmap, args map[string]interface{}) *rowSegment { - data := UnwrapBitmap(op([]ext.Bitmap{WrapBitmap(s.data)}, args)) - - return &rowSegment{ - data: data, - shard: s.shard, - n: data.Count(), - } -} - // SetBit sets the i-th column of the row. func (s *rowSegment) SetBit(i uint64) (changed bool) { s.ensureWritable() diff --git a/server.go b/server.go index f0974c0d5..a03a76899 100644 --- a/server.go +++ b/server.go @@ -27,13 +27,11 @@ import ( "sync" "time" - "github.com/molecula/ext" uuid "github.com/satori/go.uuid" // extensions pulls in some extensions depending on build tags _ "github.com/pilosa/pilosa/v2/extensions" "github.com/pilosa/pilosa/v2/logger" - "github.com/pilosa/pilosa/v2/pql" "github.com/pilosa/pilosa/v2/roaring" "github.com/pilosa/pilosa/v2/stats" "github.com/pkg/errors" @@ -63,7 +61,6 @@ type Server struct { // nolint: maligned hosts []string clusterDisabled bool serializer Serializer - extensions []*ext.ExtensionInfo // External systemInfo SystemInfo @@ -430,30 +427,6 @@ func NewServer(opts ...ServerOption) (*Server, error) { s.cluster.confirmDownRetries = s.confirmDownRetries s.cluster.confirmDownSleep = s.confirmDownSleep s.holder.broadcaster = s - err = s.loadAllExtensions() - if err != nil { - s.logger.Printf("not all plugins loaded successfully") - } - if len(s.extensions) > 0 { - s.logger.Printf("loaded extensions:") - for _, ext := range s.extensions { - if ext == nil { - s.logger.Printf(" inexplicably, a nil extension?!?") - continue - } - s.logger.Printf(" %s %s: %s", ext.Name, ext.Version, ext.Description) - if ext.License != "" { - s.logger.Printf(" License: %s", ext.License) - } - if len(ext.BitmapOps) > 0 { - opList := make([]string, len(ext.BitmapOps)) - for i := range ext.BitmapOps { - opList[i] = ext.BitmapOps[i].Name - } - s.logger.Printf(" Ops: %s", strings.Join(opList, ", ")) - } - } - } err = s.cluster.setup() if err != nil { @@ -467,56 +440,6 @@ func (s *Server) InternalClient() InternalClient { return s.defaultClient } -// loadNewExtensions loads extensions that have been -// registered since the last call to loadNewExtensions. -func (s *Server) loadNewExtensions() error { //nolint:unused - return s.loadExtensions(ext.NewExtensions()) -} - -// loadAllExtensions loads all extensions. -func (s *Server) loadAllExtensions() error { - return s.loadExtensions(ext.AllExtensions()) -} - -func (s *Server) loadExtensions(exts []*ext.ExtensionInfo) error { - var lastError error - for _, extension := range exts { - if err := s.loadExtension(extension); err != nil { - lastError = err - } - } - return lastError -} - -func (s *Server) loadExtension(extInfo *ext.ExtensionInfo) error { - if extInfo.ExtensionAPI != "v0" { - return fmt.Errorf("%s: unsupported extension API %s", extInfo.Name, extInfo.ExtensionAPI) - } - s.extensions = append(s.extensions, extInfo) - bitmapOps := extInfo.BitmapOps - bmOps, countOps, fieldOps, unknownOps := 0, 0, 0, 0 - for i := range bitmapOps { - typ := bitmapOps[i].Func.BitmapOpType() - switch { - case typ.Input == ext.OpInputBitmap && typ.Output == ext.OpOutputCount: - countOps++ - case typ.Input == ext.OpInputBitmap && typ.Output == ext.OpOutputBitmap: - bmOps++ - case typ.Input == ext.OpInputNaryBSI && typ.Output == ext.OpOutputSignedBitmap: - fieldOps++ - default: - unknownOps++ - } - } - err := s.executor.registerOps(bitmapOps) - if err != nil { - s.logger.Printf("warning: extension registration failed: %v", err) - } else { - pql.RegisterPluginFuncs(bitmapOps) - } - return nil -} - // UpAndDown brings the server up minimally and shuts it down // again; basically, it exists for testing holder open and close. func (s *Server) UpAndDown() error { @@ -545,8 +468,12 @@ func (s *Server) UpAndDown() error { func (s *Server) Open() error { s.logger.Printf("open server") - // Start background monitoring. - s.snapshotQueue = newSnapshotQueue(10, 2, s.logger) + if s.holder.NeedsSnapshot() { + // Start background monitoring. + s.snapshotQueue = newSnapshotQueue(10, 2, s.logger) + } else { + s.snapshotQueue = defaultSnapshotQueue //TODO (twg) rethink this + } // Log startup err := s.holder.logStartup() @@ -711,6 +638,7 @@ func (s *Server) monitorAntiEntropy() { // the cluster sets its state to resizing and *then* sends to // abortAntiEntropyCh before starting to resize } + // Sync holders. s.logger.Printf("holder sync beginning") s.cluster.muAntiEntropy.Lock() diff --git a/test/index.go b/test/index.go index b90702ae2..e51850d87 100644 --- a/test/index.go +++ b/test/index.go @@ -32,7 +32,9 @@ func newIndex() *Index { if err != nil { panic(err) } - index, err := pilosa.NewIndex(pilosa.NewHolder(pilosa.DefaultPartitionN), path, "i") + h := pilosa.NewHolder(pilosa.DefaultPartitionN) + h.Path = path + index, err := h.CreateIndex("i", pilosa.IndexOptions{}) if err != nil { panic(err) } @@ -42,7 +44,7 @@ func newIndex() *Index { // MustOpenIndex returns a new, opened index at a temporary path. Panic on error. func MustOpenIndex() *Index { index := newIndex() - if err := index.Open(); err != nil { + if err := index.Open(false); err != nil { panic(err) } return index @@ -62,12 +64,14 @@ func (i *Index) Reopen() error { } path, name := i.Path(), i.Name() - i.Index, err = pilosa.NewIndex(pilosa.NewHolder(pilosa.DefaultPartitionN), path, name) + h := pilosa.NewHolder(pilosa.DefaultPartitionN) + h.Path = h.HolderPathFromIndexPath(path, name) + i.Index, err = h.CreateIndex(name, pilosa.IndexOptions{}) if err != nil { return err } - if err := i.Open(); err != nil { + if err := i.Open(false); err != nil { return err } return nil diff --git a/tx.go b/tx.go index 9fcf6fd99..1dbea6daf 100644 --- a/tx.go +++ b/tx.go @@ -15,7 +15,12 @@ package pilosa import ( + "bytes" "fmt" + "io" + "os" + "path/filepath" + "strconv" "sync" "github.com/pilosa/pilosa/v2/roaring" @@ -50,6 +55,10 @@ const writable = true // that have not been committed. type Tx interface { + // Type returns "roaring", "rbf", "badger", "badger_roaring", or one of the other + // blue-green Tx types at the top of txfactory.go + Type() string + // Rollback must be called the end of read-only transactions. Either // Rollback or Commit must be called at the end of writable transactions. // It is safe to call Rollback multiple times, but it must be @@ -97,6 +106,11 @@ type Tx interface { // Calling Next() on the returned roaring.ContainerIterator gives // you a roaring.Container that is either run, array, or raw bitmap. // Return value 'found' is true when the ckey container was present. + // ckey of 0 gives all containers (in the fragment). + // + // ContainerIterator must not have side-effects. blueGreenTx will + // call it at the very beginning of commit to verify db contents. + // ContainerIterator(index, field, view string, shard uint64, ckey uint64) (citer roaring.ContainerIterator, found bool, err error) // RoaringBitmap retreives the roaring.Bitmap for the entire shard. @@ -162,10 +176,52 @@ type Tx interface { OffsetRange(index, field, view string, shard uint64, offset, start, end uint64) (*roaring.Bitmap, error) // ImportRoaringBits does efficient bulk import using rit, a roaring.RoaringIterator. + // // See the roaring package for details of the RoaringIterator. + // // If clear is true, the bits from rit are cleared, otherwise they are set in the // specifed fragment. - ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64) (changed int, rowSet map[uint64]int, err error) + // + // The data argument can be nil, its ignored for RBF/BadgerTx. It is supplied to + // RoaringTx.ImportRoaringBits() in fragment.go fragment.fillFragmentFromArchive() + // to do the traditional fragment.readStorageFromArchive() which + // does some in memory field/view/fragment metadata updates. + // It makes blueGreenTx testing viable too. + // + // ImportRoaringBits return values changed and rowSet may be inaccurate if + // the data []byte is supplied (the RoaringTx implementation neglects this for speed). + ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64, data []byte) (changed int, rowSet map[uint64]int, err error) + + RoaringBitmapReader(index, field, view string, shard uint64, fragmentPathForRoaring string) (r io.ReadCloser, sz int64, err error) + + // SliceOfShards returns all of the shards for the specified index, field, view triple. + // Use within pilosa supposes a new read-only transaction was created just + // for the SliceOfShards() call. The original Roaring version is the only + // one that needs optionalViewPath; any other Tx implementation can ignore that. + SliceOfShards(index, field, view, optionalViewPath string) (sliceOfShards []uint64, err error) +} + +// TxStore has operations that will create and commit multiple +// Tx on a backing store. +type TxStore interface { + + // DeleteFragment deletes all the containers in a fragment. + // + // This is not in a Tx because it will often do too many deletes for a single + // transaction, and clients would be suprised to find their Tx had already + // been commited and they are getting an error on double-Commit. + // Instead each TxStore implementation creates and commits as many + // transactions as needed. + // + // Argument frag should be passed by any RoaringTx user, but for RBF/Badger it can be nil. + // If not nil, it must be of type *fragment. If frag is supplied, then + // index must be equal to frag.index, field equal to frag.field, view equal + // to frag.view, and shard equal to frag.shard. + // + DeleteFragment(index, field, view string, shard uint64, frag interface{}) error + + // Close shuts down the database. + Close() error } // RawRoaringData used by ImportRoaringBits. @@ -207,10 +263,26 @@ func NewMultiTxWithIndex(writable bool, index *Index) *MultiTx { var _ Tx = (*MultiTx)(nil) +func (mtx *MultiTx) Type() string { + return RoaringTxn +} + +func (mtx *MultiTx) SliceOfShards(index, field, view, optionalViewPath string) (sliceOfShards []uint64, err error) { + tx, err := mtx.txNoShard(index) + panicOn(err) + return tx.SliceOfShards(index, field, view, optionalViewPath) +} + func (mtx *MultiTx) UseRowCache() bool { return true } +func (mtx *MultiTx) RoaringBitmapReader(index, field, view string, shard uint64, fragmentPathForRoaring string) (r io.ReadCloser, sz int64, err error) { + tx, err := mtx.tx(index, shard) + panicOn(err) + return tx.RoaringBitmapReader(index, field, view, shard, fragmentPathForRoaring) +} + func (mtx *MultiTx) NewTxIterator(index, field, view string, shard uint64) *roaring.Iterator { tx, err := mtx.tx(index, shard) panicOn(err) @@ -226,8 +298,10 @@ func (mtx *MultiTx) Pointer() string { return fmt.Sprintf("%p", mtx) } -func (tx *MultiTx) ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64) (changed int, rowSet map[uint64]int, err error) { - panic("not done") +func (mtx *MultiTx) ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64, data []byte) (changed int, rowSet map[uint64]int, err error) { + tx, err := mtx.tx(index, shard) + panicOn(err) + return tx.ImportRoaringBits(index, field, view, shard, rit, clear, log, rowSize, data) } func (mtx *MultiTx) IncrementOpN(index, field, view string, shard uint64, changedN int) { @@ -412,6 +486,20 @@ func (mtx *MultiTx) tx(index string, shard uint64) (_ Tx, err error) { return tx, nil } +// version of the above for SliceOfShards(), where we don't have a shard. +func (mtx *MultiTx) txNoShard(index string) (_ Tx, err error) { + mtx.mu.Lock() + defer mtx.mu.Unlock() + + // Lookup transaction from cache. + for _, tx := range mtx.txs { + if tx.(*RoaringTx).Index.name == index { + return tx, nil + } + } + panic(fmt.Sprintf("no prior RoaringTx available, looking up index='%v'", index)) +} + type multiTxKey struct { index string shard uint64 @@ -426,10 +514,46 @@ type RoaringTx struct { fragment *fragment } +func (mtx *RoaringTx) Type() string { + return RoaringTxn +} + func (tx *RoaringTx) UseRowCache() bool { return true } +func (tx *RoaringTx) SliceOfShards(index, field, view, optionalViewPath string) (sliceOfShards []uint64, err error) { + + // SliceOfShards is based on view.openFragments() + + file, err := os.Open(filepath.Join(optionalViewPath, "fragments")) + if os.IsNotExist(err) { + return + } else if err != nil { + return nil, errors.Wrap(err, "opening fragments directory") + } + defer file.Close() + + fis, err := file.Readdir(0) + if err != nil { + return nil, errors.Wrap(err, "reading fragments directory") + } + + for _, fi := range fis { + if fi.IsDir() { + continue + } + // Parse filename into integer. + shard, err := strconv.ParseUint(filepath.Base(fi.Name()), 10, 64) + if err != nil { + //v.holder.Logger.Debugf("WARNING: couldn't use non-integer file as shard in index/field/view %s/%s/%s: %s", v.index, v.field, v.name, fi.Name()) + continue + } + sliceOfShards = append(sliceOfShards, shard) + } + return +} + func (tx *RoaringTx) Pointer() string { return fmt.Sprintf("%p", tx) } @@ -442,13 +566,25 @@ func (tx *RoaringTx) NewTxIterator(index, field, view string, shard uint64) *roa return b.Iterator() } -func (tx *RoaringTx) ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64) (changed int, rowSet map[uint64]int, err error) { - b, err := tx.bitmap(index, field, view, shard) - panicOn(err) +// ImportRoaringBits return values changed and rowSet will be inaccurate if +// the data []byte is supplied. This mimics the traditional roaring-per-file +// and should be faster. +func (tx *RoaringTx) ImportRoaringBits(index, field, view string, shard uint64, rit roaring.RoaringIterator, clear bool, log bool, rowSize uint64, data []byte) (changed int, rowSet map[uint64]int, err error) { + + f, err := tx.getFragment(index, field, view, shard) if err != nil { return 0, nil, err } - return b.ImportRoaringRawIterator(rit, clear, true, rowSize) + if len(data) > 0 { + // changed and rowSet are ignored anyway when len(data) > 0; + // when we are called from fragment.fillFragmentFromArchive() + // which is the only place the data []byte is supplied. + // blueGreenTx also turns off the checks in this case. + return 0, nil, f.readStorageFromArchive(bytes.NewBuffer(data)) + } + + changed, rowSet, err = f.storage.ImportRoaringRawIterator(rit, clear, true, rowSize) + return } func (tx *RoaringTx) Readonly() bool { @@ -625,8 +761,16 @@ func (tx *RoaringTx) OffsetRange(index, field, view string, shard uint64, offset // getFragment is used by IncrementOpN() and by bitmap() func (tx *RoaringTx) getFragment(index, field, view string, shard uint64) (*fragment, error) { - // If a fragment is attached, always use it. + // If a fragment is attached, always use it. Since it was set at Tx creation, + // it is highly likely to be correct. if tx.fragment != nil { + // but still a basic sanity check. + if tx.fragment.index != index || + tx.fragment.field != field || + tx.fragment.view != view || + tx.fragment.shard != shard { + panic(fmt.Sprintf("different fragment cached vs requested. tx.fragment='%#v', index='%v', field='%v'; view='%v'; shard='%v'", tx.fragment, index, field, view, shard)) + } return tx.fragment, nil } @@ -659,8 +803,10 @@ func (tx *RoaringTx) getFragment(index, field, view string, shard uint64) (*frag } frag := v.Fragment(shard) + if frag == nil { - panic(fmt.Sprintf("fragment not found: %q / %q / %d", field, view, shard)) + return nil, fmt.Errorf("fragment not found: %q / %q / %d", field, view, shard) + //panic(fmt.Sprintf("fragment not found: %q / %q / %d", field, view, shard)) } // Note: we cannot cache frag into tx.fragment. @@ -677,3 +823,52 @@ func (tx *RoaringTx) bitmap(index, field, view string, shard uint64) (*roaring.B } return frag.storage, nil } + +type RoaringStore struct{} + +func NewRoaringStore() *RoaringStore { + return &RoaringStore{} +} + +func (db *RoaringStore) Close() error { + return nil +} + +// frag should be passed by any RoaringTx user, but for RBF/Badger it can be nil. +func (db *RoaringStore) DeleteFragment(index, field, view string, shard uint64, frag interface{}) error { + + fragment, ok := frag.(*fragment) + if !ok { + return fmt.Errorf("RoaringStore.DeleteFragment must get frag of type *fragment, but got '%T'", frag) + } + + // Close data files before deletion. + if err := fragment.Close(); err != nil { + return errors.Wrap(err, "closing fragment") + } + + // Delete fragment file. + if err := os.Remove(fragment.path); err != nil { + return errors.Wrap(err, "deleting fragment file") + } + + // Delete fragment cache file. + if err := os.Remove(fragment.cachePath()); err != nil { + return errors.Wrap(err, fmt.Sprintf("no cache file to delete for shard %d", fragment.shard)) + } + return nil +} + +func (tx *RoaringTx) RoaringBitmapReader(index, field, view string, shard uint64, fragmentPathForRoaring string) (r io.ReadCloser, sz int64, err error) { + file, err := os.Open(fragmentPathForRoaring) // open the fragment file + if err != nil { + return nil, -1, err + } + fi, err := file.Stat() + if err != nil { + return nil, -1, errors.Wrap(err, "statting") + } + sz = fi.Size() + r = file + return +} diff --git a/txfactory.go b/txfactory.go index f8af55c6c..0f49982ec 100644 --- a/txfactory.go +++ b/txfactory.go @@ -22,6 +22,7 @@ import ( "strings" "syscall" + "github.com/pilosa/pilosa/v2/rbf" "github.com/pilosa/pilosa/v2/roaring" "github.com/pkg/errors" ) @@ -55,15 +56,34 @@ var sep = string(os.PathSeparator) type TxFactory struct { typeOfTx txtype - bw *BadgerDBWrapper + badgerDB *BadgerDBWrapper + + rbfDB *rbf.DB + + roaringDB *RoaringStore // could have more than one *Index, but for now keep it simple, // and allow blueGreenTx to report badger contents via idx idx *Index - - // TODO: put RBF database handle here. } +/* want glue-green to multiplex, so don't do this directly +// but rather f.CloseStore() +func (f *TxFactory) Store() TxStore { + switch f.typeOfTx { + case roaringFragmentFilesTxn: + return &RoaringStore{} + case badgerTxn: + return f.badgerDB + case rbfTxn: + return f.rbfDB + // case blueGreenBadgerRoaring: + // case blueGreenRoaringBadger: + } + panic(fmt.Sprintf("unknown f.typeOfTx type: '%v'", f.typeOfTx)) +} +*/ + // integer types for fast switch{} type txtype int @@ -85,6 +105,32 @@ const ( blueGreenRBFBadger txtype = 9 ) +func (txf *TxFactory) NeedsSnapshot() bool { + switch txf.typeOfTx { + case noneTxn: + panic("noneTxn should not occur") + case roaringFragmentFilesTxn: + return true + case badgerTxn: + return false + case rbfTxn: + return false + case blueGreenBadgerRoaring: + return true + case blueGreenRoaringBadger: + return true + case blueGreenRBFRoaring: + return true + case blueGreenRoaringRBF: + return true + case blueGreenBadgerRBF: + return false + case blueGreenRBFBadger: + return false + } + panic(fmt.Sprintf("unknown typeOfTx '%v'", txf.typeOfTx)) +} + func MustTxsrcToTxtype(txsrc string) txtype { switch txsrc { case RoaringTxn: // "roaring" @@ -109,29 +155,52 @@ func MustTxsrcToTxtype(txsrc string) txtype { panic(fmt.Sprintf("unknown txsrc '%v'", txsrc)) } -func newTxFactory(txsrc string, path string) (f *TxFactory, err error) { +// always store files in a subdir of dir. If we are having one +// database or many can depend on name. +func NewTxFactory(txsrc string, dir, name string) (f *TxFactory, err error) { + ty := MustTxsrcToTxtype(txsrc) if ty < 1 || ty > 9 { panic(fmt.Sprintf("invalid txtype '%v'", int(ty))) } - var bw *BadgerDBWrapper - if ty == badgerTxn || ty == 4 || ty == 5 || ty == 8 || ty == 9 { - bw, err = openBadgerDBWrapper(path) + f = &TxFactory{ + typeOfTx: ty, + roaringDB: NewRoaringStore(), + } + switch ty { + case badgerTxn, blueGreenBadgerRoaring, blueGreenRoaringBadger, blueGreenBadgerRBF, blueGreenRBFBadger: + + // one, big, bad-ass badger for all data: the honeyBadger. + // + // Note that having a single Tx backing store for all indexes + // enables cross-index Tx, which are important and are tested for. + path := dir + sep + "honeyBadger" + + f.badgerDB, err = globalBadgerReg.openBadgerDBWrapper(path) // TODO(jea): figure out what the appropriate error path is here. //fmt.Printf("warning: could not open badgerdb on path '%v': '%v'. For safety, we are opening a new '%v-fallback' instead\n", path, err, path+"-fallback") if err != nil { - //bw, err = newBadgerDBWrapper(path + "-fallback") - bw, err = newBadgerDBWrapper(path) + f.badgerDB, err = globalBadgerReg.newBadgerDBWrapper(path) } - panicOn(err) - - bw.doAllocZero = true + if err != nil { + return nil, errors.Wrap(err, fmt.Sprintf("cannot open badger db. path='%v'", path)) + } + // electric-fence like finding of access to mmapped data beyond + // transaction end time. + f.badgerDB.doAllocZero = true } - return &TxFactory{ - typeOfTx: ty, - bw: bw, - }, err + + switch ty { + case rbfTxn, blueGreenRBFRoaring, blueGreenRoaringRBF, blueGreenBadgerRBF, blueGreenRBFBadger: + path := dir + sep + name + ".rbf" + f.rbfDB = rbf.NewDB(path) + if err := f.rbfDB.Open(); err != nil { + return nil, errors.Wrap(err, fmt.Sprintf("cannot open rbf db. path='%v'", path)) + } + } + + return f, err } // Txo holds the transaction options @@ -153,34 +222,35 @@ func (f *TxFactory) DeleteIndex(name string) error { // from holder.go:955, by default is already done there with os.RemoveAll() return nil case badgerTxn: - return f.bw.DeleteIndex(name) + return f.badgerDB.DeleteIndex(name) case rbfTxn: panic("todo rbfTxn DeleteIndex(name)") case blueGreenBadgerRoaring: - return f.bw.DeleteIndex(name) + return f.badgerDB.DeleteIndex(name) case blueGreenRoaringBadger: - return f.bw.DeleteIndex(name) + return f.badgerDB.DeleteIndex(name) } panic(fmt.Sprintf("unknown f.typeOfTx type: '%v'", f.typeOfTx)) } -func (f *TxFactory) Close() error { +func (f *TxFactory) DeleteFragmentFromStore(index, field, view string, shard uint64, frag *fragment) error { switch f.typeOfTx { case roaringFragmentFilesTxn: - return nil + return f.roaringDB.DeleteFragment(index, field, view, shard, frag) case badgerTxn: - // note cannot actually close Badger here. - // causes problems b/c tries holder.DeleteIndex tries to delete the index after db is closed. - //return f.bw.Close() - return nil + return f.badgerDB.DeleteFragment(index, field, view, shard, frag) case rbfTxn: - panic("todo rbfTxn Close()") + //return f.rbfDB.DeleteFragment(index, field, view, shard, frag) + return nil case blueGreenBadgerRoaring: - return nil + _ = f.badgerDB.DeleteFragment(index, field, view, shard, frag) + return f.roaringDB.DeleteFragment(index, field, view, shard, frag) case blueGreenRoaringBadger: - return nil + _ = f.roaringDB.DeleteFragment(index, field, view, shard, frag) + return f.badgerDB.DeleteFragment(index, field, view, shard, frag) } panic(fmt.Sprintf("unknown f.typeOfTx type: '%v'", f.typeOfTx)) + } func (f *TxFactory) CloseIndex(idx *Index) error { @@ -188,10 +258,14 @@ func (f *TxFactory) CloseIndex(idx *Index) error { case roaringFragmentFilesTxn: return nil case badgerTxn: + // note cannot actually close Badger here. + // causes problems b/c tries holder.DeleteIndex tries to delete the index after db is closed. + //return f.badgerDB.Close() return nil case rbfTxn: - panic("todo rbfTxn CloseIndex()") - + // for same reason as above may not be able to close here. + //return f.rbfDB.Close() + return nil case blueGreenBadgerRoaring: return nil case blueGreenRoaringBadger: @@ -202,23 +276,66 @@ func (f *TxFactory) CloseIndex(idx *Index) error { func (f *TxFactory) NewTx(o Txo) Tx { + indexName := "" + if o.Index != nil { + indexName = o.Index.name + } + switch f.typeOfTx { case roaringFragmentFilesTxn: return &RoaringTx{write: o.Write, Field: o.Field, Index: o.Index, fragment: o.Fragment} case badgerTxn: - btx := f.bw.NewBadgerTx(o.Write) + btx := f.badgerDB.NewBadgerTx(o.Write, indexName) return btx case rbfTxn: panic("todo rbfTxn creation") - + /* + rbftx, err := f.rbfDB.Begin(o.Write) + if err != nil { + errors.Wrap(err, "rbfDB.Begin transaction errored") + } + return rbftx + */ case blueGreenBadgerRoaring: - btx := f.bw.NewBadgerTx(o.Write) + btx := f.badgerDB.NewBadgerTx(o.Write, indexName) rtx := &RoaringTx{write: o.Write, Field: o.Field, Index: o.Index, fragment: o.Fragment} return newBlueGreenTx(btx, rtx, f.idx) case blueGreenRoaringBadger: - btx := f.bw.NewBadgerTx(o.Write) + btx := f.badgerDB.NewBadgerTx(o.Write, indexName) rtx := &RoaringTx{write: o.Write, Field: o.Field, Index: o.Index, fragment: o.Fragment} return newBlueGreenTx(rtx, btx, f.idx) + + /* + case blueGreenBadgerRBF: + btx := f.badgerDB.NewBadgerTx(o.Write, indexName) + rbftx, err := f.rbfDB.Begin(o.Write) + if err != nil { + errors.Wrap(err, "rbfDB.Begin transaction errored") + } + return newBlueGreenTx(btx, rbftx, f.idx) + case blueGreenRBFBadger: + btx := f.badgerDB.NewBadgerTx(o.Write, indexName) + rbftx, err := f.rbfDB.Begin(o.Write) + if err != nil { + errors.Wrap(err, "rbfDB.Begin transaction errored") + } + return newBlueGreenTx(rbftx, btx, f.idx) + + case blueGreenRBFRoaring: + rbftx, err := f.rbfDB.Begin(o.Write) + if err != nil { + errors.Wrap(err, "rbfDB.Begin transaction errored") + } + rtx := &RoaringTx{write: o.Write, Field: o.Field, Index: o.Index, fragment: o.Fragment} + return newBlueGreenTx(rbftx, rtx, f.idx) + case blueGreenRoaringRBF: + rbftx, err := f.rbfDB.Begin(o.Write) + if err != nil { + errors.Wrap(err, "rbfDB.Begin transaction errored") + } + rtx := &RoaringTx{write: o.Write, Field: o.Field, Index: o.Index, fragment: o.Fragment} + return newBlueGreenTx(rtx, rbftx, f.idx) + */ } panic(fmt.Sprintf("unknown f.typeOfTx type: '%v'", f.typeOfTx)) } @@ -255,7 +372,7 @@ func (ty txtype) String() string { // Hence to view uncommited keys, you must provide in optionalUseThisTx the // Tx in which they have been added. func (idx *Index) StringifiedBadgerKeys(optionalUseThisTx Tx) string { - return idx.Txf.bw.StringifiedBadgerKeys(optionalUseThisTx) + return idx.Txf.badgerDB.StringifiedBadgerKeys(optionalUseThisTx) } // fragmentSpecFromRoaringPath takes a path releative to the diff --git a/utils_internal_test.go b/utils_internal_test.go index e9603b222..d87499a88 100644 --- a/utils_internal_test.go +++ b/utils_internal_test.go @@ -15,8 +15,6 @@ package pilosa import ( - "bufio" - "bytes" "fmt" "io/ioutil" "path/filepath" @@ -396,9 +394,8 @@ func (b bcast) SendTo(to *Node, m Message) error { return nil } -// FollowResizeInstruction is a version of cluster.FollowResizeInstruction used for testing. +// FollowResizeInstruction is a version of cluster.followResizeInstruction used for testing. func (t *ClusterCluster) FollowResizeInstruction(instr *ResizeInstruction) error { - // Prepare the return message. complete := &ResizeInstructionComplete{ JobID: instr.JobID, @@ -420,7 +417,8 @@ func (t *ClusterCluster) FollowResizeInstruction(instr *ResizeInstruction) error } // Sync available shards. - for _, is := range instr.NodeStatus.Indexes { + for k, is := range instr.NodeStatus.Indexes { + _ = k for _, fs := range is.Fields { f := destCluster.holder.Field(is.Name, fs.Name) @@ -451,23 +449,28 @@ func (t *ClusterCluster) FollowResizeInstruction(instr *ResizeInstruction) error } } - buf := bytes.NewBuffer(nil) + // this is the *test* version of a network call, transferring fragments between + // nodes in a cluster. So it is allowed to be kind of a hack. - bw := bufio.NewWriter(buf) - br := bufio.NewReader(buf) + // there will be two -badgerdb directories/databases, we need to copy + // from src to dest the fragment. This simulates sending the fragment over the network. + srcIdx := srcCluster.holder.Index(src.Index) + srctx := srcIdx.Txf.NewTx(Txo{Write: !writable, Index: srcIdx, Fragment: srcFragment}) - // Get the fragment from source. - if _, err := srcFragment.WriteTo(bw); err != nil { - return err - } - - // Flush the bufio.buf to the io.Writer (buf). - bw.Flush() - - // Write data to destination. - if _, err := destFragment.ReadFrom(br); err != nil { - return err + destIdx := destCluster.holder.Index(src.Index) + desttx := destIdx.Txf.NewTx(Txo{Write: writable, Index: destIdx, Fragment: destFragment}) + + citer, _, err := srctx.ContainerIterator(src.Index, src.Field, src.View, src.Shard, 0) + panicOn(err) + d := destFragment + for citer.Next() { + ckey, c := citer.Value() + err := desttx.PutContainer(d.index, d.field, d.view, d.shard, ckey, c) + panicOn(err) } + citer.Close() + panicOn(desttx.Commit()) + srctx.Rollback() } return nil diff --git a/view.go b/view.go index 12210a619..106ae2b42 100644 --- a/view.go +++ b/view.go @@ -50,6 +50,7 @@ type view struct { qualifiedName string holder *Holder + idx *Index fieldType string cacheType string @@ -143,7 +144,8 @@ func (v *view) open() error { } v.holder.Logger.Debugf("open fragments for index/field/view: %s/%s/%s", v.index, v.field, v.name) - if err := v.openFragments(); err != nil { + + if err := v.openFragmentsInTx(); err != nil { return errors.Wrap(err, "opening fragments") } @@ -159,64 +161,27 @@ func (v *view) open() error { var workQueue = make(chan struct{}, runtime.NumCPU()*2) -// openFragments opens and initializes the fragments inside the view. -func (v *view) openFragments() error { - file, err := os.Open(filepath.Join(v.path, "fragments")) - if os.IsNotExist(err) { - return nil - } else if err != nil { - return errors.Wrap(err, "opening fragments directory") - } - defer file.Close() +// replaces v.openFragments() with Tx generic code. +func (v *view) openFragmentsInTx() error { - fis, err := file.Readdir(0) + tx := v.idx.Txf.NewTx(Txo{Write: !writable, Index: v.idx}) + defer tx.Rollback() + + shards, err := tx.SliceOfShards(v.index, v.field, v.name, v.path) if err != nil { - return errors.Wrap(err, "reading fragments directory") + return errors.Wrap(err, "SliceOfShards") } - - eg, ctx := errgroup.WithContext(context.Background()) - var mu sync.Mutex - -fileLoop: - for _, loopFi := range fis { - select { - case <-ctx.Done(): - break fileLoop - default: - fi := loopFi - - if fi.IsDir() { - continue - } - - // Parse filename into integer. - shard, err := strconv.ParseUint(filepath.Base(fi.Name()), 10, 64) - if err != nil { - v.holder.Logger.Debugf("WARNING: couldn't use non-integer file as shard in index/field/view %s/%s/%s: %s", v.index, v.field, v.name, fi.Name()) - continue - } - - workQueue <- struct{}{} - v.holder.Logger.Debugf("open index/field/view/fragment: %s/%s/%s/%d", v.index, v.field, v.name, shard) - eg.Go(func() error { - defer func() { - <-workQueue - }() - frag := v.newFragment(v.fragmentPath(shard), shard) - if err := frag.Open(); err != nil { - return fmt.Errorf("open fragment: shard=%d, err=%s", frag.shard, err) - } - frag.RowAttrStore = v.rowAttrStore - v.holder.Logger.Debugf("add index/field/view/fragment to view.fragments: %s/%s/%s/%d", v.index, v.field, v.name, shard) - mu.Lock() - v.fragments[frag.shard] = frag - v.addKnownShard(frag.shard) - mu.Unlock() - return nil - }) + for _, shard := range shards { + frag := v.newFragment(v.fragmentPath(shard), shard) + if err := frag.Open(); err != nil { + return fmt.Errorf("open fragment: shard=%d, err=%s", frag.shard, err) } + frag.RowAttrStore = v.rowAttrStore + v.holder.Logger.Debugf("add index/field/view/fragment to view.fragments: %s/%s/%s/%d", v.index, v.field, v.name, shard) + v.fragments[frag.shard] = frag + v.addKnownShard(frag.shard) } - return eg.Wait() + return nil } // close closes the view and its fragments. @@ -359,6 +324,16 @@ func (v *view) notifyIfNewShard(shard uint64) { } func (v *view) newFragment(path string, shard uint64) *fragment { + + if v.holder != nil && v.idx != nil { + // A view must have its v.idx *Index registered with its holder. + // Otherwise TestField_AvailableShards crashes, as one example. + hIdx := v.holder.Index(v.idx.name) + if hIdx == nil && v.idx != nil { + v.holder.addIndexFromField(v.idx) + } + } + frag := newFragment(v.holder, path, v.index, v.field, v.name, shard, v.flags()) frag.CacheType = v.cacheType frag.CacheSize = v.cacheSize @@ -375,28 +350,17 @@ func (v *view) newFragment(path string, shard uint64) *fragment { func (v *view) deleteFragment(shard uint64) error { v.mu.Lock() defer v.mu.Unlock() - fragment := v.fragments[shard] - if fragment == nil { + f := v.fragments[shard] + if f == nil { return ErrFragmentNotFound } v.holder.Logger.Printf("delete fragment: (%s/%s/%s) %d", v.index, v.field, v.name, shard) - // Close data files before deletion. - if err := fragment.Close(); err != nil { - return errors.Wrap(err, "closing fragment") + idx := f.holder.Index(v.index) + if err := idx.Txf.DeleteFragmentFromStore(f.index, f.field, f.view, f.shard, f); err != nil { + return errors.Wrap(err, "DeleteFragment") } - - // Delete fragment file. - if err := os.Remove(fragment.path); err != nil { - return errors.Wrap(err, "deleting fragment file") - } - - // Delete fragment cache file. - if err := os.Remove(fragment.cachePath()); err != nil { - v.holder.Logger.Printf("no cache file to delete for shard %d", shard) - } - delete(v.fragments, shard) v.removeKnownShard(shard) diff --git a/view_internal_test.go b/view_internal_test.go index bbb1a20f4..bcde5f871 100644 --- a/view_internal_test.go +++ b/view_internal_test.go @@ -34,7 +34,15 @@ func mustOpenView(index, field, name string) *view { CacheSize: DefaultCacheSize, } - v := newView(NewHolder(DefaultPartitionN), path, index, field, name, fo) + h := NewHolder(DefaultPartitionN) + h.Path = path + // h needs an *Index so we can call h.Index() and get Index.Txf, in TestView_DeleteFragment + idx, err := h.createIndex(index, IndexOptions{}) + _ = idx + panicOn(err) + + v := newView(h, path, index, field, name, fo) + v.idx = idx if err := v.open(); err != nil { panic(err) } diff --git a/vprint.go b/vprint.go index bc40dfe45..52130bbeb 100644 --- a/vprint.go +++ b/vprint.go @@ -111,3 +111,33 @@ func FileLine(depth int) string { func stack() string { return string(debug.Stack()) } + +func FileExists(name string) bool { + fi, err := os.Stat(name) + if err != nil { + return false + } + if fi.IsDir() { + return false + } + return true +} + +func DirExists(name string) bool { + fi, err := os.Stat(name) + if err != nil { + return false + } + if fi.IsDir() { + return true + } + return false +} + +func FileSize(name string) (int64, error) { + fi, err := os.Stat(name) + if err != nil { + return -1, err + } + return fi.Size(), nil +}