featurebase/sql3/test/defs/defs_keyed.go
Travis Turner a44b622aa0 Introduce ServiceManager and Refactor DAX Integration tests (#2320)
* Introduce ServiceManager and Refactor DAX Integration tests

The ServiceManager provides an interface with which to manage
featurebase (dax) services (mds, queryer, computer). It replaces the
confusing interface implementations in /dax/server/server.go (which
optionally used pointers to in-process objects to satisfy an interface)
with (for now) http implementations. The thought is that even if we're
running all services in-process, we should communicate between services
over http in order to mirror what we would do in a production
environment where the services are running on different nodes.

This batch of commits does quit a lot, most of which is captured here:

- Added `path` support to `dax.Address`. Address is now a string of the form [scheme]://[host]:[port]/[path].
- Added `Holder.directiveApplied` to determine (in tests) if the computer has completed applying the latest directive. This is somewhat temporary until we improve the mds-to-computer logic.
- Removed the "service prefix" code which was prepending client URL paths with the prefix. Instead, the serviceType (mds, queryer, computer[n] is now part of `dax.Address`).
- Removed, from the dax config, the top level `StorageMethod` and `StorageDSN` and now just have `MDS.Config.DataDir`.
- Added `Computer.Config.N` to specify the number of computers to run in-process.
- Moved the `pilosa.MDS` interface to `computer.Registrar`. This is an example of getting the interfaces defined in the right packages.
- Added `SnapshotTable()` method to the mds client (to align with its API).
- Changed `Balancer.AddJob()` to `Balancer.AddJobs()` to support, for example, adding 256 partitions in a single call. Refactored some of the naive Balancer to account for this.
- Added a `Seed` to the top-level config. It's not really useful because of package `crypto/rand`.
- Added an in-memory implementation of the DisCo interface and disabled etcd in a computer service.
- Create sepearte data-dirs for each in-process computer.
- Disabled grpc in dax.
- Modified the sql3 test definition format to support multiple insert steps and separate query results (to align with those steps).

* Changes necessary to get multiple computer instance running in-process

For now the config looks like this:

```
[computer]
run = true
n = 4
```

but we can probably just change that to be something like:

```
[computer]
run = 4
```

*Issues found running multiple "computers" in-process*
- grpc was trying to bind on the same port
  - changed GRPCListener from `*net.TCPListener` to `net.Listener`
  - created a nopListener and set to that for now (i.e. disabled grpc)
- etcd was starting more than once
  - changed dax to use in-memory implementations of the disco interfaces (i.e. stop using etcd)
- IDAllocator (which uses boltdb) was trying to open the `idalloc.db` file more than once
  - realized we have to set separate data-dirs for each holder. that fixed it.

* Port dax integration tests to ManagedCommand

* Modify Balancer-related methods like AddJob to AddJobs

There were (and still are) a lot of places where we were adding on job
at a time, even when we had a long list of jobs to add. This resulted in
every job add (for example adding 1 of 256 shards) taking ~40ms, or over
10s to create a keyed table. One reason was because each job add was
making multiple boltdb transactions.

* Port over more dax integration test stuff

* Add DirectiveApplied to signify that snapshot/writes have loaded.

We use this in tests to avoid using sleeps.
This should be considered temporary; we're going to need a more robust
solution for determining when a computer node is ready to serve complete
data.

* Finish porting dax integration tests

* Improve godocs

* Remove docker-based DAX integration tests.

* go mod tidy

* Move test/managed.go to avoid package conflicts

* Modify IDK integration tests to work with ServiceManager changes

This is really just computer -> computer0
And the MDS DataDir config change.

* cleanup found during review

* echo $CI_COMMIT_REF_SLUG in CI

* remove docker image arg, use build instead

(cherry picked from commit 2843f218bc)
2022-12-12 09:01:20 -08:00

227 lines
6.3 KiB
Go

package defs
var Keyed TableTest = keyed
var keyed = TableTest{
Table: tbl(
"keyed",
srcHdrs(
srcHdr("_id", fldTypeString),
srcHdr("an_int", fldTypeInt, "min 0", "max 100"),
srcHdr("an_id_set", fldTypeIDSet),
srcHdr("an_id", fldTypeID),
srcHdr("a_string", fldTypeString),
srcHdr("a_string_set", fldTypeStringSet),
),
srcRows(
srcRow("one", int64(11), []int64{11, 12, 13}, int64(101), "str1", []string{"a1", "b1", "c1"}),
srcRow("two", int64(22), []int64{11, 12, 23}, int64(201), "str2", []string{"a2", "b2", "c2"}),
srcRow("three", int64(33), []int64{11, 32, 33}, int64(301), "str3", []string{"a3", "b3", "c3"}),
srcRow("four", int64(44), []int64{41, 42, 43}, int64(401), "str4", []string{"a4", "b4", "c4"}),
),
srcRows(
srcRow("five", int64(55), []int64{51, 52, 53}, int64(501), "str5", []string{"a5", "b5", "c5"}),
srcRow("six", int64(66), []int64{61, 62, 63}, int64(601), "str6", []string{"a6", "b6", "c6"}),
),
),
SQLTests: []SQLTest{
{
// Select all.
name: "select-all",
SQLs: sqls(
"select * from keyed",
"select _id, an_int, an_id_set, an_id, a_string, a_string_set from keyed",
),
ExpHdrs: hdrs(
hdr("_id", fldTypeString),
hdr("an_int", fldTypeInt),
hdr("an_id_set", fldTypeIDSet),
hdr("an_id", fldTypeID),
hdr("a_string", fldTypeString),
hdr("a_string_set", fldTypeStringSet),
),
ExpRows: rows(
row("one", int64(11), []int64{11, 12, 13}, int64(101), "str1", []string{"a1", "b1", "c1"}),
row("two", int64(22), []int64{11, 12, 23}, int64(201), "str2", []string{"a2", "b2", "c2"}),
row("three", int64(33), []int64{11, 32, 33}, int64(301), "str3", []string{"a3", "b3", "c3"}),
row("four", int64(44), []int64{41, 42, 43}, int64(401), "str4", []string{"a4", "b4", "c4"}),
),
ExpRowsPlus1: rowSets(
rows(
row("one", int64(11), []int64{11, 12, 13}, int64(101), "str1", []string{"a1", "b1", "c1"}),
row("two", int64(22), []int64{11, 12, 23}, int64(201), "str2", []string{"a2", "b2", "c2"}),
row("three", int64(33), []int64{11, 32, 33}, int64(301), "str3", []string{"a3", "b3", "c3"}),
row("four", int64(44), []int64{41, 42, 43}, int64(401), "str4", []string{"a4", "b4", "c4"}),
row("five", int64(55), []int64{51, 52, 53}, int64(501), "str5", []string{"a5", "b5", "c5"}),
row("six", int64(66), []int64{61, 62, 63}, int64(601), "str6", []string{"a6", "b6", "c6"}),
),
),
Compare: CompareExactUnordered,
SortStringKeys: true,
},
{
// Select all with top.
name: "select-all-with-top",
SQLs: sqls(
"select top(2) * from keyed",
),
ExpHdrs: hdrs(
hdr("_id", fldTypeString),
hdr("an_int", fldTypeInt),
hdr("an_id_set", fldTypeIDSet),
hdr("an_id", fldTypeID),
hdr("a_string", fldTypeString),
hdr("a_string_set", fldTypeStringSet),
),
ExpRows: rows(
row("one", int64(11), []int64{11, 12, 13}, int64(101), "str1", []string{"a1", "b1", "c1"}),
row("two", int64(22), []int64{11, 12, 23}, int64(201), "str2", []string{"a2", "b2", "c2"}),
row("three", int64(33), []int64{11, 32, 33}, int64(301), "str3", []string{"a3", "b3", "c3"}),
row("four", int64(44), []int64{41, 42, 43}, int64(401), "str4", []string{"a4", "b4", "c4"}),
),
Compare: CompareIncludedIn,
SortStringKeys: true,
ExpRowCount: 2,
},
{
// Select all with where on int field.
name: "select-all-with-where",
SQLs: sqls(
"select * from keyed where an_int = 22",
"select * from keyed where a_string = 'str2'",
"select * from keyed where an_id = 201",
),
ExpHdrs: hdrs(
hdr("_id", fldTypeString),
hdr("an_int", fldTypeInt),
hdr("an_id_set", fldTypeIDSet),
hdr("an_id", fldTypeID),
hdr("a_string", fldTypeString),
hdr("a_string_set", fldTypeStringSet),
),
ExpRows: rows(
row("two", int64(22), []int64{11, 12, 23}, int64(201), "str2", []string{"a2", "b2", "c2"}),
),
Compare: CompareExactUnordered,
SortStringKeys: true,
},
},
PQLTests: []PQLTest{
{
name: "minrow",
Table: "keyed",
PQLs: []string{"MinRow(field=an_id_set)"},
ExpHdrs: hdrs(
hdr("an_id_set", fldTypeID),
hdr("count", fldTypeID),
),
ExpRows: rows(
row(int64(11), int64(1)),
),
},
{
name: "maxrow",
Table: "keyed",
PQLs: []string{"MaxRow(field=an_id_set)"},
ExpHdrs: hdrs(
hdr("an_id_set", fldTypeID),
hdr("count", fldTypeID),
),
ExpRows: rows(
row(int64(43), int64(1)),
),
},
{
name: "topk",
Table: "keyed",
PQLs: []string{"TopK(an_id_set, k=2)"},
ExpHdrs: hdrs(
hdr("an_id_set", fldTypeID),
hdr("count", fldTypeID),
),
ExpRows: rows(
row(int64(11), int64(3)),
row(int64(12), int64(2)),
),
},
// TODO(tlt): figure out why this sometimes fails on multi-node setups
// {
// name: "topn",
// Table: "keyed",
// PQLs: []string{"TopN(an_id_set, n=2)"},
// ExpHdrs: hdrs(
// hdr("an_id_set", fldTypeID),
// hdr("count", fldTypeID),
// ),
// ExpRows: rows(
// row(int64(11), int64(3)),
// row(int64(12), int64(2)),
// ),
// },
{
name: "rows",
Table: "keyed",
PQLs: []string{"Rows(field=an_id_set)"},
ExpHdrs: hdrs(
hdr("an_id_set", fldTypeID),
),
ExpRows: rows(
row(int64(11)),
row(int64(12)),
row(int64(13)),
row(int64(23)),
row(int64(32)),
row(int64(33)),
row(int64(41)),
row(int64(42)),
row(int64(43)),
),
},
{
name: "includescolumn",
Table: "keyed",
PQLs: []string{"IncludesColumn(Row(an_id_set=12), column='two')"},
ExpHdrs: hdrs(
hdr("result", fldTypeBool),
),
ExpRows: rows(
row(true),
),
},
{
name: "constrow",
Table: "keyed",
PQLs: []string{"Extract(ConstRow(columns=['two']), Rows(an_id))"},
ExpHdrs: hdrs(
hdr("_id", fldTypeString),
hdr("an_id", fldTypeID),
),
ExpRows: rows(
row("two", int64(201)),
),
},
{
name: "fieldvalue",
Table: "keyed",
PQLs: []string{"FieldValue(field=an_int, column='three')"},
ExpHdrs: hdrs(
hdr("value", fldTypeInt),
hdr("count", fldTypeInt),
),
ExpRows: rows(
row(int64(33), int64(1)),
),
},
{
name: "unionrows",
Table: "keyed",
PQLs: []string{"Count(UnionRows(Rows(field=an_id_set)))"},
ExpHdrs: hdrs(
hdr("count", fldTypeID),
),
ExpRows: rows(
row(int64(4)),
),
},
},
}