diff --git a/cmd/pilosa-fsck/fsck.go b/cmd/pilosa-fsck/fsck.go index daf5b7b69..d0a307c92 100644 --- a/cmd/pilosa-fsck/fsck.go +++ b/cmd/pilosa-fsck/fsck.go @@ -79,7 +79,7 @@ type FsckConfig struct { func (cfg *FsckConfig) DefineFlags(fs *flag.FlagSet) { fs.BoolVar(&cfg.Fix, "fix", false, "(warning: alters the backed-up node images on disk) copy primary data to replicas to create a consistent cluster. Implies -fixcol") fs.BoolVar(&cfg.FixCol, "fixcol", false, "(warning: alters the backed-up node images on disk) repair string key translation tables. Skip repair of index data.") - fs.BoolVar(&cfg.Verbose, "v", false, "be very verbose during analysis") + //fs.BoolVar(&cfg.Verbose, "v", false, "be very verbose during analysis") fs.BoolVar(&cfg.Quiet, "q", false, "be very quiet") fs.IntVar(&cfg.ReplicaN, "replicas", 0, "(required) manually entered replicaN; the number of replicas maintained in the cluster. Must be the same as the [cluster] 'replicas = R' entry in the pilosa.conf file for the cluster.") @@ -88,44 +88,20 @@ func (cfg *FsckConfig) DefineFlags(fs *flag.FlagSet) { fs.Usage = func() { fmt.Fprintf(os.Stderr, "pilosa-fsck version: %v\n\n", pilosa.VersionInfo()) - fmt.Fprintf(os.Stderr, `Use: pilosa-fsck -replicas R {-fix} {-fixcol} {-q} {-v} /backup/1/.pilosa /backup/2/.pilosa ... /backup/N/.pilosa + fmt.Fprintf(os.Stderr, `Use: pilosa-fsck -replicas R {-fix} {-q} /backup/1/.pilosa /backup/2/.pilosa ... /backup/N/.pilosa -fix - (warning: alters the backed-up node images on disk) copy primary data to replicas to create a consistent cluster. Implies -fixcol - - -fixcol - (warning: alters the backed-up node images on disk) repair string key translation tables. Skip repair of index data. + (warning: alters the backed-up node images on disk) copy primary data to replicas to create a consistent cluster. -replicas R (required) R is a positive integer, giving the replicaN or replicator factor for the cluster. This is the number of replicas maintained in the cluster. Must be the same as the [cluster] 'replicas = R' entry shared across all the pilosa.conf files on each node. - -v - be very verbose during analysis -q be very quiet during analysis and repair `) - /* - key translation dump options, usually only used by developers debugging key translation: - - -col - (optional) display column keys (very long output) - -dir string - (optional; requires -col), one pilosa data dir to read (default "/home/ubuntu/.pilosa") - -header - (optional; requires -col), display header - -id - (optional; requires -col), dump reverse mapping id->key - -index string - (optional; requires -col), index name (default "i") - -key - (optional; requires -col), dump forward mapping key->id - -partition int - (optional; requires -col), partition id to dump - - */ fmt.Fprintf(os.Stderr, ` Welcome to pilosa-fsck. This is a scan and repair tool that is modeled after the classic unix file @@ -140,28 +116,19 @@ Just as fsck must be run on an unmounted disk, pilosa-fsck must be run on a backup. It must not be run on the directories where a live Pilosa system is serving queries. Instead, take a backup first. -A backup is a set of N cluster-node directories that have been +A backup is a set of N Pilosa data directories that have been copied from your live system. They must all be visible and mounted on one filesystem together. -pilosa-fsck can be run in scan-mode (without -fix or -fixcol), -or in repair-mode with -fix (or -fixcol). The console output -supplies a shell script documenting the analysis -and showing what data changes would be made. If a -fix has been requested, those fixes will have -been applied during the run. The output then serves -as documentation of what has been updated. If -a fix has not been requested (in other words, neither --fix nor -fixcol was given) then no changes will -have been made to the backups. The fragment level -sync can be completed next by running the script -if you wish. The -fixcol fixes can only be -applied by doing a -fixcol run of pilosa-fsck. +pilosa-fsck can be run in scan-mode (without -fix), +or in repair-mode with -fix. The console output +supplies a log documenting the analysis +and showing what data changes would have been made. REQUIRED COMMAND LINE ARGUMENTS -The paths to all the top-level pilosa -directories in a cluster must be given on the command +The paths to all the top-level Pilosa +data directories in a cluster must be given on the command line. The -replicas R flag is also always required. It must be correct for your cluser. Here R is the same as the [cluster] stanza "replicas = R" line from your @@ -204,34 +171,32 @@ subdirectories node1/ node2/ node3/ node4/ under this: NOTE: your .pilosa directories need not be named .pilosa. They can be something else, such as when the -d flag to pilosa server was used. -The .id and .topology and index directories must be found directly underneath. +The .id file, the .topology file, and the index directories must be +found directly underneath. Then a typical invocation to scan a cluster backup for issues: $ cd /backup/molecula/ -$ pilosa-fsck -replicas 3 node1/.pilosa node2/.pilosa node3/.pilosa node4/.pilosa +$ pilosa-fsck -replicas 3 node1/.pilosa node2/.pilosa node3/.pilosa node4/.pilosa &> log A typical invocation to repair the replication in the same backup: -$ pilosa-fsck -replicas 3 -fix node1/.pilosa node2/.pilosa node3/.pilosa node4/.pilosa +$ pilosa-fsck -replicas 3 -fix node1/.pilosa node2/.pilosa node3/.pilosa node4/.pilosa &> log In both cases, the .id and .topology files must be present in the backups. -KEY REPAIR NOTE +Without -fix, no modifications will be made to the backups. Only +by running with -fix will repairs be made. The user can safely +always run with -fix to repair only if needed. -While the output of pilosa-fsck wihtout -fix or -fixcol -gives a script showing the index fragment repair operations -that can be applied by (cp/rm) shell commands subsequently, -this script alone is an incomplete repair. It does not -address string key tranlsation repairs. For a complete repair, -a run of pilosa-fsck with the -fixcol or -fix flags -will be required. +A zero error code will be returned to the shell if no repairs were needed. -A note about the -fixcol key translation repairs: these are fine -grained operations on the internal databases that do not have -corresponding (cp/rm) shell commands. Therefore a run of pilosa-fsck -with the -fix or -fixcol flag is required to repair these. +A zero error code will be also be returned to the shell if +repairs were needed and they were accomplished under -fix. + +A non-zero error code indicates that repairs were needed but +were not made. `) } } @@ -295,6 +260,7 @@ func main() { myflags := flag.NewFlagSet(ProgramName, flag.ContinueOnError) cfg := &FsckConfig{} cfg.DefineFlags(myflags) + cfg.Verbose = true err := myflags.Parse(os.Args[1:]) if err != nil { @@ -341,11 +307,15 @@ func main() { }() cfg.Dirs = dirs - _, err = cfg.Run() + fixNeeded, err := cfg.Run() if err != nil { fmt.Fprintf(os.Stderr, "error: %v\n", err) os.Exit(1) } + if fixNeeded && !cfg.Fix { + fmt.Fprintf(os.Stderr, "# pilosa-fsck exiting with non-zero error code because a repair is needed, but -fix was not given.\n") + os.Exit(1) + } } func (cfg *FsckConfig) Run() (fixNeeded bool, err error) { diff --git a/cmd/pilosa-fsck/release-pilosa-fsck/DESIGN.md b/cmd/pilosa-fsck/release-pilosa-fsck/DESIGN.md new file mode 100644 index 000000000..4fbaffe31 --- /dev/null +++ b/cmd/pilosa-fsck/release-pilosa-fsck/DESIGN.md @@ -0,0 +1,252 @@ +Design for pilosa-fsck +====================== + +Problem Background +------------------ + +Molecula Pilosa provides replication for fault-tolerance within a Pilosa cluster. + +Three kinds of data are replicated: Roaring bitmap data, Column-Key translation data, +and Row-Key data are replicated. Only the first two, Roaring data and Column-Key +data are relevant here. Broadly, the Roaring bitmap data +forms the central features -- the bits -- of a large, sparse bitmap matrix. +The Column-Keys are the labels for the columns at the top margin of this matrix. + +For speed, the Roaring bitmap data is stored separately from the +Key data. The Roaring data is stored in sharded files +within a directory heirarchy under PILOSA-DATA-DIR/index_name/field_name/... +The Key translation data is stored in sharded BoltDB databases within +the PILOSA-DATA-DIR/index_name/_key directory. + +The current approach to Roaring file replication involves an +eventually consistent mechanism that uses an Anti-Entropy agent to +fix partial or incomplete replication from the primary shard to all +replica shards. + +Unfortunately, the Anti-Entropy agent approach has proved inadequate on two +fronts. First, it does not provide for immediately consistent reads in the +event that the primary is lost. Second, the Anti-Entropy agent itself experienced +out-of-memory issues that have yet to be resolved. + +Therefore, work is now underway to replace this replication +approach with a more consistent design. + +However, in the meantime, for our customers in production with Molecula +Pilosa, we wish to provide a means to re-establish correct replication. +Thus even in the event of a node failure followed by a read from a replica, the +returned read will be correct. + +The pilosa-fsck tool can therefore been seen as a temporary, stop-gap +measure to address immediate issues while the cluster replication +mechanism is replaced. + +The second factor motivating the creation of pilosa-fsck was the discovery +of a bug in the Key-translation process. Unfortunately this was a hard +to reproduce bug. It happened only on the customer's premises. +It happened only after running the system for a long time, with a +large amount of data, and with various eccentric node failures +and recoveries. + +However, we were able to reproduce a plausible explanation. +Non-primary replicas were creating keys when they should have been +forwarding the request to the primary. Correcting this bug is impetus +for the v2.1.4 release of Molecula Pilosa. + +A fine point here: since we were not able to precisely reproduce the customer's +issue in the development environment, we cannot guarantee with 100% +certainty that we have actually addressed the bug that the customer +was seeing. + +Therefore we also desired an additional insurance +policy. We wished to be able to empower customers to pro-actively discover any +future Key-translation issues that happen in their on-premise systems. + +To do this, we proposed providing select customers with the pilosa-fsck +tool which can analyze their offline backups for issues. + +Optionally, these issues can also be repaired in-place in the +offline backup on which pilosa-fsck is run. + +The -fix flag repairs both kinds of replication issues. + +Solution Approach: mechanism of action +-------------------------------------- + +The pilosa-fsck is run offline on a full set of backups taken from +all nodes in a Pilosa cluster. It runs on a single computer that +must be separate from the production or staging Pilosa environments. + +When run, pilosa-fsck analyzes the differences between the +primary and its replicas. Both the Roaring +files and the Key translation databases are analyzed. +The computer running pilosa-fsck must have the same or more +memory as the Pilosa nodes in the cluster, as it will +"pretend" to be each Pilosa node in turn. However, as each +node's backup is closed before the next node's backup is +opened, we do not require substantially more memory than a single +production node. Short Blake3 cryptographic checksums are +computed for each Roaring fragment and each Key translation +database. These are held in memory (and printed to the log) +for comparing nodes. This comparison forms the heart of +the consistency checks, and is the basis for any subsequent +repair. + +We recommend capturing both stdout and stderr to a log. +Use `&> log` or `2>&1 > log` at the end of the +pilosa-fsck invocation to save a log of the run to disk. + +In a typical cluster, the Replication factor R may be less +than the number nodes N in the cluster. For example, while +N may be 4, the R may be only 3. In this example, within +each replicated shard, one node will be the primary for +that shard, two nodes will be "regular" non-primary replicas, and one +node will be a non-replica. Note that the designation +of primary changes for different Roaring shards within an index, +even on a single node. + +The essence of the the -fix repair operation that pilosa-fsck +can do is this: it will copy from the primary to the +the non-primary replicas. Further, it will remove data from +any non-replica node if it was mistakenly present. + +The pilosa-fsck output log will contain +a sequence of command line 'cp' and 'rm' commands. +These commands are merely a record (with +accompanying justifcation in the comment following the +command) of what actions would be performed to repair +the Roaring file data. + +Only with -fix will the repair actions actually happen +during the pilosa-fsck run. + + +Details: running pilosa-fsck +---------------------------- + +Errors in invocation are reported on stderr and the program will exit with a non-zero +error code if invocation errors are present. A non-zero error code +is returned if a repair is needed and -fix was not given. + +A -fix run will return a zero error code to the shell, if the fix was +successfully made; or if no fix was required. + +The log of the run is printed to stdout. + +The -h flag to pilosa-fsck prints a summary of its operation +and a guide to laying out the backup directories. + +The help is reproduced below. + +~~~ +$ pilosa-fsck version: Molecula Pilosa v2.2.1-43-g9dacbccf (Oct 5 2020 1:28PM, 9dacbccf) + +Use: pilosa-fsck -replicas R {-fix} {-q} /backup/1/.pilosa /backup/2/.pilosa ... /backup/N/.pilosa + + -fix + (warning: alters the backed-up node images on disk) copy primary data to replicas to create a consistent cluster. + + -replicas R + (required) R is a positive integer, giving the replicaN or replicator factor for the cluster. This is + the number of replicas maintained in the cluster. Must be the same as the + [cluster] 'replicas = R' entry shared across all the pilosa.conf files on each node. + + -q + be very quiet during analysis and repair + + +Welcome to pilosa-fsck. This is a scan and repair +tool that is modeled after the classic unix file +system utility fsck. + +WARNING: DO NOT RUN ON A LIVE SYSTEM. + +The most important point to remember is that analysis +and repair must be done *offline*. + +Just as fsck must be run on an unmounted disk, +pilosa-fsck must be run on a backup. It must +not be run on the directories where a live Pilosa system +is serving queries. Instead, take a backup first. +A backup is a set of N Pilosa data directories that have been +copied from your live system. They must all +be visible and mounted on one filesystem together. + +pilosa-fsck can be run in scan-mode (without -fix), +or in repair-mode with -fix. The console output +supplies a log documenting the analysis +and showing what data changes would have been made. + +REQUIRED COMMAND LINE ARGUMENTS + +The paths to all the top-level Pilosa +data directories in a cluster must be given on the command +line. The -replicas R flag is also always required. It +must be correct for your cluser. Here R is the same as +the [cluster] stanza "replicas = R" line from your +pilosa.conf. + +Example: + +Suppose you are ready to run pilosa-fsck: +you have taken a backup of your four node Pilosa +cluster and stored it all on one filesystem with +all nodes visible and uncompressed. This +is a pre-requisite to running pilosa-fsck. +Let's suppose we have replication R = 3 set. +In this example, have stored our backed-up directories in + +/backup/molecula + +and the four node backups are in +subdirectories node1/ node2/ node3/ node4/ under this: + +/backup/molecula/node1/ +/backup/molecula/node1/.pilosa/.id +/backup/molecula/node1/.pilosa/.topology +/backup/molecula/node1/.pilosa/myindex + +/backup/molecula/node2/ +/backup/molecula/node2/.pilosa/.id +/backup/molecula/node2/.pilosa/.topology +/backup/molecula/node2/.pilosa/myindex + +/backup/molecula/node3/ +/backup/molecula/node3/.pilosa/.id +/backup/molecula/node3/.pilosa/.topology +/backup/molecula/node3/.pilosa/myindex + +/backup/molecula/node4/ +/backup/molecula/node4/.pilosa/.id +/backup/molecula/node4/.pilosa/.topology +/backup/molecula/node4/.pilosa/myindex + +NOTE: your .pilosa directories need not be named .pilosa. They can +be something else, such as when the -d flag to pilosa server was used. +The .id file, the .topology file, and the index directories must be +found directly underneath. + +Then a typical invocation to scan a cluster backup for issues: + +$ cd /backup/molecula/ +$ pilosa-fsck -replicas 3 node1/.pilosa node2/.pilosa node3/.pilosa node4/.pilosa &> log + +A typical invocation to repair the replication in the same backup: + +$ pilosa-fsck -replicas 3 -fix node1/.pilosa node2/.pilosa node3/.pilosa node4/.pilosa &> log + +In both cases, the .id and .topology files must +be present in the backups. + +Without -fix, no modifications will be made to the backups. Only +by running with -fix will repairs be made. The user can safely +always run with -fix to repair only if needed. + +A zero error code will be returned to the shell if no repairs were needed. + +A zero error code will be also be returned to the shell if +repairs were needed and they were accomplished under -fix. + +A non-zero error code indicates that repairs were needed but +were not made. + +~~~