mirror of
https://github.com/featurebasedb/featurebase.git
synced 2026-08-28 10:54:59 +00:00
We recover from some specific panics deeper in the PEG parser, but when we added the invalid timestamp, we didn't add it to the list we catch and handle gracefully. Add test case for this, and test case for successful parsing. Also add the word "valid" to the error message so people don't get as confused by it.
161 lines
3.7 KiB
Go
161 lines
3.7 KiB
Go
// Copyright 2021 Molecula Corp. All rights reserved.
|
|
package pql
|
|
|
|
import (
|
|
"fmt"
|
|
"io"
|
|
"io/ioutil"
|
|
"strconv"
|
|
"strings"
|
|
"unicode/utf8"
|
|
|
|
"github.com/pkg/errors"
|
|
)
|
|
|
|
// error strings in the parser
|
|
const duplicateArgErrorMessage = "duplicate argument provided"
|
|
const intOutOfRangeError = "integer is not in signed 64-bit range"
|
|
const invalidTimestampError = "string is not a valid timestamp"
|
|
|
|
// parser represents a parser for the PQL language.
|
|
type parser struct {
|
|
r io.Reader
|
|
//scanner *bufScanner
|
|
PQL
|
|
}
|
|
|
|
// NewParser returns a new instance of Parser.
|
|
func NewParser(r io.Reader) *parser {
|
|
return &parser{
|
|
r: r,
|
|
// scanner: newBufScanner(r),
|
|
}
|
|
}
|
|
|
|
// ParseString parses s into a query.
|
|
func ParseString(s string) (*Query, error) {
|
|
return NewParser(strings.NewReader(s)).Parse()
|
|
}
|
|
|
|
// Parse parses the next node in the query.
|
|
func (p *parser) Parse() (*Query, error) {
|
|
buf, err := ioutil.ReadAll(p.r)
|
|
if err != nil {
|
|
return nil, errors.Wrap(err, "reading buffer to parse")
|
|
}
|
|
p.PQL = PQL{
|
|
Buffer: string(buf),
|
|
}
|
|
err = p.Init()
|
|
if err != nil {
|
|
return nil, errors.Wrap(err, "creating parser")
|
|
}
|
|
err = p.PQL.Parse()
|
|
if err != nil {
|
|
return nil, errors.Wrap(err, "parsing")
|
|
}
|
|
|
|
// Handle specific panics from the parser and return them as errors.
|
|
var v interface{}
|
|
func() {
|
|
defer func() { v = recover() }()
|
|
p.Execute()
|
|
}()
|
|
if v != nil {
|
|
errorMessage, ok := v.(string)
|
|
if !ok {
|
|
return nil, fmt.Errorf("unexpected parser error of type %T: %[1]v", v)
|
|
}
|
|
if strings.HasPrefix(errorMessage, duplicateArgErrorMessage) || strings.HasPrefix(errorMessage, intOutOfRangeError) || strings.HasPrefix(errorMessage, invalidTimestampError) {
|
|
return nil, fmt.Errorf("%s", v)
|
|
} else {
|
|
panic(v)
|
|
}
|
|
}
|
|
for _, call := range p.Query.Calls {
|
|
if call == nil {
|
|
return nil, fmt.Errorf("unexpected nil Call in query's call list")
|
|
}
|
|
if err := call.CheckCallInfo(); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
|
|
return &p.Query, nil
|
|
}
|
|
|
|
// Unquote interprets s as a single-quoted, double-quoted, or
|
|
// backquoted Go string literal, returning the string value that s
|
|
// quotes. It is a copy of stdlib's strconv.Unquote, but modified so
|
|
// that if s is single-quoted, it can still be a string rather than
|
|
// only character literal. This version of Unquote also accepts
|
|
// unquoted strings and passes them back unchanged.
|
|
func Unquote(s string) (string, error) {
|
|
n := len(s)
|
|
if n < 2 {
|
|
return s, nil
|
|
}
|
|
quote := s[0]
|
|
if quote != '"' && quote != '\'' && quote != '`' {
|
|
return s, nil
|
|
}
|
|
if quote != s[n-1] {
|
|
return "", strconv.ErrSyntax
|
|
}
|
|
s = s[1 : n-1]
|
|
|
|
if quote == '`' {
|
|
if contains(s, '`') {
|
|
return "", strconv.ErrSyntax
|
|
}
|
|
if contains(s, '\r') {
|
|
// -1 because we know there is at least one \r to remove.
|
|
buf := make([]byte, 0, len(s)-1)
|
|
for i := 0; i < len(s); i++ {
|
|
if s[i] != '\r' {
|
|
buf = append(buf, s[i])
|
|
}
|
|
}
|
|
return string(buf), nil
|
|
}
|
|
return s, nil
|
|
}
|
|
if quote != '"' && quote != '\'' {
|
|
return "", strconv.ErrSyntax
|
|
}
|
|
if contains(s, '\n') {
|
|
return "", strconv.ErrSyntax
|
|
}
|
|
|
|
// Is it trivial? Avoid allocation.
|
|
if !contains(s, '\\') && !contains(s, quote) {
|
|
switch quote {
|
|
case '"', '\'':
|
|
if utf8.ValidString(s) {
|
|
return s, nil
|
|
}
|
|
}
|
|
}
|
|
|
|
var runeTmp [utf8.UTFMax]byte
|
|
buf := make([]byte, 0, 3*len(s)/2) // Try to avoid more allocations.
|
|
for len(s) > 0 {
|
|
c, multibyte, ss, err := strconv.UnquoteChar(s, quote)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
s = ss
|
|
if c < utf8.RuneSelf || !multibyte {
|
|
buf = append(buf, byte(c))
|
|
} else {
|
|
n := utf8.EncodeRune(runeTmp[:], c)
|
|
buf = append(buf, runeTmp[:n]...)
|
|
}
|
|
}
|
|
return string(buf), nil
|
|
}
|
|
|
|
// contains reports whether the string contains the byte c.
|
|
func contains(s string, c byte) bool {
|
|
return strings.ContainsRune(s, rune(c))
|
|
}
|