gjson/gjson.go

651 lines
13 KiB
Go
Raw Normal View History

2016-08-11 06:07:45 +03:00
// Package gjson provides searching for json strings.
package gjson
import "strconv"
// Type is Result type
2016-08-18 17:18:24 +03:00
type Type int
2016-08-11 06:07:45 +03:00
const (
// Null is a null json value
Null Type = iota
// False is a json false boolean
False
// Number is json number
Number
// String is a json string
String
// True is a json true boolean
True
// JSON is a raw block of JSON
JSON
)
// Result represents a json value that is returned from Get().
type Result struct {
// Type is the json type
Type Type
// Raw is the raw json
Raw string
// Str is the json string
Str string
// Num is the json number
Num float64
}
// String returns a string representation of the value.
func (t Result) String() string {
switch t.Type {
default:
return "null"
case False:
return "false"
case Number:
return strconv.FormatFloat(t.Num, 'f', -1, 64)
case String:
return t.Str
case JSON:
return t.Raw
case True:
return "true"
}
}
2016-08-12 18:39:08 +03:00
// Exists returns true if value exists.
//
// if gjson.Get(json, "name.last").Exists(){
// println("value exists")
// }
func (t Result) Exists() bool {
2016-08-12 18:51:56 +03:00
return t.Type != Null || len(t.Raw) != 0
2016-08-12 18:39:08 +03:00
}
2016-08-11 06:07:45 +03:00
// Value returns one of these types:
//
// bool, for JSON booleans
// float64, for JSON numbers
// Number, for JSON numbers
// string, for JSON string literals
// nil, for JSON null
//
func (t Result) Value() interface{} {
switch t.Type {
default:
return nil
case False:
return false
case Number:
return t.Num
case String:
return t.Str
case JSON:
return t.Raw
case True:
return true
}
}
2016-08-11 20:39:38 +03:00
type part struct {
wild bool
key string
}
type frame struct {
key string
count int
stype byte
}
2016-08-11 06:07:45 +03:00
// Get searches json for the specified path.
// A path is in dot syntax, such as "name.last" or "age".
// This function expects that the json is well-formed, and does not validate.
// Invalid json will not panic, but it may return back unexpected results.
// When the value is found it's returned immediately.
//
// A path is a series of keys seperated by a dot.
// A key may contain special wildcard characters '*' and '?'.
// To access an array value use the index as the key.
// To get the number of elements in an array use the '#' character.
2016-08-12 18:39:08 +03:00
// The dot and wildcard character can be escaped with '\'.
//
2016-08-11 06:07:45 +03:00
// {
// "name": {"first": "Tom", "last": "Anderson"},
// "age":37,
// "children": ["Sara","Alex","Jack"]
// }
// "name.last" >> "Anderson"
// "age" >> 37
// "children.#" >> 3
// "children.1" >> "Alex"
// "child*.2" >> "Jack"
// "c?ildren.0" >> "Sara"
//
func Get(json string, path string) Result {
2016-08-11 20:39:38 +03:00
var s int
2016-08-11 06:07:45 +03:00
var wild bool
2016-08-11 20:39:38 +03:00
var parts = make([]part, 0, 4)
2016-08-11 06:07:45 +03:00
// do nothing when no path specified
if len(path) == 0 {
return Result{} // nothing
}
2016-08-11 20:39:38 +03:00
// parse the path. just split on the dot
for i := 0; i < len(path); i++ {
2016-08-12 04:51:29 +03:00
next_part:
2016-08-19 21:22:59 +03:00
// be optimistic that the path mostly contains lowercase and
// underscore characters.
if path[i] <= '\\' {
if path[i] == '\\' {
// go into escape mode.
epart := []byte(path[s:i])
2016-08-12 04:51:29 +03:00
i++
2016-08-19 21:22:59 +03:00
if i < len(path) {
epart = append(epart, path[i])
i++
for ; i < len(path); i++ {
if path[i] == '\\' {
i++
if i < len(path) {
epart = append(epart, path[i])
}
continue
} else if path[i] == '.' {
parts = append(parts, part{wild: wild, key: string(epart)})
if wild {
wild = false
}
s = i + 1
i++
goto next_part
} else if path[i] == '*' || path[i] == '?' {
wild = true
2016-08-12 04:51:29 +03:00
}
2016-08-19 21:22:59 +03:00
epart = append(epart, path[i])
2016-08-12 04:51:29 +03:00
}
}
2016-08-19 21:22:59 +03:00
parts = append(parts, part{wild: wild, key: string(epart)})
goto end_parts
} else if path[i] == '.' {
parts = append(parts, part{wild: wild, key: path[s:i]})
if wild {
wild = false
}
s = i + 1
} else if path[i] == '*' || path[i] == '?' {
wild = true
2016-08-12 04:51:29 +03:00
}
2016-08-11 20:39:38 +03:00
}
}
parts = append(parts, part{wild: wild, key: path[s:]})
2016-08-12 04:51:29 +03:00
end_parts:
2016-08-11 20:39:38 +03:00
var i, depth int
var f frame
var matched bool
2016-08-18 17:18:24 +03:00
var stack = make([]frame, 1, 4)
var value Result
var vc byte
2016-08-11 20:39:38 +03:00
2016-08-11 06:07:45 +03:00
depth = 1
// look for first delimiter
for ; i < len(json); i++ {
2016-08-19 21:22:59 +03:00
if json[i] == '{' {
f.stype = '{'
2016-08-11 06:07:45 +03:00
i++
2016-08-19 21:22:59 +03:00
stack[0].stype = f.stype
2016-08-11 06:07:45 +03:00
break
2016-08-19 21:22:59 +03:00
} else if json[i] == '[' {
f.stype = '['
stack[0].stype = f.stype
i++
break
} else if json[i] <= ' ' {
continue
} else {
return Result{}
2016-08-11 06:07:45 +03:00
}
}
2016-08-19 21:22:59 +03:00
// read the next key
2016-08-11 06:07:45 +03:00
read_key:
2016-08-11 20:39:38 +03:00
if f.stype == '[' {
2016-08-19 21:22:59 +03:00
// for arrays we use the index of the value as the key.
// so "0" is the key for the first value, and "10" is the
// key for the 10th value.
2016-08-11 20:39:38 +03:00
f.key = strconv.FormatInt(int64(f.count), 10)
f.count++
2016-08-11 06:07:45 +03:00
} else {
2016-08-19 21:22:59 +03:00
// for objects we must parse the next string.
2016-08-11 06:07:45 +03:00
for ; i < len(json); i++ {
2016-08-19 21:22:59 +03:00
// read string
2016-08-11 06:07:45 +03:00
if json[i] == '"' {
i++
// the first double-quote has already been read
s = i
for ; i < len(json); i++ {
if json[i] == '"' {
2016-08-11 20:39:38 +03:00
f.key = json[s:i]
2016-08-11 06:07:45 +03:00
i++
break
}
if json[i] == '\\' {
i++
for ; i < len(json); i++ {
if json[i] == '"' {
// look for an escaped slash
if json[i-1] == '\\' {
n := 0
for j := i - 2; j > s-1; j-- {
if json[j] != '\\' {
break
}
n++
}
if n%2 == 0 {
continue
}
}
break
}
}
2016-08-11 20:39:38 +03:00
f.key = unescape(json[s:i])
2016-08-11 06:07:45 +03:00
i++
break
}
}
break
}
2016-08-19 21:22:59 +03:00
// end read string
2016-08-11 06:07:45 +03:00
}
}
2016-08-19 21:22:59 +03:00
// we have a brand new (possibly shiny) key.
2016-08-11 06:07:45 +03:00
// is it the key that we are looking for?
2016-08-11 20:39:38 +03:00
if parts[depth-1].wild {
2016-08-11 06:07:45 +03:00
// it's a wildcard path element
2016-08-11 20:39:38 +03:00
matched = wildcardMatch(f.key, parts[depth-1].key)
2016-08-11 06:07:45 +03:00
} else {
2016-08-19 21:22:59 +03:00
// just a straight up equality check
2016-08-11 20:39:38 +03:00
matched = parts[depth-1].key == f.key
2016-08-11 06:07:45 +03:00
}
// read to the value token
// there's likely a colon here, but who cares. just burn past it.
for ; i < len(json); i++ {
2016-08-12 04:51:29 +03:00
if json[i] < '"' { // control character
continue
}
if json[i] < '-' { // string
2016-08-11 06:07:45 +03:00
i++
// we read the val below
vc = '"'
goto proc_val
2016-08-12 04:51:29 +03:00
}
if json[i] < '[' { // number
if json[i] == ':' {
continue
}
2016-08-11 06:07:45 +03:00
vc = '0'
s = i
i++
// look for characters that cannot be in a number
for ; i < len(json); i++ {
switch json[i] {
default:
continue
case ' ', '\t', '\r', '\n', ',', ']', '}':
}
break
}
2016-08-18 17:18:24 +03:00
value.Raw = json[s:i]
2016-08-11 06:07:45 +03:00
goto proc_val
}
2016-08-12 04:51:29 +03:00
if json[i] < ']' { // '['
i++
vc = '['
goto proc_delim
}
if json[i] < 'u' { // true, false, null
vc = json[i]
s = i
i++
for ; i < len(json); i++ {
// let's pick up any character. it doesn't matter.
if json[i] < 'a' || json[i] > 'z' {
break
}
}
2016-08-18 17:18:24 +03:00
value.Raw = json[s:i]
2016-08-12 04:51:29 +03:00
goto proc_val
}
// must be an open objet
i++
vc = '{'
goto proc_delim
2016-08-11 06:07:45 +03:00
}
2016-08-18 17:18:24 +03:00
vc = 0
2016-08-11 06:07:45 +03:00
// sanity check before we move on
if i >= len(json) {
return Result{}
}
proc_delim:
if (matched && depth == len(parts)) || !matched {
2016-08-19 21:22:59 +03:00
// begin squash
2016-08-11 06:07:45 +03:00
// squash the value, ignoring all nested arrays and objects.
s = i - 1
// the first '[' or '{' has already been read
depth := 1
2016-08-19 21:22:59 +03:00
squash:
2016-08-11 06:07:45 +03:00
for ; i < len(json); i++ {
2016-08-12 04:51:29 +03:00
if json[i] >= '"' && json[i] <= '}' {
2016-08-19 21:22:59 +03:00
switch json[i] {
case '"':
2016-08-11 06:07:45 +03:00
i++
2016-08-12 04:51:29 +03:00
s2 := i
for ; i < len(json); i++ {
if json[i] == '"' {
// look for an escaped slash
if json[i-1] == '\\' {
n := 0
for j := i - 2; j > s2-1; j-- {
if json[j] != '\\' {
break
}
n++
}
if n%2 == 0 {
continue
2016-08-11 06:07:45 +03:00
}
}
2016-08-12 04:51:29 +03:00
break
2016-08-11 06:07:45 +03:00
}
2016-08-12 04:51:29 +03:00
}
2016-08-19 21:22:59 +03:00
case '{', '[':
depth++
case '}', ']':
depth--
if depth == 0 {
i++
break squash
2016-08-11 06:07:45 +03:00
}
}
}
}
2016-08-19 21:22:59 +03:00
// end squash
// the 'i' and 's' values should fall-though to the proc_val function
2016-08-11 06:07:45 +03:00
}
// process the value
proc_val:
if matched {
// hit, that's good!
if depth == len(parts) {
switch vc {
case '{', '[':
value.Type = JSON
2016-08-19 21:22:59 +03:00
value.Raw = json[s:i]
2016-08-11 06:07:45 +03:00
case 'n':
value.Type = Null
case 't':
value.Type = True
case 'f':
value.Type = False
case '"':
value.Type = String
// readstr
// the val has not been read yet
// the first double-quote has already been read
s = i
for ; i < len(json); i++ {
if json[i] == '"' {
2016-08-18 17:18:24 +03:00
value.Raw = json[s:i]
value.Str = value.Raw
2016-08-11 06:07:45 +03:00
i++
break
}
if json[i] == '\\' {
i++
for ; i < len(json); i++ {
if json[i] == '"' {
// look for an escaped slash
if json[i-1] == '\\' {
n := 0
for j := i - 2; j > s-1; j-- {
if json[j] != '\\' {
break
}
n++
}
if n%2 == 0 {
continue
}
}
break
}
}
2016-08-18 17:18:24 +03:00
value.Raw = json[s:i]
value.Str = unescape(value.Raw)
2016-08-11 06:07:45 +03:00
i++
break
}
}
// end readstr
case '0':
value.Type = Number
2016-08-18 17:18:24 +03:00
value.Num, _ = strconv.ParseFloat(value.Raw, 64)
2016-08-11 06:07:45 +03:00
}
return value
} else {
2016-08-11 20:39:38 +03:00
f.stype = vc
2016-08-15 14:56:55 +03:00
f.count = 0
2016-08-11 20:39:38 +03:00
stack = append(stack, f)
2016-08-11 06:07:45 +03:00
depth++
goto read_key
}
}
if vc == '"' {
// readstr
// the val has not been read yet. we can read and throw away.
// the first double-quote has already been read
s = i
for ; i < len(json); i++ {
if json[i] == '"' {
// look for an escaped slash
if json[i-1] == '\\' {
n := 0
for j := i - 2; j > s-1; j-- {
if json[j] != '\\' {
break
}
n++
}
if n%2 == 0 {
continue
}
}
break
}
}
i++
// end readstr
}
// read to the comma or end of object
for ; i < len(json); i++ {
switch json[i] {
case '}', ']':
2016-08-11 20:39:38 +03:00
if parts[depth-1].key == "#" {
return Result{Type: Number, Num: float64(f.count)}
2016-08-11 06:07:45 +03:00
}
// step the stack back
depth--
if depth == 0 {
return Result{}
}
2016-08-11 20:39:38 +03:00
stack = stack[:len(stack)-1]
f = stack[len(stack)-1]
2016-08-11 06:07:45 +03:00
case ',':
i++
goto read_key
}
}
return Result{}
}
// unescape unescapes a string
func unescape(json string) string { //, error) {
var str = make([]byte, 0, len(json))
for i := 0; i < len(json); i++ {
switch {
default:
str = append(str, json[i])
case json[i] < ' ':
return "" //, errors.New("invalid character in string")
case json[i] == '\\':
i++
if i >= len(json) {
return "" //, errors.New("invalid escape sequence")
}
switch json[i] {
default:
return "" //, errors.New("invalid escape sequence")
case '\\':
str = append(str, '\\')
case '/':
str = append(str, '/')
case 'b':
str = append(str, '\b')
case 'f':
str = append(str, '\f')
case 'n':
str = append(str, '\n')
case 'r':
str = append(str, '\r')
case 't':
str = append(str, '\t')
case '"':
str = append(str, '"')
case 'u':
if i+5 > len(json) {
return "" //, errors.New("invalid escape sequence")
}
i++
// extract the codepoint
var code int
for j := i; j < i+4; j++ {
switch {
default:
return "" //, errors.New("invalid escape sequence")
case json[j] >= '0' && json[j] <= '9':
code += (int(json[j]) - '0') << uint(12-(j-i)*4)
case json[j] >= 'a' && json[j] <= 'f':
code += (int(json[j]) - 'a' + 10) << uint(12-(j-i)*4)
case json[j] >= 'a' && json[j] <= 'f':
code += (int(json[j]) - 'a' + 10) << uint(12-(j-i)*4)
}
}
str = append(str, []byte(string(code))...)
i += 3 // only 3 because we will increment on the for-loop
}
}
}
return string(str) //, nil
}
// Less return true if a token is less than another token.
// The caseSensitive paramater is used when the tokens are Strings.
// The order when comparing two different type is:
//
// Null < False < Number < String < True < JSON
//
func (t Result) Less(token Result, caseSensitive bool) bool {
if t.Type < token.Type {
return true
}
if t.Type > token.Type {
return false
}
switch t.Type {
default:
return t.Raw < token.Raw
case String:
if caseSensitive {
return t.Str < token.Str
}
return stringLessInsensitive(t.Str, token.Str)
case Number:
return t.Num < token.Num
}
}
func stringLessInsensitive(a, b string) bool {
for i := 0; i < len(a) && i < len(b); i++ {
if a[i] >= 'A' && a[i] <= 'Z' {
if b[i] >= 'A' && b[i] <= 'Z' {
// both are uppercase, do nothing
if a[i] < b[i] {
return true
} else if a[i] > b[i] {
return false
}
} else {
// a is uppercase, convert a to lowercase
if a[i]+32 < b[i] {
return true
} else if a[i]+32 > b[i] {
return false
}
}
} else if b[i] >= 'A' && b[i] <= 'Z' {
// b is uppercase, convert b to lowercase
if a[i] < b[i]+32 {
return true
} else if a[i] > b[i]+32 {
return false
}
} else {
// neither are uppercase
if a[i] < b[i] {
return true
} else if a[i] > b[i] {
return false
}
}
}
return len(a) < len(b)
}
// wilcardMatch returns true if str matches pattern. This is a very
// simple wildcard match where '*' matches on any number characters
// and '?' matches on any one character.
func wildcardMatch(str, pattern string) bool {
if pattern == "*" {
return true
}
return deepMatch(str, pattern)
}
func deepMatch(str, pattern string) bool {
for len(pattern) > 0 {
switch pattern[0] {
default:
if len(str) == 0 || str[0] != pattern[0] {
return false
}
case '?':
if len(str) == 0 {
return false
}
case '*':
return wildcardMatch(str, pattern[1:]) ||
(len(str) > 0 && wildcardMatch(str[1:], pattern))
}
str = str[1:]
pattern = pattern[1:]
}
return len(str) == 0 && len(pattern) == 0
}