jmespath

    A faithful MoonBit port of jmespath.py, a query language for JSON.

    json
    jmespath
    query
    Download zip
    Author
    Version
    0.1.0
    License
    MIT
    Last updated
    yesterday
    Downloads
    3

    #bobzhang/jmespath

    A faithful MoonBit port of jmespath.py 1.1.0, the Python implementation of JMESPath, a query language for JSON.

    The port mirrors the upstream structure — lexer, Pratt parser with the same binding powers, AST, TreeInterpreter, the function table with its signature/type-checking machinery, and the exception classes with the same messages — and reproduces Python's semantics: truthiness, == (where 1 == 1.0), ordering of numbers and strings, code-point based string functions, json.dumps output for to_string, and so on. Values are core Json.

    #Searching

    ///|
    test "search" {
    let data : Json = {
    "people": [
    { "name": "alice", "age": 30 },
    { "name": "bob", "age": 25 },
    { "name": "carol", "age": 35 },
    ],
    }
    json_inspect(@jmespath.search("people[?age > `26`].name", data), content=[
    "alice", "carol",
    ])
    json_inspect(
    @jmespath.search("max_by(people, &age).name", data),
    content="carol",
    )
    json_inspect(
    @jmespath.search("people[*].{n: name, adult: age >= `18`} | [0]", data),
    content={ "n": "alice", "adult": true },
    )
    }

    compile parses once; the result can be evaluated many times. Compiled expressions are also kept in a small global cache, like upstream.

    ///|
    test "compile" {
    let expr = @jmespath.compile("sort_by(@, &to_number(v))[*].k")
    json_inspect(expr.search([{ "k": "a", "v": "10" }, { "k": "b", "v": "9" }]), content=[
    "b", "a",
    ])
    inspect(expr.expression, content="sort_by(@, &to_number(v))[*].k")
    }

    #Errors

    All errors are constructors of JMESPathError, named after the upstream exception classes; to_string() gives the same text as Python's str(e).

    ///|
    test "errors" {
    try @jmespath.compile("foo]bar") catch {
    e =>
    inspect(
    e,
    content=(
    #|Unexpected token: ]: Parse error at column 3, token "]" (RBRACKET), for expression:
    #|"foo]bar"
    #| ^
    ),
    )
    } noraise {
    _ => fail("expected a parse error")
    }
    try @jmespath.search("length(@)", Json::number(2.0)) catch {
    JMESPathTypeError(function_name~, actual_type~, ..) as e => {
    inspect(function_name, content="length")
    inspect(actual_type, content="number")
    inspect(
    e,
    content="In function length(), invalid type for value: 2, expected one of: ['string', 'array', 'object'], received: \"number\"",
    )
    }
    e => fail("unexpected error: \{e}")
    } noraise {
    _ => fail("expected a type error")
    }
    }

    #Custom functions

    Upstream adds functions by subclassing jmespath.functions.Functions and decorating _func_<name> methods with @signature(...). Here a Functions table starts with the builtins and register_function adds entries; pass it with Options. Arguments are validated against the signature before the implementation runs, and expression references arrive as Value::Expref.

    ///|
    test "custom functions" {
    let functions = @jmespath.Functions::new()
    functions.register_function(
    "count_if",
    [@jmespath.ArgSpec::new(["expref"]), @jmespath.ArgSpec::new(["array"])],
    args => {
    guard args is [Expref(expref), Data(Array(items))] else { Json::null() }
    let mut n = 0
    for item in items {
    if expref.visit(item) is True {
    n 1
    }
    }
    Json::number(n.to_double())
    },
    )
    let options = @jmespath.Options::new(custom_functions=functions)
    json_inspect(
    @jmespath.search("count_if(&(@ > `1`), @)", [1, 2, 3], options~),
    content=2,
    )
    // The signature is enforced like for builtins.
    try @jmespath.search("count_if(@, @)", [1], options~) catch {
    e =>
    inspect(
    e,
    content="In function count_if(), invalid type for value: [1], expected one of: ['expref'], received: \"array\"",
    )
    } noraise {
    _ => fail("expected a type error")
    }
    }

    #Python numbers and json helpers

    loads is a port of Python's json.loads (it accepts NaN/Infinity and reports Python's error messages) and dumps of json.dumps(v, separators=(',', ':')) (ASCII-only output). loads keeps the Python int/float distinction that core Json lacks (see below), so 1.0 stays a float.

    ///|
    test "python numbers" {
    let data = @jmespath.loads("{\"a\": 1, \"b\": 1.0, \"c\": \"\u{e9}\"}")
    inspect(
    @jmespath.dumps(@jmespath.search("[a, b, c]", data)),
    content="[1,1.0,\"\\u00e9\"]",
    )
    inspect(
    @jmespath.dumps(
    @jmespath.search("[to_string(b), avg(`[1, 2]`), sum(`[1, 2]`)]", data),
    ),
    content=(
    #|["1.0",1.5,3]
    ),
    )
    }

    #API overview

    jmespath.pyMoonBit
    jmespath.search(expr, data, opts)@jmespath.search(expr, data, options?)
    jmespath.compile(expr)@jmespath.compile(expr)
    ParsedResult.search(data, opts)ParsedResult::search(data, options?)
    ParsedResult._render_dot_file()ParsedResult::render_dot_file()
    parser.Parser().parse(expr)Parser::new().parse(expr)
    Parser.purge(), _MAX_SIZEParser::purge(), parser_max_size
    lexer.Lexer().tokenize(expr)Lexer::new().tokenize(expr) (Token)
    ast.field(name), ...Node::Field(name), ... (Node::to_json gives the dict form)
    visitor.TreeInterpreter(options)TreeInterpreter::new(options?)
    visitor.Options(custom_functions)Options::new(custom_functions?)
    visitor._ExpressionExpression (Expression::visit(value))
    functions.Functions, @signatureFunctions, ArgSpec, register_function
    exceptions.*ErrorJMESPathError::*Error constructors

    #Tests

    • compliance_*_test.mbt are generated by scripts/gen_compliance.py from upstream's tests/compliance/*.json and tests/legacy/*.json (the raw fixture text is embedded and parsed with loads). All 917 cases pass, including the benchmark cases (bench: parse cases are compiled only). Besides upstream's "raises a ValueError" check, error cases also check the exception class for the error category.
    • differential_test.mbt is generated by scripts/gen_differential.py, which runs a few hundred extra expressions (including randomized sorting inputs with NaN) through upstream jmespath.py and records the exact outcome (json.dumps of the result, or the exception class and message); the port must match byte for byte.
    • lexer_test.mbt, parser_test.mbt, search_test.mbt, functions_test.mbt and custom_functions_test.mbt port the upstream unit tests. Skipped: test_can_max_datetimes (Python datetime values), test_can_handle_long_ints (Python 2 long), test_can_handle_decimals_as_numeric_type (decimal.Decimal; a large integer test replaces it) and test_thread_safety_of_cache (no threads). dict_cls=OrderedDict tests run without the option, since Map is ordered.

    #Differences from jmespath.py

    • int vs float. Core Json has no int/float distinction. This port treats a number as a Python float when its repr field looks like a float literal (1.0, 1e3, Infinity, NaN) or its value is not integral, and as an int otherwise. loads, literals in expressions and the builtins (avg, to_number, sum, abs, ceil, floor) maintain the marker, so to_string, type-error messages, etc. match Python. Data parsed with @json.parse (or built with Json::number) carries no marker, so an integral float such as 1.0 there behaves as the int 1 (only observable in to_string and error messages). Ints beyond 2^53 are stored as doubles; their exact digits are kept for display when they come from loads/to_number, but arithmetic on them is approximate.
    • Expression references (&expr) can only be used directly as function arguments. Upstream evaluates &expr anywhere to an _Expression object (e.g. [&a] or not_null(&a) return one); here such uses raise TypeError, because a JSON value cannot hold an expression. to_string(&a) returns "<jmespath.visitor._Expression object>" without the memory address.
    • Number tokens in expressions (indices, slices) saturate at the Int range instead of being arbitrary precision; out-of-range indices behave the same.
    • Builtin Python exceptions leaking from upstream's interpreter are modelled as ValueError (slice step 0, ceil(NaN), dict.update length errors), TypeError (ordering a number against a string, 'in <string>', dict.update errors) and OverflowError (ceil(Infinity)) constructors with the same messages. merge follows dict.update for its unchecked arguments, except that pairs with non-string keys (which JSON objects cannot hold) raise TypeError.
    • Options has no dict_cls: Map always preserves insertion order, which is what dict_cls=OrderedDict gives upstream.
    • Custom functions are registered with Functions::register_function instead of subclass methods; implementations receive the argument array (no self) and must raise JMESPathError. Unknown type names in a signature match nothing (upstream raises KeyError at call time).
    • Parser cache: a global insertion-ordered map evicting its oldest entry when it reaches parser_max_size (512); Parser(lookahead=2) has no arguments. The lexer returns an array instead of a generator and emits no PendingDeprecationWarning for JEP-12 style literals.
    • Graphviz: render_dot_file renders slice nodes without children (upstream crashes on them).
    • Python details approximated: repr() of strings (printability of non-ASCII characters), and to_number/int()/float() accept ASCII digits only (Python also accepts other Unicode decimal digits).
    • CPython algorithms reproduced: sort/sort_by port CPython 3.13's list.sort() (timsort with powersort merging, using only <), so results match even for inputs without a total order such as NaN; sum/avg follow CPython >= 3.12's sum() (exact int accumulation, then Neumaier-compensated float summation).
    • Object identity: Python compares container elements and tests in with an identity shortcut, which is observable for NaN. This port uses physical equality of Json values for it, and loads returns one shared NaN value just like Python's json module, so @ == @ on [NaN] or contains(@, `NaN`) match upstream. Other identities are not modelled (e.g. NaNs built separately with Json::number in MoonBit are distinct objects, as are NaNs produced by to_number('nan'), like upstream).
    • Strings are UTF-16: strings decoded from JSON keep lone surrogates as single code points, and length, reverse, sort, contains, starts_with, ends_with work on code points like Python. The one residue is join of a lone high surrogate followed by a lone low surrogate, which forms a pair (one code point) here but stays two code points in Python.

    FunctionImpl

    type FunctionImpl = (Array[Value]) -> Json raise JMESPathError

    The implementation of a JMESPath function. Arguments have already been validated against the signature.

    JMESPathError

    pub(all) suberror JMESPathError {
    ParseError(lex_position~ : Int, token_value~ : String, token_type~ : String, msg~ : String, expression~ : String?)
    IncompleteExpressionError(lex_position~ : Int, token_value~ : String?, token_type~ : String?, expression~ : String?)
    LexerError(lexer_position~ : Int, lexer_value~ : String, message~ : String, expression~ : String?)
    ArityError(expected_arity~ : Int, actual_arity~ : Int, function_name~ : String)
    VariadictArityError(expected_arity~ : Int, actual_arity~ : Int, function_name~ : String)
    JMESPathTypeError(function_name~ : String, current_value~ : Value, actual_type~ : String, expected_types~ : Array[String])
    EmptyExpressionError
    UnknownFunctionError(String)
    ValueError(String)
    TypeError(String)
    OverflowError(String)
    }

    All errors raised while compiling or evaluating a JMESPath expression.

    Constructors mirror the exception classes of jmespath.py:

    • ParseError (exceptions.ParseError): token_value is the str() of the offending token's value, token_type is upper-cased like upstream.
    • IncompleteExpressionError: after the parser fills in the expression, lex_position is the length of the expression and token_value / token_type are None, just like set_expression upstream.
    • LexerError
    • ArityError / VariadictArityError (sic, upstream spelling)
    • JMESPathTypeError
    • EmptyExpressionError
    • UnknownFunctionError

    Python can additionally leak builtin exceptions from the interpreter (for example ValueError: slice step cannot be zero, or a TypeError when < compares a number with a string). These are modelled by the ValueError, TypeError and OverflowError constructors.

    JMESPathError::is_arity_error

    fn JMESPathError::is_arity_error(self : JMESPathError) -> Bool

    isinstance(e, ArityError) (includes VariadictArityError).

    JMESPathError::is_parse_error

    fn JMESPathError::is_parse_error(self : JMESPathError) -> Bool

    isinstance(e, ParseError) (includes lexer, incomplete-expression and arity errors, which subclass ParseError upstream).

    JMESPathError::is_value_error

    fn JMESPathError::is_value_error(self : JMESPathError) -> Bool

    Whether upstream raises this as a subclass of Python's ValueError. Every error except the builtin TypeError / OverflowError ones is.

    JMESPathError::to_string

    fn JMESPathError::to_string(self : JMESPathError) -> String

    JSONDecodeError

    pub(all) suberror JSONDecodeError {
    JSONDecodeError(msg~ : String, doc~ : String, pos~ : Int)
    }

    Raised by loads (Python's json.JSONDecodeError). pos, lineno and colno count code points, like Python.

    JSONDecodeError::to_string

    fn JSONDecodeError::to_string(self : JSONDecodeError) -> String

    ArgSpec

    pub(all) struct ArgSpec {
    types : Array[String]
    variadic : Bool
    } derive(Eq,
    Debug
    )

    One entry of a function signature ({'types': [...], 'variadic': ...}).

    types are JMESPath type names: number, string, boolean, array, object, null, expref, or typed arrays such as array-number. An empty list accepts any value.

    ArgSpec::new

    fn ArgSpec::new(types : Array[String], variadic? : Bool) -> ArgSpec

    Expression

    pub struct Expression {
    expression : Node
    // private fields
    }

    An expression reference bound to the interpreter that created it (jmespath.visitor._Expression).

    Expression::visit

    fn Expression::visit(self : Expression, value : Json) -> Json raise JMESPathError

    Evaluates the referenced expression against value (upstream: expref.visit(expref.expression, value)).

    FunctionSpec

    pub(all) struct FunctionSpec {
    function : (Array[Value]) -> Json raise JMESPathError
    signature : Array[ArgSpec]
    }

    A function table entry ({'function': ..., 'signature': ...}).

    Functions

    pub struct Functions {
    // private fields
    }

    A table of JMESPath functions (jmespath.functions.Functions).

    Functions::call_function

    fn Functions::call_function(self : Functions, function_name : String, resolved_args : Array[Value]) -> Json raise JMESPathError

    Validates resolved_args against the function's signature and calls it.

    Functions::get

    fn Functions::get(self : Functions, name : String) -> FunctionSpec?

    Looks up a function by name.

    Functions::names

    fn Functions::names(self : Functions) -> Array[String]

    The names of all functions in the table.

    Functions::new

    fn Functions::new() -> Functions

    A function table with all the builtin functions.

    Functions::register_function

    fn Functions::register_function(self : Functions, name : String, signature : Array[ArgSpec], function : (Array[Value]) -> Json raise JMESPathError) -> Unit

    Adds a function to the table, replacing any function with the same name (upstream: define _func_<name> with @signature(...) in a subclass).

    test {
    let functions = @jmespath.Functions::new()
    functions.register_function("double", [@jmespath.ArgSpec::new(["number"])], args => {
    guard args[0] is Data(Number(n, ..)) else { Json::null() }
    Json::number(n * 2.0)
    })
    let options = @jmespath.Options::new(custom_functions=functions)
    let result = @jmespath.search("double(`21`)", Json::null(), options~)
    json_inspect(result, content=42)
    }

    Lexer

    pub struct Lexer {
    // private fields
    }

    The JMESPath tokenizer (jmespath.lexer.Lexer).

    Lexer::new

    fn Lexer::new() -> Lexer

    Lexer::tokenize

    fn Lexer::tokenize(self : Lexer, expression : String) -> Array[Token] raise JMESPathError

    Tokenizes expression. Upstream yields tokens lazily, but the parser always consumes the whole generator first, so an eager array is equivalent. The last token is always eof.

    Node

    pub(all) enum Node {
    Comparator(String, Node, Node)
    Current
    Expref(Node)
    FunctionExpression(String, Array[Node])
    Field(String)
    FilterProjection(Node, Node, Node)
    Flatten(Node)
    Identity
    Index(Int)
    IndexExpression(Array[Node])
    KeyValPair(String, Node)
    Literal(Json)
    MultiSelectDict(Array[Node])
    MultiSelectList(Array[Node])
    OrExpression(Node, Node)
    AndExpression(Node, Node)
    NotExpression(Node)
    Pipe(Node, Node)
    Projection(Node, Node)
    Subexpression(Array[Node])
    Slice(Int?, Int?, Int?)
    ValueProjection(Node, Node)
    } derive(Eq,
    Debug
    )

    A JMESPath AST node.
    impl ToJson for Node

    Node::children

    fn Node::children(self : Node) -> Array[Node]

    The upstream node['children'] that are nodes. (Upstream slice nodes store their integer bounds as children; they are not returned here.)

    Node::to_json

    fn Node::to_json(self : Node) -> Json

    Node::type_name

    fn Node::type_name(self : Node) -> String

    The upstream node['type'].

    Node::value

    fn Node::value(self : Node) -> Json?

    The upstream node['value'], for the node types that have one.

    Options

    pub struct Options {
    custom_functions : Functions?
    }

    Options to control how a JMESPath expression is evaluated (jmespath.visitor.Options).

    Upstream also has dict_cls, the mapping class used for multi-select hashes; it is not needed here because Map always preserves insertion order (the behaviour upstream gets with dict_cls=OrderedDict).

    Options::new

    fn Options::new(custom_functions? : Functions) -> Options

    ParsedResult

    pub struct ParsedResult {
    expression : String
    parsed : Node
    }

    The result of compiling an expression (jmespath.parser.ParsedResult).

    ParsedResult::render_dot_file

    fn ParsedResult::render_dot_file(self : ParsedResult) -> String

    Renders the parsed AST as a Graphviz dot file (_render_dot_file). The AST is an implementation detail; this is meant for debugging.

    ParsedResult::search

    fn ParsedResult::search(self : ParsedResult, value : Json, options? : Options) -> Json raise JMESPathError

    Evaluates the compiled expression against value.

    Parser

    pub struct Parser {
    // private fields
    }

    The JMESPath parser (jmespath.parser.Parser).

    Parser::cache_size

    fn Parser::cache_size() -> Int

    The number of expressions currently cached.

    Parser::new

    fn Parser::new() -> Parser

    Parser::parse

    fn Parser::parse(self : Parser, expression : String) -> ParsedResult raise JMESPathError

    Parses (and caches) an expression.

    Parser::purge

    fn Parser::purge() -> Unit

    Clear the expression compilation cache (Parser.purge()).

    Token

    pub(all) struct Token {
    type_ : String
    value : TokenValue
    start : Int
    end : Int
    } derive(Eq,
    Debug
    )

    A token, {'type': ..., 'value': ..., 'start': ..., 'end': ...} upstream. end reproduces upstream's values exactly, including its quirks (for literals it is the token length, for two-character operators start + 1).

    TokenValue

    pub(all) enum TokenValue {
    Str(String)
    Int(Int)
    Literal(Json)
    } derive(Eq,
    Debug
    )

    The value of a token: identifiers and operators carry their text, number tokens an integer and literal tokens a JSON value.

    TreeInterpreter

    pub struct TreeInterpreter {
    // private fields
    }

    Evaluates AST nodes against JSON values (TreeInterpreter).

    TreeInterpreter::new

    fn TreeInterpreter::new(options? : Options) -> TreeInterpreter

    TreeInterpreter::options

    fn TreeInterpreter::options(self : TreeInterpreter) -> Options

    The options this interpreter was created with.

    TreeInterpreter::visit

    fn TreeInterpreter::visit(self : TreeInterpreter, node : Node, value : Json) -> Json raise JMESPathError

    Evaluates node against value.

    Value

    pub(all) enum Value {
    Data(Json)
    Expref(Expression)
    }

    A function argument: either a JSON value or an expression reference (&expr, upstream's _Expression).
    impl Show for Value

    compile

    fn compile(expression : String) -> ParsedResult raise JMESPathError

    Compiles a JMESPath expression (jmespath.compile).

    test {
    let parsed = @jmespath.compile("foo.bar")
    json_inspect(parsed.search({ "foo": { "bar": "baz" } }), content="baz")
    }

    dumps

    fn dumps(value : Json) -> String

    Python's json.dumps(value, separators=(',', ':')) (with the default ensure_ascii=True): non-ASCII characters are written as \uXXXX escapes and Python floats keep their .0.

    loads

    fn loads(s : String) -> Json raise JSONDecodeError

    Python's json.loads(s).

    Differences from @json.parse: accepts NaN, Infinity and -Infinity, reports Python's error messages, and marks floats (1.0, 1e3) so that they keep behaving as Python floats (see dumps).

    test {
    let v = @jmespath.loads("[1, 1.0, 1e3, \"\\u2713\"]")
    inspect(@jmespath.dumps(v), content="[1,1.0,1000.0,\"\\u2713\"]")
    }

    parser_max_size

    let parser_max_size : Int

    The number of most recently compiled expressions kept in the cache (Parser._MAX_SIZE).
    fn search(expression : String, data : Json, options? : Options) -> Json raise JMESPathError

    Evaluates a JMESPath expression against data (jmespath.search).

    test {
    let data : Json = { "foo": [{ "a": 1 }, { "a": 2 }, { "b": 3 }] }
    json_inspect(@jmespath.search("foo[*].a", data), content=[1, 2])
    }

    version

    let version : String

    The version of jmespath.py this package is a port of.