diff --git a/README.md b/README.md index 0f429ac..b525575 100644 --- a/README.md +++ b/README.md @@ -52,55 +52,63 @@ defmodule MyParser do |> integer(2) |> ignore(string(":")) |> integer(2) - |> optional(string("Z")) - defparsec :datetime, date |> ignore(string("T")) |> concat(time), debug: true + # RFC 3339 allows "T", "t" or a space between the date and the time + separator = ascii_char([?T, ?t, ?\s]) + zone = ascii_char([?A..?Z]) + + defparsec :datetime, date |> ignore(separator) |> concat(time) |> optional(zone), debug: true end MyParser.datetime("2010-04-17T14:12:34Z") -#=> {:ok, [2010, 4, 17, 14, 12, 34, "Z"], "", %{}, {1, 0}, 20} +#=> {:ok, [2010, 4, 17, 14, 12, 34, ?Z], "", %{}, {1, 0}, 20} ``` If you add `debug: true` to `defparsec/3`, it will print the generated -clauses, which are shown below: +clauses, preceded by the character guards they share, as shown below: ```elixir -defp datetime__0(<>, - acc, stack, comb__context, comb__line, comb__column) - when x0 >= 48 and x0 <= 57 and (x1 >= 48 and x1 <= 57) and - (x2 >= 48 and x2 <= 57) and (x3 >= 48 and x3 <= 57) and - (x4 >= 48 and x4 <= 57) and (x5 >= 48 and x5 <= 57) and - (x6 >= 48 and x6 <= 57) and (x7 >= 48 and x7 <= 57) and - (x8 >= 48 and x8 <= 57) and (x9 >= 48 and x9 <= 57) and - (x10 >= 48 and x10 <= 57) and (x11 >= 48 and x11 <= 57) and - (x12 >= 48 and x12 <= 57) and (x13 >= 48 and x13 <= 57) do +defguardp __ascii_digit(char) when char >= ?0 and char <= ?9 + +defguardp __ascii_char_T__t__0x20(char) when char === ?T or char === ?t or char === ?\s + +defguardp __ascii_upper(char) when char >= ?A and char <= ?Z + +defp datetime__0(<>, + acc, stack, context, comb__line, comb__offset) + when __ascii_digit(x0) and __ascii_digit(x1) and __ascii_digit(x2) and + __ascii_digit(x3) and __ascii_digit(x4) and __ascii_digit(x5) and + __ascii_digit(x6) and __ascii_digit(x7) and __ascii_char_T__t__0x20(x8) and + __ascii_digit(x9) and __ascii_digit(x10) and __ascii_digit(x11) and + __ascii_digit(x12) and __ascii_digit(x13) and __ascii_digit(x14) do datetime__1( rest, - [(x13 - 48) * 1 + (x12 - 48) * 10, (x11 - 48) * 1 + (x10 - 48) * 10, - (x9 - 48) * 1 + (x8 - 48) * 10, (x7 - 48) * 1 + (x6 - 48) * 10, (x5 - 48) * 1 + (x4 - 48) * 10, - (x3 - 48) * 1 + (x2 - 48) * 10 + (x1 - 48) * 100 + (x0 - 48) * 1000] ++ acc, + [x14 - 48 + (x13 - 48) * 10, x12 - 48 + (x11 - 48) * 10, + x10 - 48 + (x9 - 48) * 10, x7 - 48 + (x6 - 48) * 10, x5 - 48 + (x4 - 48) * 10, + x3 - 48 + (x2 - 48) * 10 + (x1 - 48) * 100 + (x0 - 48) * 1000] ++ acc, stack, - comb__context, + context, comb__line, - comb__column + 19 + comb__offset + 19 ) end -defp datetime__0(rest, acc, _stack, context, line, column) do - {:error, "...", rest, context, line, column} +defp datetime__0(rest, _acc, _stack, context, line, offset) do + {:error, "...", rest, context, line, offset} end -defp datetime__1(<<"Z", rest::binary>>, acc, stack, comb__context, comb__line, comb__column) do - datetime__2(rest, ["Z"] ++ acc, stack, comb__context, comb__line, comb__column + 1) +defp datetime__1(<>, acc, stack, context, comb__line, comb__offset) + when __ascii_upper(x0) do + datetime__2(rest, [x0] ++ acc, stack, context, comb__line, comb__offset + 1) end -defp datetime__1(rest, acc, stack, context, line, column) do - datetime__2(rest, acc, stack, context, line, column) +defp datetime__1(<>, acc, stack, context, comb__line, comb__offset) do + datetime__2(rest, [] ++ acc, stack, context, comb__line, comb__offset) end -defp datetime__2(rest, acc, _stack, context, line, column) do - {:ok, acc, rest, context, line, column} +defp datetime__2(rest, acc, _stack, context, line, offset) do + {:ok, acc, rest, context, line, offset} end ``` diff --git a/lib/nimble_parsec.ex b/lib/nimble_parsec.ex index 59fd6b3..c84d27d 100644 --- a/lib/nimble_parsec.ex +++ b/lib/nimble_parsec.ex @@ -151,17 +151,17 @@ defmodule NimbleParsec do name: name, combinator: combinator ] do - {defs, inline} = NimbleParsec.Compiler.compile(name, combinator, opts) - - NimbleParsec.Recorder.record( - __MODULE__, - parser_kind, - combinator_kind, - name, - defs, - inline, - opts - ) + # One call, since every expression here is inlined once per parser. + {defs, inline} = + NimbleParsec.Compiler.compile_into( + __MODULE__, + __ENV__.file, + parser_kind, + combinator_kind, + name, + combinator, + opts + ) if opts[:export_metadata] do def __nimble_parsec__(unquote(name)), diff --git a/lib/nimble_parsec/compiler.ex b/lib/nimble_parsec/compiler.ex index 1b2869d..06afa75 100644 --- a/lib/nimble_parsec/compiler.ex +++ b/lib/nimble_parsec/compiler.ex @@ -68,22 +68,22 @@ defmodule NimbleParsec.Compiler do @doc """ Compiles the given combinators into multiple definitions. """ - def compile(name, [], _opts) do + def compile(name, [], _char_guards, _opts) do raise ArgumentError, "cannot compile #{inspect(name)} with an empty parser combinator" end - def compile(name, combinators, opts) when is_list(combinators) do + def compile(name, combinators, char_guards, opts) when is_list(combinators) do inline? = Keyword.get(opts, :inline, false) - {defs, inline} = compile(name, combinators) + {defs, inline, new_char_guards, char_guards} = compile(name, combinators, char_guards) if inline? do - {defs, inline} + {defs, inline, new_char_guards, char_guards} else - {defs, []} + {defs, [], new_char_guards, char_guards} end end - defp compile(name, combinators) do + defp compile(name, combinators, char_guards) do config = %{ acc_depth: 0, catch_all: nil, @@ -99,7 +99,10 @@ defmodule NimbleParsec.Compiler do |> Enum.reverse() |> compile([], [], next, step, config) - {Enum.reverse([build_ok(last) | defs]), [{last, @arity} | inline]} + {defs, new_char_guards, char_guards} = + extract_char_guards(Enum.reverse([build_ok(last) | defs]), char_guards) + + {defs, [{last, @arity} | inline], new_char_guards, char_guards} end defp compile([], defs, inline, current, step, _config) do @@ -254,7 +257,7 @@ defmodule NimbleParsec.Compiler do quote(do: [unquote(rest), orig, unquote_splicing(count), acc, stack, context, line, offset]) end - guards = compile_bin_ranges(var, inclusive, exclusive) + guards = maybe_char_guard(var, inclusive, exclusive, modifier) guards = if max, do: guards ++ [quote(do: count < unquote(max - min))], else: guards recur_def = @@ -935,7 +938,7 @@ defmodule NimbleParsec.Compiler do {var, counter} = build_var(counter) input = apply_bin_modifier(var, modifier) - guards = compile_bin_ranges(var, inclusive, exclusive) + guards = maybe_char_guard(var, inclusive, exclusive, modifier) offset = if modifier == :integer do @@ -1145,6 +1148,169 @@ defmodule NimbleParsec.Compiler do "#{inspect(count)} bytes" end + ## Character guards + + # Below this many comparisons, expanding is shorter than calling a guard. + @char_guard_threshold 3 + + @char_guard_classes %{ + [?0..?9] => :digit, + [?a..?z] => :lower, + [?A..?Z] => :upper, + [?a..?z, ?A..?Z] => :alpha, + [?a..?z, ?A..?Z, ?0..?9] => :alnum, + [?0..?9, ?a..?f, ?A..?F] => :hex, + [?\s, ?\t, ?\n, ?\r] => :space + } + + defp maybe_char_guard(var, inclusive, exclusive, modifier) do + inclusive = Enum.map(inclusive, &ascending/1) + exclusive = Enum.map(exclusive, &ascending/1) + + if char_guard_class(inclusive, exclusive) || + comparisons(inclusive, exclusive) >= @char_guard_threshold do + [{:__char_guard__, [], [{modifier, inclusive, exclusive}, var]}] + else + compile_bin_ranges(var, inclusive, exclusive) + end + end + + # So both spellings share a guard. Nothing else is reordered: the order given + # is the order compared. + defp ascending({:not, range}), do: {:not, ascending(range)} + defp ascending(min..max//-1), do: max..min//1 + defp ascending(range), do: range + + defp char_guard_class(inclusive, []), do: @char_guard_classes[inclusive] + defp char_guard_class(_inclusive, _exclusive), do: nil + + defp comparisons(inclusive, exclusive) do + Enum.reduce(inclusive ++ exclusive, 0, &(&2 + range_comparisons(&1))) + end + + defp range_comparisons({:not, range}), do: range_comparisons(range) + defp range_comparisons(_.._//_), do: 2 + defp range_comparisons(char) when is_integer(char), do: 1 + + defp extract_char_guards(defs, char_guards) do + {defs, {char_guards, new}} = + Enum.map_reduce(defs, {char_guards, []}, fn {name, args, guards, body}, acc -> + {guards, acc} = Macro.prewalk(guards, acc, &replace_char_guard/2) + {{name, args, guards, body}, acc} + end) + + {defs, Enum.reverse(new), char_guards} + end + + defp replace_char_guard({:__char_guard__, _, [key, var]}, {char_guards, new}) do + case char_guards do + %{^key => name} -> + {{name, [], [var]}, {char_guards, new}} + + %{} -> + name = char_guard_name(key) + {{name, [], [var]}, {Map.put(char_guards, key, name), [{name, key} | new]}} + end + end + + defp replace_char_guard(node, acc) do + {node, acc} + end + + @char_guard_body_limit 48 + + @doc """ + Returns the name of the guard matching the given ranges. + + The same ranges always get the same name and no two get one another's, so that + parsers can share guards. Underscores keep the name out of the way of the ones + the module it is defined in may use for itself. + """ + def char_guard_name({modifier, inclusive, exclusive}) do + prefix = if modifier == :integer, do: "ascii", else: modifier + + case char_guard_class(inclusive, exclusive) do + nil -> :"__#{prefix}_char_#{char_guard_body(inclusive, exclusive)}" + class -> :"__#{prefix}_#{class}" + end + end + + # `?a` is `a`, `?-` is `0x2d`, `?a..?z` is `a_z` and `[?a, ?z]` is `a__z`. Nothing + # spelling a codepoint can be read as hex or as `not`, so no two sets collide. + defp char_guard_body(inclusive, exclusive) do + body = Enum.map_join(inclusive ++ exclusive, "__", &char_guard_field/1) + + if byte_size(body) > @char_guard_body_limit do + char_guard_hash(body) + else + body + end + end + + defp char_guard_field({:not, range}), do: "not_" <> char_guard_field(range) + defp char_guard_field(min..max//_), do: "#{char_name(min)}_#{char_name(max)}" + defp char_guard_field(char) when is_integer(char), do: char_name(char) + + defp char_name(char) when char in ?0..?9 or char in ?a..?z or char in ?A..?Z, do: <> + + defp char_name(char) do + "0x" <> String.pad_leading(String.downcase(Integer.to_string(char, 16)), 2, "0") + end + + # The body, not the ranges: `term_to_binary/1` has no ordering guarantee across + # releases, and `mix nimble_parsec.compile` writes these names to disk. + defp char_guard_hash(body) do + body + |> :erlang.md5() + |> binary_part(0, 8) + |> Base.encode16(case: :lower) + end + + @doc """ + Compiles `combinator` into `module`, recording it and defining its guards. + + The definitions returned have to be added after the guards, which are macros. + """ + def compile_into(module, file, parser_kind, combinator_kind, name, combinator, opts) do + char_guards = Module.get_attribute(module, :nimble_parsec_char_guards) || %{} + {defs, inline, new_char_guards, char_guards} = compile(name, combinator, char_guards, opts) + Module.put_attribute(module, :nimble_parsec_char_guards, char_guards) + + NimbleParsec.Recorder.record( + module, + parser_kind, + combinator_kind, + name, + defs, + inline, + new_char_guards, + opts + ) + + define_char_guards(module, file, new_char_guards) + {defs, inline} + end + + defp define_char_guards(module, file, char_guards) do + # As options rather than `__ENV__`, which would inline a literal env per parser. + env = [module: module, file: file] + Enum.each(char_guards, &Code.eval_quoted(char_guard_definition(&1), [], env)) + end + + @doc """ + Returns the quoted `defguardp` for a character guard returned by `compile/4`. + """ + def char_guard_definition({name, {_modifier, inclusive, exclusive}}) do + var = Macro.var(:char, nil) + + body = + var + |> compile_bin_ranges(inclusive, exclusive) + |> guards_list_to_quoted() + + quote(do: defguardp(unquote({name, [], [var]}) when unquote(body))) + end + ## Bin segments defp compile_bin_ranges(var, ors, ands) do @@ -1165,31 +1331,48 @@ defmodule NimbleParsec.Compiler do defp bin_range_to_guard(var, range) do case range do min..min//step when abs(step) == 1 -> - quote(do: unquote(var) === unquote(min)) + quote(do: unquote(var) === unquote(char(min))) min..max//1 -> - quote(do: unquote(var) >= unquote(min) and unquote(var) <= unquote(max)) + quote(do: unquote(var) >= unquote(char(min)) and unquote(var) <= unquote(char(max))) min..max//-1 -> - quote(do: unquote(var) >= unquote(max) and unquote(var) <= unquote(min)) + quote(do: unquote(var) >= unquote(char(max)) and unquote(var) <= unquote(char(min))) min when is_integer(min) -> - quote(do: unquote(var) === unquote(min)) + quote(do: unquote(var) === unquote(char(min))) {:not, min..min//step} when abs(step) == 1 -> - quote(do: unquote(var) !== unquote(min)) + quote(do: unquote(var) !== unquote(char(min))) {:not, min..max//1} -> - quote(do: unquote(var) < unquote(min) or unquote(var) > unquote(max)) + quote(do: unquote(var) < unquote(char(min)) or unquote(var) > unquote(char(max))) {:not, min..max//-1} -> - quote(do: unquote(var) < unquote(max) or unquote(var) > unquote(min)) + quote(do: unquote(var) < unquote(char(max)) or unquote(var) > unquote(char(min))) {:not, min} when is_integer(min) -> - quote(do: unquote(var) !== unquote(min)) + quote(do: unquote(var) !== unquote(char(min))) end end + # `?a` and `97` are the same AST node, so the literal spelling only survives as + # `:token` metadata, which `Macro.to_string/1` honours when printing. + defp char(codepoint) do + token = + case codepoint do + ?\\ -> "?\\\\" + ?\s -> "?\\s" + ?\n -> "?\\n" + ?\t -> "?\\t" + ?\r -> "?\\r" + codepoint when codepoint in ?!..?~ -> "?" <> <> + codepoint -> "0x" <> String.pad_leading(Integer.to_string(codepoint, 16), 2, "0") + end + + {:__block__, [token: token], [codepoint]} + end + defp inspect_bin_range(min..max//_, printable?) do {" in the range #{inspect_char(min)} to #{inspect_char(max)}", printable? and printable?(min) and printable?(max)} diff --git a/lib/nimble_parsec/recorder.ex b/lib/nimble_parsec/recorder.ex index 9b8fa77..2af6d21 100644 --- a/lib/nimble_parsec/recorder.ex +++ b/lib/nimble_parsec/recorder.ex @@ -20,18 +20,24 @@ defmodule NimbleParsec.Recorder do @doc """ Records the given call and potentially debugs it. """ - def record(module, parser_kind, combinator_kind, name, combinators, inline, opts) do + def record(module, parser_kind, combinator_kind, name, combinators, inline, char_guards, opts) do inline? = Keyword.get(opts, :inline, false) if Keyword.get(opts, :debug, false) do - IO.puts(format_defs(combinator_kind, combinators, inline, inline?)) + IO.puts([ + format_char_guards(char_guards), + format_defs(combinator_kind, combinators, inline, inline?) + ]) end if Process.whereis(@name) do Agent.update(@name, fn state -> update_in( state[module], - &[{parser_kind, combinator_kind, name, combinators, inline, inline?} | &1 || []] + &[ + {parser_kind, combinator_kind, name, combinators, inline, inline?, char_guards} + | &1 || [] + ] ) end) end @@ -39,6 +45,13 @@ defmodule NimbleParsec.Recorder do :ok end + defp format_char_guards(char_guards) do + Enum.map(char_guards, fn char_guard -> + definition = NimbleParsec.Compiler.char_guard_definition(char_guard) + [Macro.to_string(definition), "\n\n"] + end) + end + defp format_parser_kind(nil, _name) do [] end @@ -104,8 +117,15 @@ defmodule NimbleParsec.Recorder do case String.split(acc, marker) do [pre, _middle, pos] -> + # Guards are macros: they have to precede every definition using them. + char_guards = + entries + |> Enum.reverse() + |> Enum.flat_map(fn entry -> elem(entry, 6) end) + |> format_char_guards() + replacement = Enum.map(entries, &format_recorded/1) - IO.iodata_to_binary([pre, replacement, pos]) + IO.iodata_to_binary([pre, char_guards, replacement, pos]) [_, _] -> raise ArgumentError, "expected 2 markers #{inspect(marker)} on #{inspect(id)}, got 1" @@ -116,7 +136,9 @@ defmodule NimbleParsec.Recorder do end) end - defp format_recorded({parser_kind, combinator_kind, name, combinators, inline, inline?}) do + defp format_recorded( + {parser_kind, combinator_kind, name, combinators, inline, inline?, _guards} + ) do [ format_parser_kind(parser_kind, name) | format_defs(combinator_kind, combinators, inline, inline?) diff --git a/test/mix/tasks/nimble_parsec.compile_test.exs b/test/mix/tasks/nimble_parsec.compile_test.exs index 444c86f..1fc3eec 100644 --- a/test/mix/tasks/nimble_parsec.compile_test.exs +++ b/test/mix/tasks/nimble_parsec.compile_test.exs @@ -28,6 +28,11 @@ defmodule Mix.Tasks.NimbleParsec.CompileTest do defcombinator :combinator, integer(2) defcombinatorp :combinatorp, integer(2) + defparsec :alnum, ascii_string([?a..?z, ?A..?Z, ?0..?9], min: 1) + defparsec :alnum_again, ascii_char([?a..?z, ?A..?Z, ?0..?9]) + defparsec :lower, ascii_char([?a..?z]) + defparsec :dash_or_ab, ascii_char([?-]) |> concat(ascii_char([?a, ?b])) + # parsec:Mix.Tasks.NimbleParsec.CompileTest.Parser _pos = :ok @@ -50,6 +55,20 @@ defmodule Mix.Tasks.NimbleParsec.CompileTest do refute contents =~ "defp combinatorp(binary, opts \\\\ [])" assert contents =~ "defp combinatorp__0(" assert contents =~ " _pos = :ok\nend" + + # Emitted once per module, shared by every parser using the same ranges. + assert contents =~ "defguardp __ascii_alnum(char)" + assert length(String.split(contents, "defguardp __ascii_alnum(char)")) == 2 + assert contents =~ "when __ascii_alnum(x0)" + + # Ahead of the first definition, not merely of the ones needing it. + assert [guards, defs] = String.split(contents, "@doc \"\"\"", parts: 2) + assert guards =~ "defguardp __ascii_lower(char)" + assert guards =~ "defguardp __ascii_digit(char)" + refute defs =~ "defguardp " + + # Ranges below the threshold stay expanded: the call would be longer. + refute contents =~ "defguardp __ascii_char_" end) # Ensure the output is also compilable. diff --git a/test/nimble_parsec_test.exs b/test/nimble_parsec_test.exs index e14de9a..758da3b 100644 --- a/test/nimble_parsec_test.exs +++ b/test/nimble_parsec_test.exs @@ -1638,6 +1638,319 @@ defmodule NimbleParsecTest do end end + describe "generated guards" do + test "spell codepoints out as literals" do + assert guard_source(ascii_char([?a..?f])) =~ "x0 >= ?a and x0 <= ?f" + assert guard_source(ascii_char([?\s, ?\n])) =~ "x0 === ?\\s or x0 === ?\\n" + assert guard_source(ascii_char([?\t, ?\r])) =~ "x0 === ?\\t or x0 === ?\\r" + assert guard_source(ascii_char([?\\, ?~])) =~ "x0 === ?\\\\ or x0 === ?~" + assert guard_source(ascii_char(not: ?q)) =~ "x0 !== ?q" + + # Codepoints with no printable spelling are spelled in hex, a byte wide. + assert guard_source(ascii_char([0x00, 0x1B])) =~ "x0 === 0x00 or x0 === 0x1B" + assert guard_source(utf8_char([?é, ?ą])) =~ "x0 === 0xE9 or x0 === 0x105" + assert guard_source(utf8_char([0x1F600, 0x1F601])) =~ "x0 === 0x1F600 or x0 === 0x1F601" + end + + defp guard_source(combinator) do + {defs, _inline, _new, _seen} = + NimbleParsec.Compiler.compile(:literals, combinator, %{}, []) + + Enum.map_join(defs, "\n", fn {_name, _args, guards, _body} -> Macro.to_string(guards) end) + end + end + + describe "character guards" do + defparsecp :guarded_alnum, ascii_string([?a..?z, ?A..?Z, ?0..?9], 3) + defparsecp :guarded_except, ascii_char([?a..?z, ?A..?Z, not: ?q]) + defparsecp :guarded_symbols, ascii_char([?!, ?-, ?~]) + defparsecp :guarded_utf8, utf8_char([?à..?ż, ?a..?z, not: ?q]) + defparsecp :guarded_empty, ascii_char([?z..?a//1]) + defparsecp :guarded_with_empty, ascii_char([?z..?a//1, ?0..?9, ?A..?F]) + defparsecp :digit_chars, ascii_char([?0..?5, ?a..?z]) + defparsecp :raw_codepoints, ascii_char([0..5, ?a..?z]) + defparsecp :order_as_given, ascii_char([?0..?9, ?a..?s, ?v..?w, ?y..?z]) + defparsecp :order_shuffled, ascii_char([?y..?z, ?v..?w, ?a..?s, ?0..?9]) + defparsecp :order_descending, ascii_char([?z..?y//-1, ?w..?v//-1, ?s..?a//-1, ?9..?0//-1]) + + test "keep the parser semantics" do + assert guarded_alnum("aZ9") == {:ok, ["aZ9"], "", %{}, {1, 0}, 3} + assert {:error, _, "a!9", %{}, {1, 0}, 0} = guarded_alnum("a!9") + + assert guarded_except("p") == {:ok, ~c"p", "", %{}, {1, 0}, 1} + assert {:error, _, "q", %{}, {1, 0}, 0} = guarded_except("q") + + assert guarded_symbols("~") == {:ok, ~c"~", "", %{}, {1, 0}, 1} + assert {:error, _, "a", %{}, {1, 0}, 0} = guarded_symbols("a") + + assert guarded_utf8("ż") == {:ok, ~c"ż", "", %{}, {1, 0}, 2} + assert guarded_utf8("a") == {:ok, ~c"a", "", %{}, {1, 0}, 1} + assert {:error, _, "q", %{}, {1, 0}, 0} = guarded_utf8("q") + assert {:error, _, "Ω", %{}, {1, 0}, 0} = guarded_utf8("Ω") + end + + test "keep an empty range matching nothing" do + # Dropped instead of compiled, it would match anything rather than nothing. + assert {:error, _, "q", %{}, {1, 0}, 0} = guarded_empty("q") + assert {:error, _, "z", %{}, {1, 0}, 0} = guarded_empty("z") + + assert guarded_with_empty("5") == {:ok, ~c"5", "", %{}, {1, 0}, 1} + assert {:error, _, "q", %{}, {1, 0}, 0} = guarded_with_empty("q") + end + + test "are named after the well-known class they match" do + assert [{:__ascii_digit, _}] = guards_for(ascii_char([?0..?9])) + assert [{:__ascii_lower, _}] = guards_for(ascii_char([?a..?z])) + assert [{:__ascii_upper, _}] = guards_for(ascii_char([?A..?Z])) + assert [{:__ascii_alpha, _}] = guards_for(ascii_char([?a..?z, ?A..?Z])) + assert [{:__ascii_alnum, _}] = guards_for(ascii_char([?a..?z, ?A..?Z, ?0..?9])) + assert [{:__ascii_hex, _}] = guards_for(ascii_char([?0..?9, ?a..?f, ?A..?F])) + assert [{:__ascii_space, _}] = guards_for(ascii_char([?\s, ?\t, ?\n, ?\r])) + + # The modifier separates them, though the comparisons are the same. + assert [{:__utf8_alnum, _}] = guards_for(utf8_char([?a..?z, ?A..?Z, ?0..?9])) + end + + test "are named after the ranges they match otherwise" do + # A class spelled in another order is a different guard, and so is one + # narrowed by an exclusive range. + assert [{:__ascii_char_0_9__A_Z__a_z, _}] = guards_for(ascii_char([?0..?9, ?A..?Z, ?a..?z])) + + assert [{:__ascii_char_a_z__A_Z__0_9__not_q, _}] = + guards_for(ascii_char([?a..?z, ?A..?Z, ?0..?9, not: ?q])) + + assert [{:__utf8_char_0xe0_0x17c__a_z, _}] = guards_for(utf8_char([?à..?ż, ?a..?z])) + end + + test "collapse a descending range onto its ascending counterpart" do + assert [{name, key}] = guards_for(ascii_char([?a..?z, ?A..?Z, ?0..?5])) + assert [{^name, ^key}] = guards_for(ascii_char([?z..?a//-1, ?Z..?A//-1, ?5..?0//-1])) + end + + test "are not emitted for ranges cheaper to expand inline" do + assert [] = guards_for(ascii_char([?a, ?b])) + assert [] = guards_for(ascii_char([?a..?f])) + assert [] = guards_for(ascii_char(not: ?q)) + end + + test "compare the ranges in the order they were given" do + # The comparisons short-circuit, so the order is the caller's to choose. + assert compared_codepoints(ascii_char([?a..?z, ?A..?Z])) == [?a, ?z, ?A, ?Z] + assert compared_codepoints(ascii_char([?A..?Z, ?a..?z])) == [?A, ?Z, ?a, ?z] + + assert compared_codepoints(ascii_char([?A..?Z, ?a..?z, not: ?q])) == + [?A, ?Z, ?a, ?z, ?q] + end + + test "match the same codepoints whatever order the ranges are given in" do + expected = Enum.to_list(?0..?9) ++ Enum.to_list(?a..?s) ++ [?v, ?w, ?y, ?z] + + assert accepted_by(&order_as_given/1) == expected + assert accepted_by(&order_shuffled/1) == expected + assert accepted_by(&order_descending/1) == expected + end + + test "tell a character range apart from the codepoints numbering it" do + assert [{chars, _}] = guards_for(ascii_char([?0..?5, ?a..?z])) + assert [{codepoints, _}] = guards_for(ascii_char([0..5, ?a..?z])) + refute chars == codepoints + + assert accepted_by(&digit_chars/1) == Enum.to_list(?0..?5) ++ Enum.to_list(?a..?z) + assert accepted_by(&raw_codepoints/1) == Enum.to_list(0..5) ++ Enum.to_list(?a..?z) + end + + test "are shared by every parser using the same ranges" do + ranges = [?a..?z, ?A..?Z, ?0..?9] + assert {_, _, [{name, _}], seen} = compile_with_guards(ascii_char(ranges)) + + assert {_, _, [], ^seen} = + NimbleParsec.Compiler.compile(:reuse, ascii_char(ranges), seen, []) + + assert seen == %{{:integer, ranges, []} => name} + end + + test "are never left as a placeholder in a definition" do + ranges = [?a..?z, ?A..?Z, ?0..?9] + + combinators = [ + ascii_char(ranges), + ascii_char(ranges) |> concat(utf8_char(ranges)), + ascii_string(ranges, min: 1), + ascii_string(ranges, 3), + utf8_string(ranges, max: 3), + optional(ascii_char(ranges)), + repeat(ascii_char(ranges)), + times(ascii_char(ranges), min: 2), + lookahead(ascii_char(ranges)), + lookahead_not(ascii_char(ranges)), + eventually(ascii_char(ranges)), + choice([ascii_char(ranges), ascii_char([?0..?9, ?x, ?y])]), + ascii_char(ranges) |> label("labelled"), + ascii_char(ranges) |> map({String, :to_string, []}) + ] + + for combinator <- combinators do + {defs, _inline, new, _seen} = compile_with_guards(combinator) + assert new != [] + + # A surviving placeholder is a compile error in the generated module. + refute inspect(defs) =~ "__char_guard__" + end + end + + test "are printed ahead of the definitions using them on :debug" do + output = + ExUnit.CaptureIO.capture_io(fn -> + Code.compile_string(""" + defmodule DebuggedGuards do + import NimbleParsec + defparsecp :first, ascii_char([?a..?z, ?A..?Z, ?0..?9]), debug: true + defparsecp :second, ascii_char([?a..?z, ?A..?Z, ?0..?9]), debug: true + end + """) + end) + + assert [first, second] = String.split(output, "defp second__0", parts: 2) + assert [before_defs, _] = String.split(first, "defp first__0", parts: 2) + assert before_defs =~ "defguardp __ascii_alnum(char)" + + # Only the parser introducing it prints it. + refute second =~ "defguardp" + assert second =~ "when __ascii_alnum(x0)" + end + + defp accepted_by(parser) do + for char <- 0..127, match?({:ok, _, _, _, _, _}, parser.(<>)), do: char + end + + defp compile_with_guards(combinator, char_guards \\ %{}) do + NimbleParsec.Compiler.compile(:guarded, combinator, char_guards, []) + end + + defp guards_for(combinator) do + {_defs, _inline, new, _seen} = compile_with_guards(combinator) + new + end + + defp compared_codepoints(combinator) do + [guard] = guards_for(combinator) + + guard + |> NimbleParsec.Compiler.char_guard_definition() + |> Macro.prewalk([], fn + node, acc when is_integer(node) -> {node, [node | acc]} + node, acc -> {node, acc} + end) + |> elem(1) + |> Enum.reverse() + end + end + + describe "character guard names" do + test "spell a codepoint as itself when a function name can hold it" do + assert guard_name([?1, ?2, ?3]) == :__ascii_char_1__2__3 + assert guard_name([?a, ?B, ?9]) == :__ascii_char_a__B__9 + assert guard_name([?a..?f, ?x]) == :__ascii_char_a_f__x + end + + test "spell every other codepoint in hex" do + assert guard_name([?!, ?-, ?~]) == :__ascii_char_0x21__0x2d__0x7e + assert guard_name([?\s, ?\t, ?\n]) == :__ascii_char_0x20__0x09__0x0a + assert guard_name([?_, ?a, ?b]) == :__ascii_char_0x5f__a__b + assert guard_name([?à..?ż, ?a], :utf8) == :__utf8_char_0xe0_0x17c__a + end + + test "join the bounds of a range by one underscore and ranges by two" do + assert guard_name([?a..?b, ?z]) == :__ascii_char_a_b__z + assert guard_name([?a, ?b..?z]) == :__ascii_char_a__b_z + assert guard_name([?a, ?b, ?z]) == :__ascii_char_a__b__z + assert guard_name([0..5, ?a]) == :__ascii_char_0x00_0x05__a + end + + test "prefix an excluded range by not" do + assert guard_name([?a..?z, not: ?q]) == :__ascii_char_a_z__not_q + assert guard_name([?a..?z, ?A, not: ?q..?t]) == :__ascii_char_a_z__A__not_q_t + end + + test "fall back to a hash of the name once it stops being one" do + long = [?a..?z, ?A..?Z, ?0..?9, ?!, ?-, ?~, ?., ?,, ?;, not: ?q] + assert Atom.to_string(guard_name(long)) =~ ~r/^__ascii_char_[0-9a-f]{16}$/ + + assert guard_name(long) == guard_name(long) + refute guard_name(long) == guard_name(tl(long)) + end + + test "are usable as function names in a guard" do + for ranges <- range_corpus(), name = guard_name(ranges) do + assert Atom.to_string(name) =~ ~r/^[a-z_][a-zA-Z0-9_]*$/ + + # The file mix nimble_parsec.compile writes has to parse. + assert Code.string_to_quoted!("defp f(c) when #{name}(c), do: :ok") + end + end + + test "are unique for every distinct set of ranges" do + # A shared name would silently give one parser the other's comparisons. + names = + for ranges <- range_corpus(), modifier <- [:integer, :utf8], reduce: %{} do + names -> + name = guard_name(ranges, modifier) + key = {modifier, ranges} + + case names do + %{^name => ^key} -> + names + + %{^name => other} -> + flunk("#{name} is shared by #{inspect(key)} and #{inspect(other)}") + + %{} -> + Map.put(names, name, key) + end + end + + assert map_size(names) == length(range_corpus()) * 2 + end + + # Ranges that spell each other in more than one way. They all ascend, as + # `compile/4` collapses descending ranges before naming them. + defp range_corpus do + entries = [ + ?a, + ?z, + ?0, + ?x, + ?1, + ?2, + ?n, + ?o, + ?t, + ?_, + ?-, + 0x12, + 0x1, + 0x2, + ?a..?z, + ?0..?5, + 0..5, + ?2..?x, + ?x..?2//1 + ] + + singles = for entry <- entries, do: [entry] + excluded = for entry <- entries, do: [?A..?Z, {:not, entry}] + pairs = for left <- entries, right <- entries, do: [left, right] + triples = for left <- entries, right <- entries, do: [left, ?B, right] + + Enum.uniq(singles ++ excluded ++ pairs ++ triples) + end + + defp guard_name(ranges, modifier \\ :integer) do + {inclusive, exclusive} = Enum.split_with(ranges, &(not match?({:not, _}, &1))) + NimbleParsec.Compiler.char_guard_name({modifier, inclusive, exclusive}) + end + end + describe "continuing parser" do defparsecp :digits, [?0..?9] |> ascii_char() |> times(min: 1) |> label("digits") defparsecp :chars, [?a..?z] |> ascii_char() |> times(min: 1) |> label("chars") @@ -1664,14 +1977,14 @@ defmodule NimbleParsecTest do end defp bound?(document) do - {defs, _} = NimbleParsec.Compiler.compile(:not_used, document, []) + {defs, _, _, _} = NimbleParsec.Compiler.compile(:not_used, document, %{}, []) assert length(defs) == 3, "Expected #{inspect(document)} to contain 3 clauses, got #{length(defs)}" end defp not_bound?(document) do - {defs, _} = NimbleParsec.Compiler.compile(:not_used, document, []) + {defs, _, _, _} = NimbleParsec.Compiler.compile(:not_used, document, %{}, []) assert length(defs) != 3, "Expected #{inspect(document)} to contain greater than 3 clauses" end