From 4a35711438acd7ed67ad3fdd7e17fc76b78f042c Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 12:20:08 +0000 Subject: [PATCH] Add GROUP BY and aggregates Supports `select status, count(*) as n, sum(total) from t group by status` with count(*), count(field), sum, min, max and avg. Aggregates can take an alias. An aggregate without GROUP BY covers the whole table. ORDER BY can name a group field, an aggregate or an alias, and LIMIT limits the groups. The grammar gains GROUP BY, aggregate calls in the select list and ORDER BY, and AS aliases. The translator checks grouping rules (selected fields must be grouped, known aggregate names, * only for count) and names the output columns. The planner plans the underlying scan as before, reading only the fields that are grouped or aggregated, so WHERE keeps index and primary-key pushdown. An {:aggregate, ...} operator (Efsql.Aggregate) then groups the rows, followed by sort, limit and project. SQL semantics: count(*) counts rows and the other aggregates skip NULLs. NULLs form one group. Values that compare equal group together even when their terms differ, like Decimal 1.0 and 1.00. An aggregate over no rows still returns one row. sum and avg are Decimal when any input is. HAVING stays unsupported. Completion, the help page and the README cover the new syntax, and the random query generator produces aggregates and GROUP BY for the property tests. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01PFBZUrerge6bbWVCHpS5Ko --- README.md | 26 +++++- lib/efsql/aggregate.ex | 132 +++++++++++++++++++++++++++++ lib/efsql/complete.ex | 19 +++-- lib/efsql/executor.ex | 4 + lib/efsql/logical.ex | 12 ++- lib/efsql/parser.ex | 113 +++++++++++++++++++++++- lib/efsql/physical.ex | 2 + lib/efsql/planner.ex | 40 +++++++++ lib/efsql/sql/ast.ex | 15 +++- lib/efsql/sql/parser.ex | 27 ++++-- lib/efsql/tui/help.ex | 9 ++ src/efsql_sql_grammar.yrl | 39 +++++++-- test/aggregate_test.exs | 88 +++++++++++++++++++ test/complete_test.exs | 10 +++ test/group_by_test.exs | 86 +++++++++++++++++++ test/parser_test.exs | 63 ++++++++++++++ test/sql/parser_test.exs | 52 +++++++++++- test/support/efsql_test/sql_gen.ex | 39 +++++++-- 18 files changed, 734 insertions(+), 42 deletions(-) create mode 100644 lib/efsql/aggregate.ex create mode 100644 test/aggregate_test.exs create mode 100644 test/group_by_test.exs diff --git a/README.md b/README.md index dfb79ea..3a77621 100644 --- a/README.md +++ b/README.md @@ -171,7 +171,8 @@ so if you forget. Comments (`-- ...` and `/* ... */`) can go anywhere. A syntax error reports the line and column it was found at, and features efsql doesn't support (`OR`, -`NOT`, `<>`, joins, functions, `GROUP BY`) are rejected by name. +`NOT`, `<>`, joins, `HAVING`, functions other than aggregates) are rejected by +name. ### Storage IDs @@ -303,6 +304,29 @@ Typed literals work anywhere a value does, including `IN` and `BETWEEN`: select col_a from tenant_id.table_name where inserted_at between '2024-01-01'::timestamp and '2025-01-01'::timestamp; ``` +### Group and aggregate + +```sql +-- one row for the whole table +select count(*) from tenant_id.table_name; +select min(inserted_at), max(inserted_at) from tenant_id.table_name where status = 'paid'; + +-- one row per group +select status, count(*) as n, sum(total) from tenant_id.table_name group by status; +select status, avg(total) from tenant_id.table_name group by status order by avg(total) desc limit 3; +``` + +The aggregates are `count(*)`, `count(field)`, `sum`, `min`, `max` and +`avg`, and `AS` names one. As in SQL, `count(*)` counts rows while every +other aggregate skips NULLs, and NULLs form one group. `sum` and `avg` take +numbers and `Decimal`s; `min` and `max` also work on strings and datetimes. + +A selected field must be in `GROUP BY`. `ORDER BY` can name a group field, +an aggregate or its alias, and `LIMIT` limits the groups. `WHERE` is applied +before grouping and gets the usual index and primary key pushdown; the +grouping itself happens after the rows are read. `HAVING` is not supported +yet. + ### Limit ```sql diff --git a/lib/efsql/aggregate.ex b/lib/efsql/aggregate.ex new file mode 100644 index 0000000..3ced7db --- /dev/null +++ b/lib/efsql/aggregate.ex @@ -0,0 +1,132 @@ +defmodule Efsql.Aggregate do + @moduledoc """ + `GROUP BY` and aggregates over pulled rows, for `Efsql.Executor`'s + `{:aggregate, group_by, aggregates}` operator. + + Rows are grouped by the values of the `group_by` fields; equal values + group together even when their terms differ (`Decimal` 1.0 and 1.00, or + datetimes stored at different precisions), and NULLs form one group, as + in SQL. Each group becomes one row holding its group fields and one + column per aggregate. With no group fields, every row is one group, so + an empty input still gives one row (`count(*)` of nothing is 0). + + SQL semantics: `count(*)` counts rows, every other aggregate skips NULLs, + and `sum`, `min`, `max` and `avg` of no values are NULL. `sum` and `avg` + take numbers and `Decimal`s, and are `Decimal` when any input is. + `min` and `max` order with `Efsql.Types.compare/2`, so they work on + strings and datetimes too. Groups come out ordered by their key. + """ + + alias Efsql.Exception.Unsupported + alias Efsql.Types + + @type aggregate :: {name :: atom, :count | :sum | :min | :max | :avg, atom | :star} + + @spec run([map], [atom], [aggregate]) :: [map] + def run(rows, group_by, aggregates) do + rows + |> Enum.group_by(fn row -> Enum.map(group_by, &key(Map.get(row, &1))) end) + |> ensure_one_group(group_by) + |> Enum.map(fn {_key, rows} -> + first = List.first(rows, %{}) + group = Map.new(group_by, &{&1, Map.get(first, &1)}) + + Enum.reduce(aggregates, group, fn {name, function, arg}, acc -> + Map.put(acc, name, compute(function, arg, rows)) + end) + end) + |> Enum.sort(&(compare_groups(&1, &2, group_by) != :gt)) + end + + # A whole-table aggregate answers even when no row matched. + defp ensure_one_group(groups, []) when groups == %{}, do: [{[], []}] + defp ensure_one_group(groups, _group_by), do: Enum.to_list(groups) + + # A term that is equal for values Types.compare/2 calls equal. + defp key(%Decimal{} = d), do: Decimal.normalize(d) + defp key(%NaiveDateTime{microsecond: {us, _}} = t), do: %{t | microsecond: {us, 6}} + defp key(%DateTime{microsecond: {us, _}} = t), do: %{t | microsecond: {us, 6}} + defp key(%Time{microsecond: {us, _}} = t), do: %{t | microsecond: {us, 6}} + defp key(value), do: value + + # -- aggregates -- + + defp compute(:count, :star, rows), do: length(rows) + defp compute(:count, field, rows), do: length(values(rows, field)) + + defp compute(:min, field, rows), do: extreme(values(rows, field), :lt) + defp compute(:max, field, rows), do: extreme(values(rows, field), :gt) + + defp compute(:sum, field, rows) do + case values(rows, field) do + [] -> nil + values -> sum(values, field) + end + end + + defp compute(:avg, field, rows) do + case values(rows, field) do + [] -> + nil + + values -> + case sum(values, field) do + %Decimal{} = total -> Decimal.div(total, length(values)) + total -> total / length(values) + end + end + end + + defp values(rows, field) do + rows |> Enum.map(&Map.get(&1, field)) |> Enum.reject(&is_nil/1) + end + + defp extreme([], _keep), do: nil + + defp extreme([first | rest], keep) do + Enum.reduce(rest, first, fn value, best -> + if Types.compare(value, best) == keep, do: value, else: best + end) + end + + defp sum(values, field) do + Enum.each(values, fn value -> + unless is_number(value) or is_struct(value, Decimal) do + raise Unsupported, + "sum and avg need numbers, but #{field} holds #{inspect(value, limit: 5)}" + end + end) + + if Enum.any?(values, &is_struct(&1, Decimal)), + do: values |> Enum.map(&decimal/1) |> Enum.reduce(&Decimal.add/2), + else: Enum.sum(values) + end + + defp decimal(%Decimal{} = d), do: d + defp decimal(n) when is_integer(n), do: Decimal.new(n) + defp decimal(n) when is_float(n), do: Decimal.from_float(n) + + # -- group order -- + + # By key, NULLs last, like an ascending ORDER BY. + defp compare_groups(_a, _b, []), do: :eq + + defp compare_groups(a, b, [field | rest]) do + case {Map.get(a, field), Map.get(b, field)} do + {nil, nil} -> + compare_groups(a, b, rest) + + {nil, _} -> + :gt + + {_, nil} -> + :lt + + {va, vb} -> + case Types.compare(va, vb) do + :eq -> compare_groups(a, b, rest) + cmp -> cmp + end + end + end +end diff --git a/lib/efsql/complete.ex b/lib/efsql/complete.ex index a17a8a6..fe86918 100644 --- a/lib/efsql/complete.ex +++ b/lib/efsql/complete.ex @@ -13,7 +13,7 @@ defmodule Efsql.Complete do """ @statement_start ~w(select) - @post_expr ~w(and order limit) + @post_expr ~w(and group order limit) @operators ~w(= > >= < <= like in between not is) def complete(input, context) do @@ -53,14 +53,14 @@ defmodule Efsql.Complete do last in ["where", "and", "not"] -> fields(context, tokens) - last == "order" -> + last in ["order", "group"] -> ~w(by) last == "by" -> fields(context, tokens) last in ["asc", "desc"] -> - @post_expr -- ["order"] + ~w(limit) last in @operators -> [] @@ -69,8 +69,9 @@ defmodule Efsql.Complete do true -> case section(tokens) do :select -> ~w(from) - :from -> ~w(where order limit) + :from -> ~w(where group order limit) :where -> @operators + :group_by -> ~w(order limit) :order_by -> ~w(asc desc limit) _ -> [] end @@ -95,11 +96,13 @@ defmodule Efsql.Complete do defp section(tokens) do tokens |> Enum.reverse() + |> Enum.chunk_every(2, 1) |> Enum.find_value(:start, fn - "by" -> :order_by - "where" -> :where - "from" -> :from - "select" -> :select + ["by", "group"] -> :group_by + ["by" | _] -> :order_by + ["where" | _] -> :where + ["from" | _] -> :from + ["select" | _] -> :select _ -> nil end) end diff --git a/lib/efsql/executor.ex b/lib/efsql/executor.ex index 1e9f6f0..f93ab22 100644 --- a/lib/efsql/executor.ex +++ b/lib/efsql/executor.ex @@ -64,6 +64,10 @@ defmodule Efsql.Executor do Enum.filter(rows, fn row -> Enum.all?(predicates, &eval(&1, row)) end) end + defp apply_op({:aggregate, group_by, aggregates}, rows) do + Efsql.Aggregate.run(rows, group_by, aggregates) + end + defp apply_op({:sort, sort}, rows) do Enum.sort(rows, fn a, b -> compare(a, b, sort) != :gt end) end diff --git a/lib/efsql/logical.ex b/lib/efsql/logical.ex index 384d346..7219823 100644 --- a/lib/efsql/logical.ex +++ b/lib/efsql/logical.ex @@ -14,6 +14,12 @@ defmodule Efsql.Logical do The primary key is the pseudo-field `:_`. + A grouped query (`GROUP BY`, or any aggregate) has `group_by` set to its + key fields (`[]` for one group over every row) and `aggregates` to + `{output_name, function, field | :star}`, `function` one of `:count`, + `:sum`, `:min`, `:max`, `:avg`. Its `projection` and `order` then name + output columns: group fields and aggregate names. + `Efsql.Rewrite` normalizes a logical query, `Efsql.Planner` turns it into an `Efsql.Physical.Plan`. """ @@ -29,7 +35,11 @@ defmodule Efsql.Logical do predicates: [], # [{:asc | :desc, field}] order: [], - limit: nil + limit: nil, + # nil, or the GROUP BY fields of a grouped query + group_by: nil, + # [{output_name :: atom, function :: atom, field :: atom | :star}] + aggregates: [] end def predicate_field({:cmp, _op, field, _value}), do: field diff --git a/lib/efsql/parser.ex b/lib/efsql/parser.ex index 802aed9..6d91190 100644 --- a/lib/efsql/parser.ex +++ b/lib/efsql/parser.ex @@ -24,19 +24,126 @@ defmodule Efsql.Parser do def to_logical(%AST.Select{} = select) do {prefix, source} = split_from(select.from) - %Logical.Select{ - projection: projection(select.fields), + logical = %Logical.Select{ source: source, prefix: prefix, predicates: predicates(select.where), - order: Enum.map(select.order_by, fn {name, dir} -> {dir, field(name)} end), limit: select.limit } + + if grouped?(select), + do: grouped(logical, select), + else: %Logical.Select{ + logical + | projection: projection(select.fields), + order: Enum.map(select.order_by, fn {name, dir} -> {dir, field(name)} end) + } end defp projection(:star), do: :star defp projection(fields), do: Enum.map(fields, &field/1) + # -- GROUP BY and aggregates -- + + @aggregates ~w[count sum min max avg] + + defp grouped?(%AST.Select{fields: fields, group_by: group_by, order_by: order_by}) do + group_by != [] or + (is_list(fields) and Enum.any?(fields, &aggregate?/1)) or + Enum.any?(order_by, fn {key, _dir} -> aggregate?(key) end) + end + + defp aggregate?(item), do: match?({:aggregate, _, _, _}, item) + + # Every selected name must be a group field; ORDER BY may also name an + # aggregate or its alias. An aggregate only in ORDER BY is computed, then + # projected away. + defp grouped(_logical, %AST.Select{fields: :star}) do + raise Unsupported, "SELECT * can't be used with GROUP BY or aggregates; name the fields" + end + + defp grouped(logical, select) do + group_by = Enum.map(select.group_by, &group_field/1) + + {columns, aggregates} = + Enum.map_reduce(select.fields, [], fn + {:aggregate, _, _, _} = call, aggregates -> + {name, aggregate} = aggregate(call) + {name, aggregates ++ [aggregate]} + + name, aggregates -> + field = field(name) + + unless field in group_by do + raise Unsupported, + "#{name} must be in GROUP BY or used in an aggregate (count, sum, ...)" + end + + {field, aggregates} + end) + + {order, aggregates} = + Enum.map_reduce(select.order_by, aggregates, fn + {{:aggregate, _, _, _} = call, dir}, aggregates -> + {name, {_, function, arg} = aggregate} = aggregate(call) + + case Enum.find(aggregates, &match?({_, ^function, ^arg}, &1)) do + {existing, _, _} -> {{dir, existing}, aggregates} + nil -> {{dir, name}, aggregates ++ [aggregate]} + end + + {name, dir}, aggregates -> + field = field(name) + named = Enum.map(aggregates, &elem(&1, 0)) + + unless field in group_by or field in named do + raise Unsupported, + "ORDER BY #{name} must name a GROUP BY field, an aggregate or its alias" + end + + {{dir, field}, aggregates} + end) + + %Logical.Select{ + logical + | projection: columns, + order: order, + group_by: group_by, + aggregates: aggregates + } + end + + defp group_field("_"), + do: raise(Unsupported, "GROUP BY '_' is not supported; use the primary key field name") + + defp group_field(name), do: field(name) + + defp aggregate({:aggregate, function, arg, alias}) do + unless function in @aggregates do + raise Unsupported, + "#{function}() is not supported; the aggregates are #{Enum.join(@aggregates, ", ")}" + end + + arg = + case arg do + :star when function == "count" -> + :star + + :star -> + raise Unsupported, "only count takes *; #{function} needs a field" + + "_" -> + raise Unsupported, + "#{function}('_') is not supported; use the primary key field name" + + name -> + field(name) + end + + name = field(alias || "#{function}(#{if arg == :star, do: "*", else: arg})") + {name, {name, String.to_atom(function), arg}} + end + defp split_from([table]), do: {nil, table} defp split_from([tenant, table]), do: {tenant, table} defp split_from([storage, tenant, table]), do: {{storage, tenant}, table} diff --git a/lib/efsql/physical.ex b/lib/efsql/physical.ex index 217eae1..1b71d12 100644 --- a/lib/efsql/physical.ex +++ b/lib/efsql/physical.ex @@ -18,6 +18,8 @@ defmodule Efsql.Physical do Operators: * `{:filter, [predicate]}` — `Efsql.Logical` predicates, SQL NULL semantics + * `{:aggregate, group_by, aggregates}` — one row per group, see + `Efsql.Aggregate` * `{:sort, [{:asc | :desc, field}]}` — NULLs last ascending, first descending * `{:limit, n}` * `{:project, [field]}` — trim rows to the requested fields diff --git a/lib/efsql/planner.ex b/lib/efsql/planner.ex index f0bd294..8777ef7 100644 --- a/lib/efsql/planner.ex +++ b/lib/efsql/planner.ex @@ -17,6 +17,10 @@ defmodule Efsql.Planner do `select *` behaves like any field list: index-constrained queries are served by `Repo.all_from_source` (full data objects, no Ecto select required), pk constraints by `Repo.all_range`. + + A grouped query is planned as the plain query that pulls the fields its + groups and aggregates read, so `WHERE` gets the same pushdown; grouping, + ordering and the limit then run on the pulled rows. """ alias Efsql.Exception.Unsupported @@ -27,6 +31,39 @@ defmodule Efsql.Planner do @pk_field :_ + def plan(%Logical.Select{group_by: group_by} = logical, options) when is_list(group_by) do + %Logical.Select{aggregates: aggregates, order: order, limit: limit} = logical + + read = Enum.uniq(group_by ++ for({_name, _fun, f} <- aggregates, f != :star, do: f)) + + %Plan{} = + plan = + plan( + %Logical.Select{ + logical + | projection: if(read == [], do: :star, else: read), + order: [], + limit: nil, + group_by: nil, + aggregates: [] + }, + options + ) + + columns = group_by ++ Enum.map(aggregates, &elem(&1, 0)) + + ops = + [{:aggregate, group_by, aggregates}] + |> append_if(order != [], {:sort, order}) + |> append_if(limit != nil, {:limit, limit}) + |> append_if( + Enum.sort(columns) != Enum.sort(logical.projection), + {:project, logical.projection} + ) + + %Plan{plan | ops: plan.ops ++ ops} + end + def plan(%Logical.Select{} = logical, options) do %Logical.Select{predicates: preds, order: sort, projection: projection} = logical star? = projection == :star @@ -62,6 +99,9 @@ defmodule Efsql.Planner do A plain Ecto query with every predicate as a where clause, for callers that hand the query to the adapter unplanned (e.g. `Repo.stream`). """ + def to_ecto_query(%Logical.Select{group_by: group_by}) when is_list(group_by), + do: raise(Unsupported, "GROUP BY and aggregates can't be streamed") + def to_ecto_query(%Logical.Select{} = logical) do take = if logical.projection == :star, do: nil, else: logical.projection diff --git a/lib/efsql/sql/ast.ex b/lib/efsql/sql/ast.ex index f17c784..017fe8d 100644 --- a/lib/efsql/sql/ast.ex +++ b/lib/efsql/sql/ast.ex @@ -13,18 +13,27 @@ defmodule Efsql.SQL.AST do @moduledoc "A `SELECT` statement." @type t :: %__MODULE__{ - fields: :star | [String.t()], + fields: :star | [String.t() | Efsql.SQL.AST.aggregate()], from: [String.t(), ...], where: Efsql.SQL.AST.expr() | nil, - order_by: [{String.t(), :asc | :desc}], + group_by: [String.t()], + order_by: [{String.t() | Efsql.SQL.AST.aggregate(), :asc | :desc}], limit: non_neg_integer() | nil } # `from` holds the dotted name's parts: [table], [tenant, table] or # [storage_id, tenant, table]. - defstruct fields: :star, from: [], where: nil, order_by: [], limit: nil + defstruct fields: :star, from: [], where: nil, group_by: [], order_by: [], limit: nil end + @typedoc """ + An aggregate call in the select list or ORDER BY: + `{:aggregate, function, argument, alias}`. The function is any lower-case + name (`count`, `sum`, ...), the argument a field name or `:star`, and the + alias the `AS` name or `nil`. + """ + @type aggregate :: {:aggregate, String.t(), String.t() | :star, String.t() | nil} + @typedoc """ An expression. diff --git a/lib/efsql/sql/parser.ex b/lib/efsql/sql/parser.ex index f97340e..2ba9956 100644 --- a/lib/efsql/sql/parser.ex +++ b/lib/efsql/sql/parser.ex @@ -2,7 +2,11 @@ defmodule Efsql.SQL.Parser do @moduledoc """ Parses one `SELECT` statement into an `Efsql.SQL.AST.Select`. - SELECT fields FROM name [WHERE expr] [ORDER BY items] [LIMIT n] [;] + SELECT items FROM name [WHERE expr] [GROUP BY names] [ORDER BY items] + [LIMIT n] [;] + + An item is a name or an aggregate call, `count(*)` or `sum(price)`, + optionally named with `AS alias`. The grammar is `src/efsql_sql_grammar.yrl`, compiled by yecc; this module feeds it `Efsql.SQL.Lexer` tokens and explains its errors. @@ -22,7 +26,7 @@ defmodule Efsql.SQL.Parser do turn and keeps those the grammar accepts, so messages follow the grammar with no upkeep: `expected BY, got the end of the statement`. A few hints on top name features that aren't supported yet, such as - `GROUP BY` or functions; drop a hint when its feature lands. + `HAVING` or joins; drop a hint when its feature lands. """ alias Efsql.SQL.AST @@ -56,7 +60,6 @@ defmodule Efsql.SQL.Parser do @not_select ~w[insert update delete create drop alter truncate grant revoke explain] @unsupported %{ "distinct" => "DISTINCT", - "group" => "GROUP BY", "having" => "HAVING", "offset" => "OFFSET", "join" => "JOIN", @@ -72,6 +75,9 @@ defmodule Efsql.SQL.Parser do "fetch" => "FETCH" } + @functions_only_aggregates "functions are only supported as aggregates, " <> + "in the select list and ORDER BY" + def reserved_words, do: @reserved @spec parse(String.t()) :: {:ok, AST.Select.t()} | {:error, SyntaxError.t()} @@ -80,9 +86,16 @@ defmodule Efsql.SQL.Parser do grammar_tokens = tokens |> Enum.with_index() |> Enum.map(&to_grammar/1) case :efsql_sql_grammar.parse(grammar_tokens) do - {:ok, {:select, fields, from, where, order_by, limit}} -> + {:ok, {:select, fields, from, where, group_by, order_by, limit}} -> {:ok, - %AST.Select{fields: fields, from: from, where: where, order_by: order_by, limit: limit}} + %AST.Select{ + fields: fields, + from: from, + where: where, + group_by: group_by, + order_by: order_by, + limit: limit + }} {:error, {at, :efsql_sql_grammar, _}} -> tokens = List.to_tuple(tokens) @@ -183,9 +196,9 @@ defmodule Efsql.SQL.Parser do defp hint({:word, "fetch", _}, _, _, _), do: "FETCH is not supported" defp hint({:op, :"(", _}, {:word, word, _}, _, _) when word not in @keywords, - do: "functions are not supported" + do: @functions_only_aggregates - defp hint({:op, :"(", _}, {:quoted, _, _}, _, _), do: "functions are not supported" + defp hint({:op, :"(", _}, {:quoted, _, _}, _, _), do: @functions_only_aggregates defp hint({:op, :., _}, {type, _, _}, _, _) when type in [:word, :quoted], do: "qualified column names are not supported" diff --git a/lib/efsql/tui/help.ex b/lib/efsql/tui/help.ex index b6507ec..cb77878 100644 --- a/lib/efsql/tui/help.ex +++ b/lib/efsql/tui/help.ex @@ -68,6 +68,15 @@ defmodule Efsql.Tui.Help do {:sql, "select id from users order by city asc, name desc;"}, {:note, "efsql sorts after reading unless an index can serve the order"} ]}, + {"Grouping", + [ + {:sql, "select count(*) from users;"}, + {:sql, "select city, count(*) as n from users group by city;"}, + {:sql, "select status, sum(total) from orders group by status;"}, + {:sql, "select city from users group by city order by count(*) desc;"}, + "Aggregates: count(*), count(f), sum, min, max, avg.", + {:note, "selected fields must be grouped; having is not supported"} + ]}, {"How a query runs", [ "Constraints an index or the key can answer are pushed to", diff --git a/src/efsql_sql_grammar.yrl b/src/efsql_sql_grammar.yrl index 5db49a1..887b18a 100644 --- a/src/efsql_sql_grammar.yrl +++ b/src/efsql_sql_grammar.yrl @@ -1,7 +1,11 @@ %% Grammar for efsql's SQL dialect. Efsql.SQL.Parser drives it with %% tokens from Efsql.SQL.Lexer: %% -%% SELECT fields FROM name [WHERE expr] [ORDER BY items] [LIMIT n] [;] +%% SELECT items FROM name [WHERE expr] [GROUP BY names] [ORDER BY items] +%% [LIMIT n] [;] +%% +%% A select item is a name or an aggregate call, `count(*)` or `sum(price)`, +%% optionally `AS alias`; which functions exist is Efsql.Parser's business. %% %% A token is {Category, Index, Value}, where Index is its place in the %% token list, so an error names its token exactly. Words in the reserved @@ -17,7 +21,8 @@ %% struct. Nonterminals -statement select_stmt fields field_list table where_clause order_clause order_items order_item limit_clause +statement select_stmt fields field_list field table where_clause group_clause names +order_clause order_items order_item order_key aggregate limit_clause expr or_expr and_expr not_expr predicate operand primary name type_name type_word keyword element elements expr_list comparison. @@ -37,14 +42,21 @@ Endsymbol '$end'. statement -> select_stmt : '$1'. statement -> select_stmt ';' : '$1'. -select_stmt -> 'select' fields 'from' table where_clause order_clause limit_clause : - {select, '$2', '$4', '$5', '$6', '$7'}. +select_stmt -> 'select' fields 'from' table where_clause group_clause order_clause limit_clause : + {select, '$2', '$4', '$5', '$6', '$7', '$8'}. fields -> '*' : star. fields -> field_list : '$1'. -field_list -> name : ['$1']. -field_list -> name ',' field_list : ['$1' | '$3']. +field_list -> field : ['$1']. +field_list -> field ',' field_list : ['$1' | '$3']. + +field -> name : '$1'. +field -> aggregate : '$1'. +field -> aggregate 'as' name : setelement(4, '$1', '$3'). + +aggregate -> ident '(' '*' ')' : {aggregate, value('$1'), star, nil}. +aggregate -> ident '(' name ')' : {aggregate, value('$1'), '$3', nil}. table -> name : ['$1']. table -> name '.' name : ['$1', '$3']. @@ -53,15 +65,24 @@ table -> name '.' name '.' name : ['$1', '$3', '$5']. where_clause -> '$empty' : nil. where_clause -> 'where' expr : '$2'. +group_clause -> '$empty' : []. +group_clause -> 'group' 'by' names : '$3'. + +names -> name : ['$1']. +names -> name ',' names : ['$1' | '$3']. + order_clause -> '$empty' : []. order_clause -> 'order' 'by' order_items : '$3'. order_items -> order_item : ['$1']. order_items -> order_item ',' order_items : ['$1' | '$3']. -order_item -> name : {'$1', asc}. -order_item -> name 'asc' : {'$1', asc}. -order_item -> name 'desc' : {'$1', desc}. +order_item -> order_key : {'$1', asc}. +order_item -> order_key 'asc' : {'$1', asc}. +order_item -> order_key 'desc' : {'$1', desc}. + +order_key -> name : '$1'. +order_key -> aggregate : '$1'. limit_clause -> '$empty' : nil. limit_clause -> 'limit' integer : value('$2'). diff --git a/test/aggregate_test.exs b/test/aggregate_test.exs new file mode 100644 index 0000000..1ffb100 --- /dev/null +++ b/test/aggregate_test.exs @@ -0,0 +1,88 @@ +defmodule Efsql.AggregateTest do + use ExUnit.Case, async: true + + alias Efsql.Aggregate + alias Efsql.Exception.Unsupported + + @rows [ + %{city: "Osaka", age: 30, name: "a"}, + %{city: "Lima", age: 40, name: "b"}, + %{city: "Osaka", age: nil, name: "c"}, + %{city: nil, age: 25, name: "d"}, + %{city: "Osaka", age: 50, name: "e"} + ] + + test "one row per group, ordered by key with NULLs last" do + assert Aggregate.run(@rows, [:city], [{:n, :count, :star}]) == [ + %{city: "Lima", n: 1}, + %{city: "Osaka", n: 3}, + %{city: nil, n: 1} + ] + end + + test "count(*) counts rows, every other aggregate skips NULLs" do + assert [_, osaka, _] = + Aggregate.run(@rows, [:city], [ + {:rows, :count, :star}, + {:ages, :count, :age}, + {:sum, :sum, :age}, + {:avg, :avg, :age}, + {:min, :min, :age}, + {:max, :max, :age} + ]) + + assert osaka == %{city: "Osaka", rows: 3, ages: 2, sum: 80, avg: 40.0, min: 30, max: 50} + end + + test "no group fields means one group over every row" do + assert Aggregate.run(@rows, [], [{:n, :count, :star}, {:first, :min, :name}]) == [ + %{n: 5, first: "a"} + ] + end + + test "an aggregate over no rows is still one row; a grouping over none is no rows" do + assert Aggregate.run([], [], [{:n, :count, :star}, {:c, :count, :age}, {:s, :sum, :age}]) == + [%{n: 0, c: 0, s: nil}] + + assert Aggregate.run([], [:city], [{:n, :count, :star}]) == [] + end + + test "sum, min and max of only NULLs are NULL" do + rows = [%{age: nil}, %{}] + + assert Aggregate.run(rows, [], [{:s, :sum, :age}, {:a, :avg, :age}, {:m, :min, :age}]) == + [%{s: nil, a: nil, m: nil}] + end + + test "sum and avg are Decimal when any input is" do + rows = [%{p: Decimal.new("1.50")}, %{p: 2}, %{p: 0.5}] + + assert [%{s: s, a: a}] = Aggregate.run(rows, [], [{:s, :sum, :p}, {:a, :avg, :p}]) + assert Decimal.equal?(s, Decimal.new("4.0")) + assert Decimal.equal?(Decimal.round(a, 3), Decimal.new("1.333")) + end + + test "sum of anything but numbers is unsupported" do + assert_raise Unsupported, ~r/sum and avg need numbers, but name holds "a"/, fn -> + Aggregate.run(@rows, [], [{:s, :sum, :name}]) + end + end + + test "equal values group together whatever their term" do + rows = [ + %{d: Decimal.new("1.0")}, + %{d: Decimal.new("1.00")}, + %{d: ~N[2024-01-01 00:00:00]}, + %{d: ~N[2024-01-01 00:00:00.000000]} + ] + + assert [%{n: 2}, %{n: 2}] = Aggregate.run(rows, [:d], [{:n, :count, :star}]) + end + + test "min and max order datetimes chronologically" do + rows = [%{at: ~D[2024-05-01]}, %{at: ~D[2023-12-31]}, %{at: ~D[2024-01-15]}] + + assert Aggregate.run(rows, [], [{:first, :min, :at}, {:last, :max, :at}]) == + [%{first: ~D[2023-12-31], last: ~D[2024-05-01]}] + end +end diff --git a/test/complete_test.exs b/test/complete_test.exs index 238046c..6d6c64b 100644 --- a/test/complete_test.exs +++ b/test/complete_test.exs @@ -59,6 +59,16 @@ defmodule Efsql.CompleteTest do test "order by offers fields" do assert {_, ["by"]} = Complete.complete("select id from users order ", @context) + assert {_, ["by"]} = Complete.complete("select id from users group ", @context) + + {_, candidates} = Complete.complete("select id from users group by ", @context) + assert "name" in candidates + + assert {_, ["order", "limit"]} = + Complete.complete("select name from users group by name ", @context) + + {_, candidates} = Complete.complete("select id from users ", @context) + assert "group" in candidates {_, candidates} = Complete.complete("select id from users order by ", @context) assert "name" in candidates end diff --git a/test/group_by_test.exs b/test/group_by_test.exs new file mode 100644 index 0000000..5f0e9f8 --- /dev/null +++ b/test/group_by_test.exs @@ -0,0 +1,86 @@ +defmodule EfsqlTest.Integration.GroupBy do + use EfsqlTest.Case, async: true + + alias Efsql.Exception.Unsupported + alias Efsql.Physical.Plan + + # Alice "Lorem ipsum", Bob "foobar", Charles with no notes. + + test "count(*) over the whole table", context do + assert [%{"count(*)": 3}] = Efsql.all("select count(*) from #{context[:tenant_id]}.users;") + end + + test "count of a field skips NULLs", context do + assert [%{n: 2}] = Efsql.all("select count(notes) as n from #{context[:tenant_id]}.users;") + end + + test "min and max", context do + assert [%{"min(name)": "Alice", "max(name)": "Charles"}] = + Efsql.all("select min(name), max(name) from #{context[:tenant_id]}.users;") + end + + test "one row per group, NULLs grouped together", context do + rows = + Efsql.all("select notes, count(*) as n from #{context[:tenant_id]}.users group by notes;") + + assert rows == [ + %{notes: "Lorem ipsum", n: 1}, + %{notes: "foobar", n: 1}, + %{notes: nil, n: 1} + ] + end + + test "WHERE narrows the rows before grouping", context do + assert [%{n: 2}] = + Efsql.all( + "select count(*) as n from #{context[:tenant_id]}.users where name > 'Alice';" + ) + end + + test "an aggregate of no rows is one row", context do + assert [%{n: 0, "max(name)": nil}] = + Efsql.all( + "select count(*) as n, max(name) from #{context[:tenant_id]}.users where name = 'Zed';" + ) + end + + test "ORDER BY and LIMIT apply to the groups", context do + assert [%{notes: nil}, %{notes: "foobar"}] = + Efsql.all( + "select notes from #{context[:tenant_id]}.users group by notes " <> + "order by notes desc limit 2;" + ) + end + + test "fields grouped but not selected are projected away", context do + assert [%{n: 1}, %{n: 1}, %{n: 1}] = + rows = + Efsql.all("select count(*) as n from #{context[:tenant_id]}.users group by notes;") + + assert Enum.all?(rows, &(Map.keys(&1) == [:n])) + end + + test "sum of text is unsupported", context do + assert_raise Unsupported, ~r/sum and avg need numbers/, fn -> + Efsql.all("select sum(name) from #{context[:tenant_id]}.users;") + end + end + + test "the scan keeps its pushdown and reads only the fields it needs", context do + {%Plan{} = plan, _rows, _tenants} = + Efsql.qall( + "select notes, count(*) from #{context[:tenant_id]}.users where name = 'Bob' " <> + "group by notes order by notes limit 5;" + ) + + assert {:index_scan, %Ecto.Query{wheres: [_]} = query, _opts} = plan.access + assert %{take: %{0 => {:any, [:notes]}}} = query.select + assert query.limit == nil + + assert [ + {:aggregate, [:notes], [{:"count(*)", :count, :star}]}, + {:sort, [asc: :notes]}, + {:limit, 5} + ] = plan.ops + end +end diff --git a/test/parser_test.exs b/test/parser_test.exs index 1631015..10db5e9 100644 --- a/test/parser_test.exs +++ b/test/parser_test.exs @@ -327,6 +327,69 @@ defmodule Efsql.ParserTest do end end + describe "group by" do + test "group fields and aggregates become output columns" do + assert %Logical.Select{ + projection: [:notes, :"count(*)", :total], + group_by: [:notes], + aggregates: [{:"count(*)", :count, :star}, {:total, :sum, :price}] + } = parse("select notes, count(*), sum(price) as total from t.users group by notes;") + end + + test "an aggregate without GROUP BY is one group over every row" do + assert %Logical.Select{group_by: [], aggregates: [{:"max(name)", :max, :name}]} = + parse("select max(name) from t.users;") + end + + test "every aggregate function" do + assert %Logical.Select{aggregates: aggregates} = + parse("select count(a), sum(a), min(a), max(a), avg(a) from t.users;") + + assert Enum.map(aggregates, &elem(&1, 1)) == [:count, :sum, :min, :max, :avg] + end + + test "ORDER BY names a group field, an aggregate or its alias" do + assert %Logical.Select{order: [desc: :n, asc: :notes, desc: :"max(name)"]} = + parse( + "select notes, count(*) as n, max(name) from t.users group by notes " <> + "order by n desc, notes, max(name) desc;" + ) + end + + test "an aggregate only in ORDER BY is computed and projected away" do + assert %Logical.Select{ + projection: [:notes], + aggregates: [{:"count(*)", :count, :star}], + order: [desc: :"count(*)"] + } = parse("select notes from t.users group by notes order by count(*) desc;") + end + + test "the same aggregate in ORDER BY is the selected one, alias included" do + assert %Logical.Select{aggregates: [{:n, :count, :star}], order: [desc: :n]} = + parse("select count(*) as n from t.users order by count(*) desc;") + end + + test "WHERE and LIMIT carry over" do + assert %Logical.Select{predicates: [{:cmp, :==, :name, "x"}], limit: 5} = + parse("select count(*) from t.users where name = 'x' limit 5;") + end + + test "grouping errors say what is wrong" do + for {sql, message} <- [ + {"select name, count(*) from t.users", "name must be in GROUP BY"}, + {"select name from t.users group by notes", "name must be in GROUP BY"}, + {"select * from t.users group by name", "SELECT * can't be used with GROUP BY"}, + {"select count(*) from t.users order by name", "ORDER BY name must name"}, + {"select lower(name) from t.users", "lower() is not supported"}, + {"select sum(*) from t.users", "only count takes *"}, + {"select count(_) from t.users", "count('_') is not supported"}, + {"select count(*) from t.users group by _", "GROUP BY '_' is not supported"} + ] do + assert_raise Unsupported, ~r/#{Regex.escape(message)}/, fn -> parse(sql) end + end + end + end + describe "limit" do test "limit" do assert %Logical.Select{limit: 2} = parse("select id from t.users limit 2;") diff --git a/test/sql/parser_test.exs b/test/sql/parser_test.exs index 6f91355..505b666 100644 --- a/test/sql/parser_test.exs +++ b/test/sql/parser_test.exs @@ -95,12 +95,14 @@ defmodule Efsql.SQL.ParserTest do test "unsupported shapes say what is unsupported" do assert reason("select distinct a from t") == "DISTINCT is not supported" - assert reason("select count(a) from t") == "functions are not supported" assert reason("select t.a from t") == "qualified column names are not supported" end test "a missing FROM" do - assert %SyntaxError{reason: "expected FROM or ',', got the end of the statement", column: 9} = + assert %SyntaxError{ + reason: "expected FROM, '(' or ',', got the end of the statement", + column: 9 + } = error("select a") end end @@ -346,6 +348,50 @@ defmodule Efsql.SQL.ParserTest do end end + describe "GROUP BY and aggregates" do + test "aggregates in the select list, with or without an alias" do + assert %Select{ + fields: [ + "a", + {:aggregate, "count", :star, nil}, + {:aggregate, "sum", "price", "total"} + ], + group_by: ["a"] + } = parse("select a, COUNT(*), Sum(price) as total from t group by a") + end + + test "any function name parses; the translator checks it" do + assert %Select{fields: [{:aggregate, "lower", "a", nil}]} = parse("select lower(a) from t") + end + + test "GROUP BY lists names and comes between WHERE and ORDER BY" do + assert %Select{where: {:compare, :=, _, _}, group_by: ["a", "b"], order_by: [{"a", :asc}]} = + parse("select a, b from t where c = 1 group by a, b order by a limit 5") + end + + test "ORDER BY takes an aggregate" do + assert %Select{order_by: [{{:aggregate, "count", :star, nil}, :desc}, {"a", :asc}]} = + parse("select a from t group by a order by count(*) desc, a") + end + + test "errors" do + assert reason("select * from t group a") == "expected BY, got 'a'" + assert reason("select * from t group by") == "expected a name, got the end of the statement" + assert reason("select count() from t") == "expected a name or '*', got ')'" + assert reason("select count(a from t") == "expected ')', got 'from'" + assert reason("select count(*) as from t") =~ "expected a name, got the keyword 'from'" + + assert reason("select a from t order by a group by a") == + "expected the end of the statement, got 'group'" + + assert reason("select a from t where count(*) > 1") == + "functions are only supported as aggregates, in the select list and ORDER BY" + + assert reason("select a from t group by a having count(*) > 1") == + "HAVING is not supported" + end + end + describe "ORDER BY and LIMIT" do test "directions default to ascending" do assert %Select{order_by: [{"a", :asc}, {"b", :desc}, {"c", :asc}]} = @@ -396,8 +442,6 @@ defmodule Efsql.SQL.ParserTest do end test "unsupported clauses are named" do - assert reason("select * from t group by a") == "GROUP BY is not supported" - assert reason("select * from t where a = 1 having a > 1") == "HAVING is not supported" diff --git a/test/support/efsql_test/sql_gen.ex b/test/support/efsql_test/sql_gen.ex index ac9b4a3..96708f9 100644 --- a/test/support/efsql_test/sql_gen.ex +++ b/test/support/efsql_test/sql_gen.ex @@ -3,7 +3,9 @@ defmodule EfsqlTest.SQLGen do # Random `Efsql.SQL.AST.Select` trees, and a printer that renders one as # SQL text with randomized spelling: keyword case, whitespace and # comments between tokens, optional quoting, `!=` for `<>`, `CAST(...)` - # for `::`, `ISNULL` for `IS NULL`. Parsing the text must give back the + # for `::`, `ISNULL` for `IS NULL`. Trees include aggregates and GROUP BY + # whatever their meaning (`select a, lower(*)`), since the parser takes + # any function name. Parsing the text must give back the # tree. Uses `:rand`, which ExUnit seeds per test from `--seed`. alias Efsql.SQL.AST.Select @@ -17,16 +19,27 @@ defmodule EfsqlTest.SQLGen do def select() do %Select{ - fields: if(one_in(4), do: :star, else: list(1..4, &name/0)), + fields: if(one_in(4), do: :star, else: list(1..4, &field/0)), from: list(1..3, &name/0), where: maybe(fn -> expr(3) end), - order_by: list(0..3, fn -> {name(), Enum.random([:asc, :desc])} end), + group_by: if(one_in(3), do: list(1..3, &name/0), else: []), + order_by: list(0..3, fn -> {order_key(), Enum.random([:asc, :desc])} end), limit: maybe(fn -> Enum.random([0, 1, 15, 1000, 123_456_789]) end) } end # -- trees -- + # Function names must lex as plain words that aren't keywords. + @functions ~w[count sum min max avg lower] + + defp field(), do: if(one_in(3), do: aggregate(maybe(&name/0)), else: name()) + defp order_key(), do: if(one_in(4), do: aggregate(nil), else: name()) + + defp aggregate(alias) do + {:aggregate, Enum.random(@functions), if(one_in(3), do: :star, else: name()), alias} + end + def expr(0), do: predicate() def expr(depth) do @@ -90,6 +103,7 @@ defmodule EfsqlTest.SQLGen do kw("from"), Enum.intersperse(Enum.map(select.from, &print_name/1), p(".")), if(select.where, do: [kw("where"), print(select.where)], else: []), + group_by(select.group_by), order_by(select.order_by), if(select.limit, do: [kw("limit"), w(Integer.to_string(select.limit))], else: []), if(one_in(2), do: [p(";")], else: []) @@ -99,7 +113,20 @@ defmodule EfsqlTest.SQLGen do end defp fields(:star), do: [p("*")] - defp fields(names), do: comma(Enum.map(names, &print_name/1)) + defp fields(items), do: comma(Enum.map(items, &print_field/1)) + + defp print_field({:aggregate, _, _, nil} = call), do: print_aggregate(call) + + defp print_field({:aggregate, _, _, alias} = call), + do: [print_aggregate(call), kw("as"), print_name(alias)] + + defp print_field(name), do: print_name(name) + + defp print_aggregate({:aggregate, function, arg, _alias}), + do: [kw(function), p("("), if(arg == :star, do: p("*"), else: print_name(arg)), p(")")] + + defp group_by([]), do: [] + defp group_by(names), do: [kw("group"), kw("by"), comma(Enum.map(names, &print_name/1))] defp order_by([]), do: [] @@ -109,8 +136,8 @@ defmodule EfsqlTest.SQLGen do kw("by"), comma( Enum.map(items, fn - {name, :asc} -> [print_name(name) | Enum.random([[], [kw("asc")]])] - {name, :desc} -> [print_name(name), kw("desc")] + {key, :asc} -> [print_field(key) | Enum.random([[], [kw("asc")]])] + {key, :desc} -> [print_field(key), kw("desc")] end) ) ]