From ae8719e6554552876728f6781e640b5fcb215055 Mon Sep 17 00:00:00 2001 From: Niols Date: Tue, 15 Sep 2026 17:21:00 +0100 Subject: [PATCH 1/2] Add support for PostgreSQL's CREATE/DROP EXTENSION Both statements are accepted with -dialect postgresql, so that migration files that install pg_trgm and friends can be fed to sqlgg unchanged. sqlgg keeps no extension state, so there is nothing to track: the optional SCHEMA, VERSION and CASCADE clauses are parsed and discarded, an extension can be dropped without having been created, and the statements evaluate to Stmt.Other rather than getting a statement kind of their own. They are gated behind a new `extension` dialect feature, PostgreSQL only, like the existing `user_defined_type` gate on CREATE TYPE. `extension`, `schema` and `version` become keywords, so they are added to `ident` to keep them unreserved, the way `type` already is. Being unreserved, they also join the filter in lsp/completion.ml that already excluded `type`: the recovering parser accepts an unreserved keyword wherever an identifier fits, so without this they would be offered as keyword completions in every completion list. CREATE EXTENSION's trailing `[ WITH ] ...` clause is ambiguous with a following statement that starts with a WITH (a CTE), which LR(1) cannot resolve. Both statements therefore live in a new `top_statement` rule used only by `input`, where EOF bounds the clause. Extensions are database-global DDL that has no business in a routine body anyway. Co-Authored-By: Claude Opus 5 (1M context) --- doc/_cli-options.md | 2 +- doc/sql/ddl.md | 15 +++++++ lib/dialect.ml | 5 +++ lib/sql.ml | 2 + lib/sql_lexer.mll | 3 ++ lib/sql_parser.mly | 23 +++++++++- lib/syntax.ml | 3 ++ lsp/completion.ml | 12 +++++- lsp/document.ml | 3 +- test/cram/create_extension.t | 82 ++++++++++++++++++++++++++++++++++++ 10 files changed, 144 insertions(+), 6 deletions(-) create mode 100644 test/cram/create_extension.t diff --git a/doc/_cli-options.md b/doc/_cli-options.md index cce123dc..760f5534 100644 --- a/doc/_cli-options.md +++ b/doc/_cli-options.md @@ -22,7 +22,7 @@ Dialect and checks: -dialect mysql|postgresql|sqlite|tidb Set SQL dialect. Queries can only use its features - -no-check {all|{,}+} Disable dialect feature checks (possible features: collation|join_on_subquery|create_table_as_select|on_duplicate_key|on_conflict|straight_join|lock_in_share_mode|fulltext_index|unsigned_types|autoincrement|replace_into|row_locking|default_expr|ttl|cached_table|alter_column|user_defined_type) + -no-check {all|{,}+} Disable dialect feature checks (possible features: collation|join_on_subquery|create_table_as_select|on_duplicate_key|on_conflict|straight_join|lock_in_share_mode|fulltext_index|unsigned_types|autoincrement|replace_into|row_locking|default_expr|ttl|cached_table|alter_column|user_defined_type|extension) -allow-write-notnull-null Accept writing a nullable value into a NOT NULL column, instead of failing (MySQL, TiDB and SQLite only) Generated header: diff --git a/doc/sql/ddl.md b/doc/sql/ddl.md index 5b9e0597..e0e48c61 100644 --- a/doc/sql/ddl.md +++ b/doc/sql/ddl.md @@ -104,6 +104,21 @@ DROP TYPE mood; Enum types get the same treatment as inline `ENUM(...)` columns. See [Literals](./literals.md) for enum literal validation and the OCaml mapping to polymorphic variants. +## CREATE EXTENSION (PostgreSQL) + +`CREATE EXTENSION` and `DROP EXTENSION` are accepted with `-dialect postgresql` so that +migration files can be fed to sqlgg unchanged: + +```sql +CREATE EXTENSION IF NOT EXISTS pg_trgm WITH SCHEMA public; +DROP EXTENSION pg_trgm; +``` + +sqlgg keeps no extension state, so these statements are only checked for syntax: an +extension can be dropped without having been created, and the optional `SCHEMA`, +`VERSION` and `CASCADE` clauses are parsed and discarded. They are reported as plain +statements with no particular kind. + ## Metadata Metadata (`-- [sqlgg] key=value`) can be attached to columns and propagates through DDL/DML/DQL. See [Metadata](./metadata.md) for details. diff --git a/lib/dialect.ml b/lib/dialect.ml index 70ae44ac..86cfc374 100644 --- a/lib/dialect.ml +++ b/lib/dialect.ml @@ -33,6 +33,7 @@ type feature = | CachedTable [@as "cached_table"] | AlterColumn [@as "alter_column"] | UserDefinedType [@as "user_defined_type"] + | Extension [@as "extension"] [@@deriving show { with_path = false }, enumerate, to_string, of_string] let show_feature x = @@ -153,6 +154,8 @@ let get_alter_column (change : Sql.Alter_column_pg.t) pos = let get_user_defined_type pos = only UserDefinedType [PostgreSQL] pos +let get_extension pos = only Extension [PostgreSQL] pos + let get_default_expr ~kind ~expr pos = let open Sql in let tidb_only_functions = @@ -480,3 +483,5 @@ let rec analyze stmt = process_params acc params | CreateType _ -> [get_user_defined_type (0, 0)] | DropType _ -> [get_user_defined_type (0, 0)] + | CreateExtension _ -> [get_extension (0, 0)] + | DropExtension _ -> [get_extension (0, 0)] diff --git a/lib/sql.ml b/lib/sql.ml index e0e62b48..a6c01165 100644 --- a/lib/sql.ml +++ b/lib/sql.ml @@ -1099,6 +1099,8 @@ type stmt = | CreateRoutine of table_name * Source_type.kind collated located option * (string * Source_type.kind collated located * expr option) list (* table_name represents possibly namespaced function name *) | CreateType of string * create_type_target | DropType of string * bool + | CreateExtension of string + | DropExtension of string list [@@deriving show {with_path=false}] (* diff --git a/lib/sql_lexer.mll b/lib/sql_lexer.mll index 532e4317..faa005a6 100644 --- a/lib/sql_lexer.mll +++ b/lib/sql_lexer.mll @@ -69,6 +69,7 @@ let keywords = "escape",ESCAPE; "except",EXCEPT; "exists",EXISTS; + "extension", EXTENSION "extension"; "extract",EXTRACT; "false", FALSE; "first",FIRST; @@ -142,6 +143,7 @@ let keywords = "returns", RETURNS; "row", ROW; "rows", ROWS; + "schema", SCHEMA "schema"; "second_microsecond", SECOND_MICROSECOND; "select",SELECT; "set",SET; @@ -168,6 +170,7 @@ let keywords = "using",USING; "values",VALUES; "varying",VARYING; + "version", VERSION "version"; "view",VIEW; "when", WHEN; "where",WHERE; diff --git a/lib/sql_parser.mly b/lib/sql_parser.mly index 74fae590..b0542f8e 100644 --- a/lib/sql_parser.mly +++ b/lib/sql_parser.mly @@ -48,6 +48,7 @@ SHARED EXCLUSIVE NONE TTL TTL_ENABLE REMOVE CACHE NOCACHE %token FUNCTION PROCEDURE LANGUAGE RETURNS OUT INOUT BEGIN COMMENT +%token EXTENSION SCHEMA VERSION %token SECOND_MICROSECOND MINUTE_MICROSECOND MINUTE_SECOND HOUR_MICROSECOND HOUR_SECOND HOUR_MINUTE DAY_MICROSECOND DAY_SECOND DAY_MINUTE DAY_HOUR EXTRACT @@ -99,7 +100,17 @@ %% -input: statement EOF { $1 } +input: top_statement EOF { $1 } + +(* CREATE/DROP EXTENSION are database-global PostgreSQL DDL, so they are only + accepted as a whole statement and not inside a routine body. Keeping them out + of [statement] also keeps CREATE EXTENSION's trailing [ WITH ] ... clause from + being ambiguous with a following statement that starts with a WITH (CTE). *) +top_statement: s=statement { s } + | CREATE EXTENSION if_not_exists? name=ident create_extension_opts + { CreateExtension name } + | DROP EXTENSION if_exists? names=commas(ident) drop_behavior? + { DropExtension names } param: | QSTN { { value=None; pos = ($startofs, $endofs) } } @@ -213,6 +224,14 @@ proc_parameter: parameter_mode? p=func_parameter { p } or_replace: OR REPLACE { } +(* CREATE EXTENSION options: sqlgg tracks no extension state, so these are + accepted and discarded. Order is left loose on purpose. *) +create_extension_opts: WITH? create_extension_opt* { } +create_extension_opt: SCHEMA ident { } + | VERSION extension_version { } + | CASCADE { } +extension_version: TEXT { } | ident { } + routine_body: TEXT | compound_stmt { } compound_stmt: BEGIN statement+ END { } (* mysql *) @@ -221,7 +240,7 @@ routine_extra: LANGUAGE IDENT { } (* cf. ColId / unreserved_keyword in PostgreSQL's gram.y (TYPE_P is unreserved there too): https://github.com/postgres/postgres/blob/REL_18_0/src/backend/parser/gram.y#L17632 *) -ident: x=IDENT | x=TYPE { x } +ident: x=IDENT | x=TYPE | x=EXTENSION | x=SCHEMA | x=VERSION { x } table_ident: x=ident { x } qual_ident: x=ident { x } diff --git a/lib/syntax.ml b/lib/syntax.ml index f50fba71..f7c40677 100644 --- a/lib/syntax.ml +++ b/lib/syntax.ml @@ -2082,6 +2082,9 @@ let rec eval (stmt:Sql.stmt) = | DropType (name, if_exists) -> User_types.drop ~if_exists name; [], [], DropType name, no_stmt_annotations + | CreateExtension _ | DropExtension _ -> + (* sqlgg keeps no extension state: accepted, and invisible to codegen *) + [], [], Other, no_stmt_annotations type var_shape = | Shape_param diff --git a/lsp/completion.ml b/lsp/completion.ml index 6ad4af5c..6cb91477 100644 --- a/lsp/completion.ml +++ b/lsp/completion.ml @@ -232,11 +232,19 @@ let make (document : Document.t) offset = ~some:(column_items ~rank:Rank.exact) (Symbol.find_opt sources q) | Name roles -> - let is_type = function Sql_tokens.TYPE _ -> true | _ -> false in + (* Tokens that [ident] accepts too (see the ident rule in sql_parser.mly): + the recovering parser takes them wherever an identifier fits, so + offering them as keywords is noise in every completion list. Keep in + step with [ident]. *) + let is_unreserved = function + | Sql_tokens.TYPE _ | Sql_tokens.EXTENSION _ | Sql_tokens.SCHEMA _ + | Sql_tokens.VERSION _ -> true + | _ -> false + in let keywords = Sql_lexer.Keywords.to_seq Sql_lexer.keywords |> Seq.filter (fun (_, token) -> - not (is_type token) && Recover_parser.accepts run token) + not (is_unreserved token) && Recover_parser.accepts run token) |> Seq.map (fun (keyword, _) -> if List.exists (String.equal keyword) functions then function_item keyword else diff --git a/lsp/document.ml b/lsp/document.ml index 83a8d66f..1b171bbf 100644 --- a/lsp/document.ml +++ b/lsp/document.ml @@ -142,7 +142,8 @@ let check ~file (stmt : Statements.t) = | Sql.DeleteMulti (_, tables, _) -> scope (Some tables) | Sql.Insert { action = (`Set _ | `Values _ | `Param _); _ } | Sql.Create _ | Sql.Drop _ | Sql.Alter _ | Sql.Rename _ | Sql.CreateIndex _ | Sql.Set _ - | Sql.CreateRoutine _ | Sql.CreateType _ | Sql.DropType _ -> [], [], [] + | Sql.CreateRoutine _ | Sql.CreateType _ | Sql.DropType _ + | Sql.CreateExtension _ | Sql.DropExtension _ -> [], [], [] in match exn with | Parser_utils.Error _ -> recovery_scope stmt.text diff --git a/test/cram/create_extension.t b/test/cram/create_extension.t new file mode 100644 index 00000000..15519d39 --- /dev/null +++ b/test/cram/create_extension.t @@ -0,0 +1,82 @@ +CREATE EXTENSION, all optional clauses + $ sqlgg -gen none -dialect=postgresql - <<'EOF' 2>&1 + > CREATE EXTENSION pg_trgm; + > CREATE EXTENSION IF NOT EXISTS pg_trgm; + > CREATE EXTENSION IF NOT EXISTS pg_trgm WITH SCHEMA public; + > CREATE EXTENSION pg_trgm SCHEMA public; + > CREATE EXTENSION pg_trgm WITH SCHEMA public VERSION '1.6' CASCADE; + > CREATE EXTENSION pg_trgm WITH VERSION unquoted_version; + > EOF + +DROP EXTENSION, all optional clauses + $ sqlgg -gen none -dialect=postgresql - <<'EOF' 2>&1 + > DROP EXTENSION pg_trgm; + > DROP EXTENSION IF EXISTS pg_trgm; + > DROP EXTENSION pg_trgm CASCADE; + > DROP EXTENSION pg_trgm RESTRICT; + > DROP EXTENSION IF EXISTS pg_trgm, btree_gin CASCADE; + > EOF + +sqlgg keeps no extension state, so an extension needs no prior CREATE to be +dropped, and may be created twice + + $ sqlgg -gen none -dialect=postgresql - <<'EOF' 2>&1 + > DROP EXTENSION pg_trgm; + > CREATE EXTENSION pg_trgm; + > CREATE EXTENSION pg_trgm; + > EOF + +Both are emitted as plain unprepared statements + + $ sqlgg -gen caml -no-header -dialect=postgresql - <<'EOF' 2>&1 + > CREATE EXTENSION IF NOT EXISTS pg_trgm WITH SCHEMA public; + > DROP EXTENSION pg_trgm; + > EOF + module Sqlgg (T : Sqlgg_traits.M) = struct + + module IO = Sqlgg_io.Blocking + + let statement_0 db = + T.execute_unprepared db (Sqlgg_traits.Query.make ~sql:("CREATE EXTENSION IF NOT EXISTS pg_trgm WITH SCHEMA public") ~name:"statement_0" ~kind:Sqlgg_traits.Query.Other ()) + + let statement_1 db = + T.execute_unprepared db (Sqlgg_traits.Query.make ~sql:("DROP EXTENSION pg_trgm") ~name:"statement_1" ~kind:Sqlgg_traits.Query.Other ()) + + end (* module Sqlgg *) + +A following CTE statement is not mistaken for CREATE EXTENSION's WITH clause + + $ sqlgg -gen none -dialect=postgresql - <<'EOF' 2>&1 + > CREATE TABLE t (id INTEGER NOT NULL); + > CREATE EXTENSION pg_trgm; + > WITH c AS (SELECT id FROM t) SELECT id FROM c; + > EOF + +Extensions are PostgreSQL-only + + $ sqlgg -gen none -dialect=mysql - <<'EOF' 2>&1 + > CREATE EXTENSION pg_trgm; + > EOF + Feature Extension is not supported for dialect MySQL (supported by: PostgreSQL) at + Errors encountered, no code generated + [1] + + $ sqlgg -gen none -dialect=mysql - <<'EOF' 2>&1 + > DROP EXTENSION pg_trgm; + > EOF + Feature Extension is not supported for dialect MySQL (supported by: PostgreSQL) at + Errors encountered, no code generated + [1] + + $ sqlgg -gen none -dialect=mysql -no-check extension - <<'EOF' 2>&1 + > CREATE EXTENSION pg_trgm; + > EOF + Warning: Feature Extension is not supported for dialect MySQL, proceeding anyway at + +extension, schema and version stay unreserved: still usable as identifiers + + $ sqlgg -gen none -dialect=postgresql - <<'EOF' 2>&1 + > CREATE TABLE kw (extension TEXT, schema TEXT, version INTEGER NOT NULL); + > SELECT extension, schema, version FROM kw WHERE schema = 'public'; + > CREATE EXTENSION schema; + > EOF From 246b7676378260557887c583809d569efee55395 Mon Sep 17 00:00:00 2001 From: jongleb <10594395+jongleb@users.noreply.github.com> Date: Tue, 22 Sep 2026 12:36:23 +0000 Subject: [PATCH 2/2] allow extension keywords as names --- lib/sql_parser.mly | 10 +++++----- lsp/completion.ml | 12 ++---------- lsp/recover_parser.ml | 3 ++- lsp/test/unreserved.t/q.sql | 2 ++ lsp/test/unreserved.t/run.t | 33 +++++++++++++++++++++++++++++++++ test/cram/create_extension.t | 10 ++++++++++ 6 files changed, 54 insertions(+), 16 deletions(-) create mode 100644 lsp/test/unreserved.t/q.sql create mode 100644 lsp/test/unreserved.t/run.t diff --git a/lib/sql_parser.mly b/lib/sql_parser.mly index b0542f8e..49e52289 100644 --- a/lib/sql_parser.mly +++ b/lib/sql_parser.mly @@ -235,7 +235,7 @@ extension_version: TEXT { } | ident { } routine_body: TEXT | compound_stmt { } compound_stmt: BEGIN statement+ END { } (* mysql *) -routine_extra: LANGUAGE IDENT { } +routine_extra: LANGUAGE ident { } | COMMENT TEXT { } (* cf. ColId / unreserved_keyword in PostgreSQL's gram.y (TYPE_P is unreserved there too): @@ -502,7 +502,7 @@ column_def_extra: PRIMARY? KEY { Some (Alter_action_attr.Syntax_constraint Prima } | on_conflict { None } | CHECK LPAREN expr RPAREN { None } - | COLLATE IDENT { None } + | COLLATE ident { None } | pair(GENERATED,ALWAYS)? AS LPAREN expr RPAREN either(VIRTUAL,STORED)? { None } (* FIXME params and typing ignored *) default_value: e=single_literal_value @@ -609,7 +609,7 @@ c_expr_: | f=INTERVAL_UNIT LPAREN e=expr RPAREN { call f [e] } | EXTRACT LPAREN interval_unit FROM e=expr RPAREN { call "extract" [e] } | DEFAULT LPAREN a=attr_name RPAREN { fn "default" fun_identity [Column (make_collated ~collated:a ())] } - | CONVERT LPAREN e=expr USING IDENT RPAREN { e } + | CONVERT LPAREN e=expr USING ident RPAREN { e } | CONVERT LPAREN e=expr COMMA f=cast_as RPAREN { f e } | GROUP_CONCAT LPAREN p=func_params order=loption(order) preceded(SEPARATOR, TEXT)? RPAREN { fn "group_concat" (Agg (With_order { with_order_kind = Group_concat; order })) p } @@ -805,11 +805,11 @@ cast_as: %inline sequence(X): l=sequence_(X) RPAREN { l } %inline charset_kw: CHARSET {} | CHARACTER SET {} -charset: charset_kw c=IDENT { Named c } +charset: charset_kw c=ident { Named c } | charset_kw BINARY { Binary } | charset_kw? ASCII { Ascii } | charset_kw? UNICODE { Unicode } -collate: COLLATE c=IDENT { make_located ~value:c ~pos:($startofs, $endofs) } +collate: COLLATE c=ident { make_located ~value:c ~pos:($startofs, $endofs) } collate_opt: %prec LOWEST { None } | c=collate { Some c } sql_type: t=sql_type_flavor c=collate_opt { make_collated ?collation:c ~collated:t () } diff --git a/lsp/completion.ml b/lsp/completion.ml index 6cb91477..d6d9c76f 100644 --- a/lsp/completion.ml +++ b/lsp/completion.ml @@ -232,19 +232,11 @@ let make (document : Document.t) offset = ~some:(column_items ~rank:Rank.exact) (Symbol.find_opt sources q) | Name roles -> - (* Tokens that [ident] accepts too (see the ident rule in sql_parser.mly): - the recovering parser takes them wherever an identifier fits, so - offering them as keywords is noise in every completion list. Keep in - step with [ident]. *) - let is_unreserved = function - | Sql_tokens.TYPE _ | Sql_tokens.EXTENSION _ | Sql_tokens.SCHEMA _ - | Sql_tokens.VERSION _ -> true - | _ -> false - in let keywords = Sql_lexer.Keywords.to_seq Sql_lexer.keywords |> Seq.filter (fun (_, token) -> - not (is_unreserved token) && Recover_parser.accepts run token) + Option.is_none (Recover_parser.ident_name token) + && Recover_parser.accepts run token) |> Seq.map (fun (keyword, _) -> if List.exists (String.equal keyword) functions then function_item keyword else diff --git a/lsp/recover_parser.ml b/lsp/recover_parser.ml index 598660e9..e21976e0 100644 --- a/lsp/recover_parser.ml +++ b/lsp/recover_parser.ml @@ -23,7 +23,8 @@ let make_lexeme lexbuf token = { token; pos = Sql_lexer.pos lexbuf } let position offset = { Lexing.dummy_pos with pos_cnum = offset } let ident_name : Sql_tokens.token -> string option = function - | IDENT name | TYPE name -> Some name + | IDENT name | TYPE name | EXTENSION name | SCHEMA name | VERSION name -> + Some name | _ -> None let qualifier_before = function diff --git a/lsp/test/unreserved.t/q.sql b/lsp/test/unreserved.t/q.sql new file mode 100644 index 00000000..e9594bb4 --- /dev/null +++ b/lsp/test/unreserved.t/q.sql @@ -0,0 +1,2 @@ +CREATE TABLE schema (extension INTEGER, version TEXT); +SELECT extension, version FROM schema; diff --git a/lsp/test/unreserved.t/run.t b/lsp/test/unreserved.t/run.t new file mode 100644 index 00000000..3575b5b4 --- /dev/null +++ b/lsp/test/unreserved.t/run.t @@ -0,0 +1,33 @@ +PostgreSQL unreserved keywords behave as identifiers in IDE features: + + $ ../ask.exe q.sql hover:'SELECT extension^' hover:'extension, version^' hover:'FROM schema^' def:'SELECT extension^' def:'extension, version^' def:'FROM schema^' + ### hover:SELECT extension^ + 2:7-2:16 + ```sql + schema.extension Int? + ``` + + Declared in `q.sql` + ### hover:extension, version^ + 2:18-2:25 + ```sql + schema.version Text? + ``` + + Declared in `q.sql` + ### hover:FROM schema^ + 2:31-2:37 + **table** `schema` + + ```sql + extension Int? + version Text? + ``` + + Declared in `q.sql` + ### def:SELECT extension^ + q.sql 1:21-1:30 + ### def:extension, version^ + q.sql 1:40-1:47 + ### def:FROM schema^ + q.sql 1:13-1:19 diff --git a/test/cram/create_extension.t b/test/cram/create_extension.t index 15519d39..24cb74f0 100644 --- a/test/cram/create_extension.t +++ b/test/cram/create_extension.t @@ -80,3 +80,13 @@ extension, schema and version stay unreserved: still usable as identifiers > SELECT extension, schema, version FROM kw WHERE schema = 'public'; > CREATE EXTENSION schema; > EOF + +They also stay usable in identifier positions that predate extension support + + $ sqlgg -gen none -dialect=postgresql -no-check=all - <<'EOF' 2>&1 + > CREATE TABLE kw_contexts (col TEXT COLLATE schema); + > CREATE TABLE kw_charset (col TEXT CHARACTER SET extension); + > CREATE FUNCTION extension(arg INTEGER) RETURNS INTEGER AS 'body' LANGUAGE version; + > SELECT CONVERT('x' USING extension); + > EOF + Warning: Assuming custom collation implementation for PostgreSQL